From 60721667beee603d12e84109d7a73c8a036c9d8f Mon Sep 17 00:00:00 2001
From: deadmanoz <62584182+deadmanoz@users.noreply.github.com>
Date: Thu, 24 Sep 2026 21:47:53 +0800
Subject: [PATCH] feat(site): generate a static website from the dataset
Give every record a page at block/{hash}/ so a block can be linked and found by searching its hash, as asked in #6.
The index lists the records, sortable by column and filterable by text, rule or kind of evidence; notes/ renders docs/notes.md and reported/ lists the reported blocks.
Pages follow the visitor's light or dark preference, with a toggle in the nav.
Block pages take the rule's Core check and CI evidence from the tables in docs/schema.md, say whether the evidence on file is a full block, a P2SH spend proof, a coinbase proof or the header alone, and quote the incident note that names the block's height, or names a block with the same failing_prevout.
ci/generate-website.py writes site/ using pinned Python-Markdown, with ci/website.css and ci/website.js alongside.
The sanity-check workflow builds the site on every run; on main, a deploy job that needs the validation job publishes it to GitHub Pages.
---
.github/workflows/sanitycheck.yml | 29 +-
.gitignore | 1 +
README.md | 6 +
ci/generate-website.py | 496 ++++++++++++++++++++++++++++++
ci/test_generate_website.py | 71 +++++
ci/website.css | 107 +++++++
ci/website.js | 62 ++++
requirements.txt | 1 +
8 files changed, 772 insertions(+), 1 deletion(-)
create mode 100644 ci/generate-website.py
create mode 100644 ci/test_generate_website.py
create mode 100644 ci/website.css
create mode 100644 ci/website.js
diff --git a/.github/workflows/sanitycheck.yml b/.github/workflows/sanitycheck.yml
index 417cbf0..04cd7bf 100644
--- a/.github/workflows/sanitycheck.yml
+++ b/.github/workflows/sanitycheck.yml
@@ -13,6 +13,12 @@ on:
permissions:
contents: read
+# Runs on a branch go one at a time, so an older run on main never deploys over a newer one.
+# A newer push to a pull request cancels its superseded run.
+concurrency:
+ group: ${{ github.workflow }}-${{ github.ref }}
+ cancel-in-progress: ${{ github.event_name == 'pull_request' }}
+
jobs:
sanity-check:
runs-on: ubuntu-latest
@@ -29,13 +35,14 @@ jobs:
path: .cache/prevouts
key: prevouts-v2-${{ hashFiles('blocks/*.bin', 'proofs/*.json') }}
restore-keys: prevouts-v2-
- - name: validate data and test validator
+ - name: validate data, test validator and generate website
run: |
python -m venv .venv
. .venv/bin/activate
python -m pip install -r requirements.txt
python ci/sanity-check.py --fetch-prevouts
python -m unittest discover -s ci -p 'test_*.py'
+ python ci/generate-website.py
# Save once per block set, after validation and tests pass on the default branch.
- name: save verified previous transactions
if: >-
@@ -46,3 +53,23 @@ jobs:
with:
path: .cache/prevouts
key: ${{ steps.prevouts-cache.outputs.cache-primary-key }}
+ # Every validated run uploads the site; only the deploy job publishes it.
+ - name: upload website
+ uses: actions/upload-pages-artifact@v5
+ with:
+ path: site
+
+ deploy-website:
+ needs: sanity-check
+ if: github.ref == 'refs/heads/main'
+ runs-on: ubuntu-latest
+ permissions:
+ pages: write
+ id-token: write
+ environment:
+ name: github-pages
+ url: ${{ steps.deployment.outputs.page_url }}
+
+ steps:
+ - id: deployment
+ uses: actions/deploy-pages@v5
diff --git a/.gitignore b/.gitignore
index e8f5bc7..1d670a7 100644
--- a/.gitignore
+++ b/.gitignore
@@ -2,3 +2,4 @@
.venv/
__pycache__/
.cache/prevouts/
+site/
diff --git a/README.md b/README.md
index eb3466c..bccfbfa 100644
--- a/README.md
+++ b/README.md
@@ -18,6 +18,12 @@ This dataset covers blocks that fail those rules, including failures that can be
Merge-mined recoveries generally provide a header and coinbase rather than a full Bitcoin block.
+## Website
+
+[bitcoin-data.github.io/invalid-blocks](https://bitcoin-data.github.io/invalid-blocks/) is generated from this repository and deployed once validation passes on `main`.
+It has a page for each block at `block/{hash}/`, the rendered [notes](docs/notes.md) and the reported blocks.
+`python ci/generate-website.py` writes it to `site/`; CI runs the same command on pull requests, so a change that breaks the site fails before merge.
+
## Contributing
Add one record to [`data/invalid-blocks.jsonl`](data/invalid-blocks.jsonl), sorted by height then hash.
diff --git a/ci/generate-website.py b/ci/generate-website.py
new file mode 100644
index 0000000..5f5ba2b
--- /dev/null
+++ b/ci/generate-website.py
@@ -0,0 +1,496 @@
+#!/usr/bin/env python3
+"""Generate the static website in site/ from the dataset, its notes and its schema.
+
+Each record gets a page at block/{hash}/, so a search for a block hash can find it.
+The index lists every record; notes/ renders docs/notes.md and reported/ lists the reported blocks.
+"""
+
+import html
+import json
+import posixpath
+import re
+import shutil
+from collections import Counter
+from datetime import datetime, timezone
+from itertools import dropwhile, takewhile
+from pathlib import Path
+from typing import Any
+from urllib.parse import urlparse
+
+import markdown
+from markdown.extensions.toc import TocExtension
+from bitcoin.core import CBlockHeader, CTransaction, b2lx
+
+REPO_ROOT = Path(__file__).resolve().parent.parent
+DATA_PATH = REPO_ROOT / "data" / "invalid-blocks.jsonl"
+REPORTED_PATH = REPO_ROOT / "data" / "reported-blocks.jsonl"
+NOTES_PATH = REPO_ROOT / "docs" / "notes.md"
+SCHEMA_PATH = REPO_ROOT / "docs" / "schema.md"
+BLOCKS_DIR = REPO_ROOT / "blocks"
+PROOFS_DIR = REPO_ROOT / "proofs"
+CSS_PATH = Path(__file__).with_name("website.css")
+JS_PATH = Path(__file__).with_name("website.js")
+OUT_DIR = REPO_ROOT / "site"
+
+SITE_URL = "https://bitcoin-data.github.io/invalid-blocks"
+REPO_URL = "https://github.com/bitcoin-data/invalid-blocks"
+BLOB_URL = f"{REPO_URL}/blob/main"
+RAW_URL = f"{REPO_URL}/raw/main"
+BLOCK_ROOT = "../../" # block pages sit at block/{hash}/
+
+RELATIVE_HREF = re.compile(r'href="(?!https?://|mailto:|#)([^"]+)"')
+BLOCK_FILE = re.compile(r"\.\./blocks/\d+-([0-9a-f]{64})\.bin")
+HEADING = re.compile(r'
"
+
+
+def schema_rows(schema: str, heading: str) -> dict[str, list[str]]:
+ """Read the first Markdown table under heading, keyed by its first cell."""
+ lines = schema.split(heading, 1)[1].splitlines()
+ table = list(takewhile(lambda line: line.startswith("|"), dropwhile(lambda line: not line.startswith("|"), lines)))
+ cells = [[cell.strip() for cell in line.strip("|").split("|")] for line in table[2:]]
+ return {row[0].strip("`"): row for row in cells}
+
+
+def rule_text(schema: str) -> dict[str, tuple[str, str]]:
+ """Map each rule to its Core check and CI evidence from the schema tables, rendered for a block page."""
+ checks = schema_rows(schema, "## Rules and Core reject strings")
+ evidence = schema_rows(schema, "## Evidence enforced by CI")
+
+ def render(cell: str) -> str:
+ inline = markdown.markdown(cell).removeprefix("
").removesuffix("
")
+ return rewrite_links(inline, BLOCK_ROOT)
+
+ return {rule: (render(checks[rule][2]), render(evidence[rule][1])) for rule in checks}
+
+
+def rewrite_links(body: str, root: str) -> str:
+ """Resolve relative hrefs rendered from docs/*.md for a page at root: block files to block pages, notes to notes/, the rest to GitHub."""
+
+ def resolve(match: re.Match[str]) -> str:
+ path, _, anchor = match.group(1).partition("#")
+ fragment = f"#{anchor}" if anchor else ""
+ block = BLOCK_FILE.fullmatch(path)
+ if block:
+ return f'href="{block_href(root, block.group(1))}"'
+ if path == "notes.md":
+ return f'href="{root}notes/{fragment}"'
+ return f'href="{BLOB_URL}/{posixpath.normpath(posixpath.join("docs", path))}{fragment}"'
+
+ return RELATIVE_HREF.sub(resolve, body)
+
+
+def render_notes(notes: str) -> tuple[str, str, list[dict[str, Any]]]:
+ """Render docs/notes.md without its title, with its contents list and its incident sections' heights and block-page excerpts."""
+ converter = markdown.Markdown(extensions=["tables", TocExtension(toc_depth="2-3")])
+ body = converter.convert(notes)
+ body = re.sub(r"
]*>.*?
\s*", "", body, count=1)
+ body = body.replace("
", '
').replace("
", "
")
+ incidents = next((t for t in converter.toc_tokens if t["name"] == "Incident notes"), None)
+ if incidents is None:
+ raise ValueError(f"{NOTES_PATH} has no Incident notes section")
+ sections = []
+ for token in incidents["children"]:
+ paragraph = re.search(rf'
.*?
\s*
(.*?)
', body, re.S)
+ sections.append({
+ "id": token["id"],
+ "html": token["html"],
+ "heights": [int(h) for h in re.findall(r"\d+", token["name"].split(" - ")[0])],
+ "excerpt": rewrite_links(paragraph.group(1), BLOCK_ROOT) if paragraph else "",
+ })
+ return body, converter.toc, sections
+
+
+def link_heading_heights(body: str, root: str, records: list[dict[str, Any]]) -> str:
+ """Link the heights in each incident heading to their block pages, where a height names a single record."""
+ heights = Counter(r["height"] for r in records)
+ single = {r["height"]: r["hash"] for r in records if heights[r["height"]] == 1}
+
+ def link(height: re.Match[str]) -> str:
+ block_hash = single.get(int(height.group(0)))
+ return f'{height.group(0)}' if block_hash else height.group(0)
+
+ def heading(match: re.Match[str]) -> str:
+ heights, dash, title = match.group(2).partition(" - ")
+ return f'
{re.sub(r"\d+", link, heights)}{dash}{title}
'
+
+ return HEADING.sub(heading, body)
+
+
+def note_links(records: list[dict[str, Any]], incidents: list[dict[str, Any]]) -> dict[str, dict[str, Any]]:
+ """Map each record hash to the incident note naming its height, or naming a block with the same failing_prevout."""
+ by_height = {height: incident for incident in incidents for height in incident["heights"]}
+ links = {r["hash"]: by_height[r["height"]] for r in records if r["height"] in by_height}
+ spends = {r["hash"]: r.get("context", {}).get("failing_prevout") for r in records}
+ by_spend = {spends[block_hash]: incident for block_hash, incident in links.items() if spends[block_hash]}
+ return {block_hash: by_spend[spend] for block_hash, spend in spends.items() if spend in by_spend} | links
+
+
+# Evidence kinds, in index order, with their index card labels; a kind's badge reads as the name with spaces.
+EVIDENCE_CARDS = {
+ "full-block": "with full block",
+ "spend-proof": "with P2SH spend proof",
+ "coinbase-proof": "with coinbase proof",
+ "header-only": "header only",
+}
+
+
+def stem(record: dict[str, Any]) -> str:
+ return f"{record['height']}-{record['hash']}"
+
+
+def evidence_on_file(record: dict[str, Any]) -> tuple[str, dict[str, Any] | None]:
+ """Name the kind of evidence on file for a record, with its proof file when it has one."""
+ if (BLOCKS_DIR / f"{stem(record)}.bin").exists():
+ return "full-block", None
+ path = PROOFS_DIR / f"{stem(record)}.json"
+ if not path.exists():
+ return "header-only", None
+ proof = json.loads(path.read_text())
+ coinbase = CTransaction.deserialize(bytes.fromhex(proof["transaction"])).is_coinbase()
+ return ("coinbase-proof" if coinbase else "spend-proof"), proof
+
+
+def badge(kind: str, href: str = "", label: str = "") -> str:
+ """Badge styled as kind; its text is label, or the kind with spaces for hyphens."""
+ text = esc(label) if label else kind.replace("-", " ")
+ if href:
+ return f'{text}'
+ return f'{text}'
+
+
+def site_url(path: str) -> str:
+ return f"{SITE_URL}/{path.removesuffix('index.html')}"
+
+
+def page(path: str, title: str, description: str, body: str) -> str:
+ root = root_of(path)
+
+ def nav(href: str, label: str) -> str:
+ current = ' aria-current="page"' if path == f"{href}index.html" else ""
+ return f'{label}'
+
+ return f"""
+
+
+
+
+{esc(title)}
+
+
+
+
+
+
+
+
+
+{body}
+
+
+
+
+
+"""
+
+
+def index_page(
+ records: list[dict[str, Any]],
+ reported: list[dict[str, Any]],
+ on_file: dict[str, tuple[str, dict[str, Any] | None]],
+) -> tuple[str, str]:
+ path = "index.html"
+ counts = Counter(kind for kind, _ in on_file.values())
+ counts[""] = len(records)
+ cards = "\n".join(
+ f''
+ for kind, label in {"": "invalid blocks", **EVIDENCE_CARDS}.items()
+ )
+ chips = "".join(
+ f''
+ for rule, count in Counter(r["rule"] for r in records).most_common()
+ )
+ attributes = {"height": ' class="num" aria-sort="ascending"', "observations": ' class="num"'}
+ headers = "".join(
+ f'
'
+ for name in ("height", "date", "hash", "rule", "pool", "evidence", "observations")
+ )
+ rows = []
+ for record in records:
+ kind, _ = on_file[record["hash"]]
+ pool = record.get("context", {}).get("pool", "")
+ date = utc(record["nTime"], "%Y-%m-%d")
+ search = " ".join([str(record["height"]), record["hash"], record["rule"], record["core_reject_reason"], pool, date]).lower()
+ rows.append(f"""
Headers and blocks with valid proof of work that fail a named consensus rule.
+
+
A stale block follows the rules but ends up outside the active chain; these blocks break them.
+They were observed on the Bitcoin network or recovered from other sources, including chains that merge-mine with Bitcoin and archived block explorers.
+The data is in data/invalid-blocks.jsonl, and contributions are welcome in the repository.
"""
+ description = f"{len(records)} Bitcoin blocks with valid proof of work that fail a named consensus rule, with headers, full blocks, evidence and observations."
+ return path, page(path, "Bitcoin invalid blocks", description, body)
+
+
+def context_rows(context: dict[str, Any]) -> list[tuple[str, str]]:
+ """List a record's context for display: the pool fields share a row, and parent_kind shows with the previous block."""
+ rows = []
+ if "pool" in context:
+ source = f' (source)' if "pool_provenance" in context else ""
+ rows.append(("pool", f'{esc(context["pool"])} by {esc(context["pool_basis"])}{source}'))
+ for key, value in context.items():
+ if key in ("pool", "pool_basis", "pool_provenance", "parent_kind"):
+ continue
+ shown = f'{esc(str(value))}'
+ if key == "parent_mtp":
+ shown += f' {utc(value)}'
+ rows.append((key, shown))
+ if key == "coinbase_scriptsig_hex":
+ text = "".join(chr(b) if 32 <= b < 127 else "·" for b in bytes.fromhex(value))
+ rows.append(("scriptSig as text", f'{esc(text)}'))
+ return rows
+
+
+def observations_panel(observations: list[dict[str, Any]]) -> str:
+ if not observations:
+ return ""
+ rows = []
+ for obs in observations:
+ child = esc(obs.get("child_chain", ""))
+ if "child_height" in obs:
+ child += f' {obs["child_height"]}'
+ seen = utc(obs["first_seen"]) if "first_seen" in obs else ""
+ url = esc(obs["provenance"])
+ rows.append(
+ f'