diff --git a/mirror/html/pages.py b/mirror/html/pages.py index c0f25ca..3eaa77f 100644 --- a/mirror/html/pages.py +++ b/mirror/html/pages.py @@ -91,12 +91,14 @@ def render_entry_page( data: dict[str, Any], config: Config, md: MarkdownRenderer, + linked: list[EntryMeta] | None = None, ) -> str: """Render a single issue or PR page.""" + _linked = linked or [] if meta.is_pr: - content = _render_pull_content(meta, data, config, md) + content = _render_pull_content(meta, data, config, md, _linked) else: - content = _render_issue_content(meta, data, config, md) + content = _render_issue_content(meta, data, config, md, _linked) return base_page(f"{meta.title} #{meta.number}", content, config) @@ -105,6 +107,7 @@ def _render_issue_content( data: dict[str, Any], config: Config, md: MarkdownRenderer, + linked: list[EntryMeta], ) -> str: """Render issue page content (port of partials/render-issue.html).""" b = config.base_url @@ -139,7 +142,7 @@ def _render_issue_content( events_html = "\n".join(events_html_parts) # Sidebar - sidebar = _render_sidebar(contributors, meta, issue, config) + sidebar = _render_sidebar(contributors, meta, issue, config, linked=linked) return f"""\
@@ -147,9 +150,12 @@ def _render_issue_content(

{html_escape(meta.title)} #{meta.number} - + {svg_icon(b, "box-arrow-up-right")} + + {svg_icon(b, "diagram-2")} +

{badge} @@ -181,6 +187,7 @@ def _render_pull_content( data: dict[str, Any], config: Config, md: MarkdownRenderer, + linked: list[EntryMeta], ) -> str: """Render PR page content (port of partials/render-pull.html).""" b = config.base_url @@ -249,7 +256,7 @@ def _render_pull_content( """ # Sidebar - sidebar = _render_sidebar(contributors, meta, pull, config, reviewers_html) + sidebar = _render_sidebar(contributors, meta, pull, config, reviewers_html, linked=linked) # Tabs tabs = f"""\ @@ -281,9 +288,12 @@ def _render_pull_content(

{html_escape(meta.title)} #{meta.number} - + {svg_icon(b, "box-arrow-up-right")} + + {svg_icon(b, "diagram-2")} +

{badge} @@ -318,15 +328,26 @@ def _render_pull_content( """ +def _sidebar_section(title: str, items_html: str) -> str: + return ( + f'
' + f'{title}' + f'
{items_html}
' + f'
' + ) + + def _render_sidebar( contributors: list[dict[str, Any]], meta: EntryMeta, item: dict[str, Any], config: Config, extra_html: str = "", + linked: list[EntryMeta] | None = None, ) -> str: - """Render the right sidebar: contributors, labels, milestone, optional extras.""" + """Render the right sidebar: contributors, labels, milestone, linked items.""" b = config.base_url + linked = linked or [] # Deduplicate contributors by login seen: set[str] = set() @@ -344,20 +365,21 @@ def _render_sidebar( login = c.get("login", "unknown") avatar = c.get("avatar_url", "?") contrib_parts.append( - f'' - f'' - f'{html_escape(login)}' + f'' + f'' + f'{html_escape(login)}' ) labels_html = "" if meta.labels: - labels_html = f"""\ -

- Labels -
- {issue_labels(meta.labels, b)} -

-""" + label_items = "".join( + f'' + f'{render_label(name)}' + for name in meta.labels + ) + labels_html = _sidebar_section("Labels", label_items) milestone_html = "" if item.get("milestone"): @@ -372,15 +394,32 @@ def _render_sidebar(

""" + linked_html = "" + if linked: + graph_link = ( + f'' + f'view graph' + f'' + ) + items_html = "".join( + f'' + f'' + f'' + f'#{e.number} {html_escape(e.title)}' + f'' + f'' + for e in linked + ) + linked_html = _sidebar_section(f"Linked ({graph_link})", items_html) + + contributors_html = "".join(contrib_parts) return f"""\ -

- -
- {"".join(contrib_parts)} -

+ {_sidebar_section("Contributors", contributors_html)} {extra_html} {labels_html} {milestone_html} + {linked_html} """ diff --git a/mirror/html/renderer.py b/mirror/html/renderer.py index 2ab6221..a362499 100644 --- a/mirror/html/renderer.py +++ b/mirror/html/renderer.py @@ -79,11 +79,25 @@ def _render_home(self) -> None: def _render_entries(self) -> None: """Render each issue/PR page. Re-reads full JSON one at a time.""" + # Build undirected adjacency: number -> set of linked numbers. + linked: dict[int, set[int]] = {} + for link in self.index.graph.links: + src, tgt = link["source"], link["target"] + linked.setdefault(src, set()).add(tgt) + linked.setdefault(tgt, set()).add(src) + total = len(self.index.entries) for i, meta in enumerate(self.index.entries): if (i + 1) % 100 == 0 or i == 0 or i == total - 1: print(f"Rendering entries... {i + 1}/{total}", end="\r", flush=True) + # Resolve linked EntryMeta objects (only those present in the index). + linked_entries = sorted( + (self.index.by_number[n] for n in linked.get(meta.number, set()) + if n in self.index.by_number), + key=lambda e: e.number, + ) + # Re-read full JSON data = _read_json(meta.json_path) remove_nested_keys(data) @@ -91,7 +105,7 @@ def _render_entries(self) -> None: if meta.is_pr and "comments" in data: build_pull_timeline(data) - html = render_entry_page(meta, data, self.config, self.md) + html = render_entry_page(meta, data, self.config, self.md, linked_entries) self._write(self.config.output_dir / str(meta.number) / "index.html", html) del data # Release memory diff --git a/mirror/html/templates.py b/mirror/html/templates.py index 779bdb0..1441b3c 100644 --- a/mirror/html/templates.py +++ b/mirror/html/templates.py @@ -57,6 +57,13 @@ def style_css() -> str: padding: 0.4em; } +.state-dot { + width: 6px; + height: 6px; + padding: 0; + border-radius: 50%; +} + .state-complete { background-color: var(--bs-indigo) !important; color: var(--bs-light); @@ -314,10 +321,48 @@ def search_script(config: Config) -> str: def graph_script(config: Config) -> str: - """Return the force-graph JS (port of partials/graph.html).""" + """Return the force-graph JS. + + Reads ?start=N from the URL. The graph starts with that node and its + immediate neighbours rendered as cards. Clicking a card expands its hidden + neighbours; ctrl/cmd-clicking opens the issue/PR page. + """ b = config.base_url return f"""\ -
+ + +
+
+ +
+ click card to open · click “Load +N connected” to expand
@@ -326,80 +371,311 @@ def graph_script(config: Config) -> str: """ diff --git a/mirror/index.py b/mirror/index.py index 034a329..53cdf9f 100644 --- a/mirror/index.py +++ b/mirror/index.py @@ -3,6 +3,7 @@ from __future__ import annotations import json +import re from pathlib import Path from typing import Any @@ -18,21 +19,70 @@ def _read_json(path: Path) -> dict[str, Any]: return json.load(f) +# Bots whose cross-references are noise and should be excluded from the graph. +_GRAPH_EXCLUDED_ACTORS = frozenset({"drahtbot"}) + +# Matches any GitHub issue/PR URL: captures (owner, repo, number). +_RE_GITHUB_URL = re.compile( + r"https?://github\.com/([^/\s]+)/([^/\s#]+)/(?:issues|pull)/(\d+)" +) +# Matches standalone #NNN references (same-repo convention on GitHub). +# Applied only after GitHub URLs have been stripped from the body. +_RE_BODY_REF = re.compile(r"(? list[dict[str, int]]: - """Extract cross-reference links from events for the graph.""" + """Extract cross-reference links for the graph. + + Sources: + - cross-referenced events (GitHub records these on the *target* when + something references it, so they capture incoming links) + - same-repo GitHub URLs in the body (cross-referenced events are sometimes + absent from backups; URL parsing fills the gap for outgoing links) + - plain #NNN references in the body (same-repo convention on GitHub) + """ + seen: set[int] = set() links: list[dict[str, int]] = [] + for event in data.get("events", []): if event.get("event") != "cross-referenced": continue + actor = (event.get("actor") or {}).get("login", "") + if actor.lower() in _GRAPH_EXCLUDED_ACTORS: + continue source_issue = event.get("source", {}).get("issue", {}) repo_url = source_issue.get("repository_url", "") - if owner_repo in repo_url: - links.append({ - "source": source_number, - "target": source_issue["number"], - }) + target = source_issue.get("number") + if target and owner_repo in repo_url and target not in seen: + seen.add(target) + links.append({"source": source_number, "target": target}) + + if body: + owner_lower, repo_lower = (s.lower() for s in owner_repo.split("/", 1)) + + def _on_github_url(m: re.Match) -> str: + # Strip every GitHub URL regardless of repo; only record same-repo ones. + if m.group(1).lower() == owner_lower and m.group(2).lower() == repo_lower: + target = int(m.group(3)) + if target != source_number and target not in seen: + seen.add(target) + links.append({"source": source_number, "target": target}) + return "" + + # Pass 1: full GitHub URLs. Strip them all so their numbers are not + # picked up by the plain #NNN scan below. + cleaned = _RE_GITHUB_URL.sub(_on_github_url, body) + + # Pass 2: plain #NNN references in the URL-stripped body. + for m in _RE_BODY_REF.finditer(cleaned): + target = int(m.group(1)) + if target != source_number and target not in seen: + seen.add(target) + links.append({"source": source_number, "target": target}) + return links @@ -71,8 +121,9 @@ def build_index(config: Config) -> SiteIndex: "is_pr": False, "url": f"{config.base_url}{meta.number}/", }) + body = (data.get("issue") or {}).get("body") or "" index.graph.links.extend( - _extract_graph_links(data, meta.number, owner_repo) + _extract_graph_links(data, meta.number, owner_repo, body) ) _index_entry(index, meta) @@ -97,8 +148,9 @@ def build_index(config: Config) -> SiteIndex: "is_pr": True, "url": f"{config.base_url}{meta.number}/", }) + body = (data.get("pull") or {}).get("body") or "" index.graph.links.extend( - _extract_graph_links(data, meta.number, owner_repo) + _extract_graph_links(data, meta.number, owner_repo, body) ) _index_entry(index, meta) @@ -113,7 +165,9 @@ def build_index(config: Config) -> SiteIndex: def _index_entry(index: SiteIndex, meta: EntryMeta) -> None: - """Add an entry to the contributor and label indexes.""" + """Add an entry to the contributor, label, and number indexes.""" + index.by_number[meta.number] = meta + login_lower = meta.contributor.lower() index.by_contributor.setdefault(login_lower, []).append(meta) index.contributor_avatars[login_lower] = meta.avatar_url diff --git a/mirror/models.py b/mirror/models.py index 8ebb713..80cdcc7 100644 --- a/mirror/models.py +++ b/mirror/models.py @@ -39,6 +39,7 @@ class SiteIndex: """In-memory index built during Pass 1. Used by all renderers.""" entries: list[EntryMeta] = field(default_factory=list) # sorted by date desc + by_number: dict[int, EntryMeta] = field(default_factory=dict) # number -> entry by_contributor: dict[str, list[EntryMeta]] = field(default_factory=dict) # lowercase login -> entries by_label: dict[str, list[EntryMeta]] = field(default_factory=dict) # label name -> entries contributor_avatars: dict[str, str] = field(default_factory=dict) # lowercase login -> avatar URL diff --git a/tests/test_index.py b/tests/test_index.py index a64a998..2933772 100644 --- a/tests/test_index.py +++ b/tests/test_index.py @@ -11,50 +11,199 @@ from mirror.index import build_index, _extract_graph_links -class TestExtractGraphLinks(unittest.TestCase): - def test_cross_reference_same_repo(self) -> None: +def _xref_event(number: int, repo_url: str, actor: str | None = None) -> dict: + """Build a minimal cross-referenced event.""" + return { + "event": "cross-referenced", + "actor": {"login": actor} if actor else None, + "source": { + "issue": { + "number": number, + "repository_url": f"https://api.github.com/repos/{repo_url}", + } + }, + } + + +class TestExtractGraphLinksEvents(unittest.TestCase): + """cross-referenced event handling.""" + + def test_same_repo(self) -> None: + data = {"events": [_xref_event(99, "owner/repo")]} + self.assertEqual( + _extract_graph_links(data, 42, "owner/repo"), + [{"source": 42, "target": 99}], + ) + + def test_different_repo_excluded(self) -> None: + data = {"events": [_xref_event(99, "other/repo")]} + self.assertEqual(_extract_graph_links(data, 42, "owner/repo"), []) + + def test_excluded_bot_actor(self) -> None: + data = {"events": [_xref_event(99, "owner/repo", actor="drahtbot")]} + self.assertEqual(_extract_graph_links(data, 42, "owner/repo"), []) + + def test_excluded_bot_actor_case_insensitive(self) -> None: + data = {"events": [_xref_event(99, "owner/repo", actor="DrahtBot")]} + self.assertEqual(_extract_graph_links(data, 42, "owner/repo"), []) + + def test_null_actor_allowed(self) -> None: + # actor field can be null in the JSON + data = {"events": [_xref_event(99, "owner/repo", actor=None)]} + self.assertEqual( + _extract_graph_links(data, 42, "owner/repo"), + [{"source": 42, "target": 99}], + ) + + def test_non_cross_reference_events_ignored(self) -> None: + data = {"events": [{"event": "labeled", "created_at": "2023-01-01T00:00:00Z"}]} + self.assertEqual(_extract_graph_links(data, 42, "owner/repo"), []) + + def test_no_events_key(self) -> None: + self.assertEqual(_extract_graph_links({}, 42, "owner/repo"), []) + + def test_multiple_events_deduplication(self) -> None: + # Same target appears twice (e.g. two cross-refs) — should produce one link. data = { "events": [ - { - "event": "cross-referenced", - "source": { - "issue": { - "number": 99, - "repository_url": "https://api.github.com/repos/owner/repo", - } - }, - } + _xref_event(99, "owner/repo"), + _xref_event(99, "owner/repo"), ] } links = _extract_graph_links(data, 42, "owner/repo") self.assertEqual(links, [{"source": 42, "target": 99}]) - def test_cross_reference_different_repo(self) -> None: + def test_multiple_distinct_events(self) -> None: data = { "events": [ - { - "event": "cross-referenced", - "source": { - "issue": { - "number": 99, - "repository_url": "https://api.github.com/repos/other/repo", - } - }, - } + _xref_event(10, "owner/repo"), + _xref_event(20, "owner/repo"), + _xref_event(30, "other/repo"), # excluded ] } links = _extract_graph_links(data, 42, "owner/repo") + self.assertEqual(links, [ + {"source": 42, "target": 10}, + {"source": 42, "target": 20}, + ]) + + +class TestExtractGraphLinksBody(unittest.TestCase): + """Body parsing: same-repo GitHub URLs and plain #NNN references.""" + + # --- same-repo GitHub URLs --- + + def test_same_repo_issue_url(self) -> None: + body = "Fixes https://github.com/owner/repo/issues/100" + links = _extract_graph_links({}, 42, "owner/repo", body) + self.assertEqual(links, [{"source": 42, "target": 100}]) + + def test_same_repo_pull_url(self) -> None: + body = "Based on https://github.com/owner/repo/pull/200" + links = _extract_graph_links({}, 42, "owner/repo", body) + self.assertEqual(links, [{"source": 42, "target": 200}]) + + def test_same_repo_url_case_insensitive_owner(self) -> None: + # GitHub owner/repo comparisons are case-insensitive. + body = "See https://github.com/Owner/Repo/issues/55" + links = _extract_graph_links({}, 42, "owner/repo", body) + self.assertEqual(links, [{"source": 42, "target": 55}]) + + def test_other_repo_url_excluded(self) -> None: + body = "See https://github.com/other/repo/issues/99 for background." + links = _extract_graph_links({}, 42, "owner/repo", body) self.assertEqual(links, []) - def test_non_cross_reference_events_ignored(self) -> None: - data = {"events": [{"event": "labeled", "created_at": "2023-01-01T00:00:00Z"}]} - links = _extract_graph_links(data, 42, "owner/repo") + def test_other_repo_url_number_not_picked_up_by_hash_regex(self) -> None: + # The number inside an other-repo URL must NOT be captured by #NNN scan. + # e.g. "github.com/secp256k1/issues/7" — 7 should not become a link. + body = "Similar to https://github.com/other/repo/issues/7" + links = _extract_graph_links({}, 42, "owner/repo", body) + self.assertEqual(links, []) + + def test_multiple_same_repo_urls(self) -> None: + body = ( + "This fixes https://github.com/owner/repo/issues/10 " + "and https://github.com/owner/repo/pull/20." + ) + links = _extract_graph_links({}, 42, "owner/repo", body) + self.assertIn({"source": 42, "target": 10}, links) + self.assertIn({"source": 42, "target": 20}, links) + self.assertEqual(len(links), 2) + + def test_mixed_repos_in_urls(self) -> None: + body = ( + "Builds on https://github.com/owner/repo/issues/5. " + "Related: https://github.com/other/project/issues/999." + ) + links = _extract_graph_links({}, 42, "owner/repo", body) + self.assertEqual(links, [{"source": 42, "target": 5}]) + + # --- plain #NNN references --- + + def test_plain_hash_ref(self) -> None: + body = "This is related to #123." + links = _extract_graph_links({}, 42, "owner/repo", body) + self.assertEqual(links, [{"source": 42, "target": 123}]) + + def test_hash_ref_at_start_of_body(self) -> None: + body = "#10 is the predecessor." + links = _extract_graph_links({}, 42, "owner/repo", body) + self.assertEqual(links, [{"source": 42, "target": 10}]) + + def test_hash_ref_self_excluded(self) -> None: + # A body mentioning its own number should not create a self-link. + body = "This is PR #42 which fixes things." + links = _extract_graph_links({}, 42, "owner/repo", body) self.assertEqual(links, []) - def test_no_events(self) -> None: - links = _extract_graph_links({}, 42, "owner/repo") + def test_hash_ref_inside_word_not_matched(self) -> None: + # "issue#42" or "PR#42" should not match — the lookbehind requires non-word char. + body = "See issue#42 for context." + links = _extract_graph_links({}, 1, "owner/repo", body) self.assertEqual(links, []) + def test_hash_ref_multiple(self) -> None: + body = "Relates to #10, #20, and #30." + links = _extract_graph_links({}, 42, "owner/repo", body) + targets = {l["target"] for l in links} + self.assertEqual(targets, {10, 20, 30}) + + def test_hash_ref_deduplicated(self) -> None: + body = "See #99 and also #99 again." + links = _extract_graph_links({}, 42, "owner/repo", body) + self.assertEqual(links, [{"source": 42, "target": 99}]) + + def test_no_body(self) -> None: + links = _extract_graph_links({}, 42, "owner/repo", None) + self.assertEqual(links, []) + + def test_empty_body(self) -> None: + links = _extract_graph_links({}, 42, "owner/repo", "") + self.assertEqual(links, []) + + # --- interaction between URL pass and #NNN pass --- + + def test_same_repo_url_and_hash_ref_deduplication(self) -> None: + # A full URL and a plain #NNN pointing to the same issue should not + # produce duplicate links. + body = "See https://github.com/owner/repo/issues/77 (i.e. #77)." + links = _extract_graph_links({}, 42, "owner/repo", body) + self.assertEqual(links, [{"source": 42, "target": 77}]) + + def test_event_and_body_deduplication(self) -> None: + # cross-referenced event + body mention of the same target → one link. + data = {"events": [_xref_event(99, "owner/repo")]} + links = _extract_graph_links(data, 42, "owner/repo", body="Also see #99.") + self.assertEqual(links, [{"source": 42, "target": 99}]) + + def test_event_and_body_combined(self) -> None: + # Event brings in 99, body brings in 100 → both present. + data = {"events": [_xref_event(99, "owner/repo")]} + links = _extract_graph_links(data, 42, "owner/repo", body="#100 is related.") + targets = {l["target"] for l in links} + self.assertEqual(targets, {99, 100}) + class TestBuildIndex(unittest.TestCase): def setUp(self) -> None: