From abc58c452624e56fd8d5eb31351a46acf0609456 Mon Sep 17 00:00:00 2001 From: rohit-jsfreaky Date: Wed, 26 Aug 2026 01:37:05 +0530 Subject: [PATCH] fix(wiki): link index rows to their own article when labels collide (#3032, #3033) --- graphify/wiki.py | 41 ++++++++++++++++++++++---- tests/test_wiki.py | 73 ++++++++++++++++++++++++++++++++++++++++++++++ 2 files changed, 108 insertions(+), 6 deletions(-) diff --git a/graphify/wiki.py b/graphify/wiki.py index 11c3735b14..8adbc991a9 100644 --- a/graphify/wiki.py +++ b/graphify/wiki.py @@ -83,8 +83,19 @@ def _md_link(label: str, resolver: dict[str, str]) -> str: god nodes get article files — render as plain text instead of a dead link that points nowhere even inside Obsidian. """ + return _md_link_to(label, resolver.get(label)) + + +def _md_link_to(label: str, slug: str | None) -> str: + """Render ``[label](slug.md)`` for a slug that is already known. + + Same output as :func:`_md_link`, minus the label lookup. The index knows the + exact slug of every article it lists — two same-labelled articles have + distinct slugs — so it links by that slug instead of re-deriving one from the + label, which collapses duplicates onto whichever was registered first + (#3032/#3033). ``None`` means no article exists, rendered as plain text. + """ text = label.replace("[", r"\[").replace("]", r"\]") - slug = resolver.get(label) if slug is None: return text return f"[{text}]({slug}.md)" @@ -225,9 +236,21 @@ def _index_md( god_nodes_data: list[dict], total_nodes: int, total_edges: int, - resolver: dict[str, str] | None = None, + community_slugs: dict[int, str], + god_articles: list[tuple[str, str]], ) -> str: - resolver = resolver or {} + """Render index.md. + + Takes the slugs the articles were actually written under — + ``community_slugs`` (cid -> slug) and ``god_articles`` ((node_id, slug)) — + rather than the label -> slug resolver the articles use for prose links. + Two communities, or two god nodes, can share a label while owning separate + files, and a label lookup cannot tell them apart: it points every row at + whichever was registered first and orphans the rest (#3032/#3033). + """ + # A node with no article of its own is absent here, so it renders as plain + # text rather than linking to a same-labelled article that isn't it. + god_slugs = dict(god_articles) lines: list[str] = [ "# Knowledge Graph Index", "", @@ -244,13 +267,15 @@ def _index_md( for cid, nodes in sorted(communities.items(), key=lambda x: -len(x[1])): label = labels.get(cid, f"Community {cid}") - lines.append(f"- {_md_link(label, resolver)} — {len(nodes)} nodes") + slug = community_slugs.get(cid) + lines.append(f"- {_md_link_to(label, slug)} — {len(nodes)} nodes") lines.append("") if god_nodes_data: lines += ["## God Nodes", "(most connected concepts — the load-bearing abstractions)", ""] for node in god_nodes_data: - lines.append(f"- {_md_link(node['label'], resolver)} — {node['degree']} connections") + slug = god_slugs.get(node.get("id")) + lines.append(f"- {_md_link_to(node['label'], slug)} — {node['degree']} connections") lines.append("") lines += [ @@ -389,7 +414,11 @@ def _unique_slug(base: str) -> str: # Index (out / "index.md").write_text( - _index_md(communities, labels, god_nodes_data, G.number_of_nodes(), G.number_of_edges(), resolver), + _index_md( + communities, labels, god_nodes_data, + G.number_of_nodes(), G.number_of_edges(), + community_slugs, god_articles, + ), encoding="utf-8", ) diff --git a/tests/test_wiki.py b/tests/test_wiki.py index 130da843dc..adaed68a19 100644 --- a/tests/test_wiki.py +++ b/tests/test_wiki.py @@ -428,3 +428,76 @@ def test_wiki_links_use_collision_suffixed_slug(tmp_path): assert "parser_2.md" in index_targets # link points at the suffixed file... for t in index_targets: assert (tmp_path / t).exists(), t # ...and every target is a real file + + +# Regression tests for #3032 / #3033 - two articles whose labels are *identical* +# (not merely case-variant, as above) each get their own file, so the index has +# to link each row to the file that row's article was actually written to. + + +def test_index_links_same_labelled_communities_to_own_articles(tmp_path): + """Two communities may carry the same label. Each is written to its own file + (`Core.md`, `Core_2.md`), so the index rows must target one each. Resolving + the target from the label instead points both rows at whichever community + was slugged first, orphaning the second article (#3032).""" + G = nx.Graph() + G.add_node("n1", label="a", file_type="code", source_file="a.py", community=0) + G.add_node("n2", label="b", file_type="code", source_file="b.py", community=0) + G.add_node("n3", label="c", file_type="code", source_file="c.py", community=1) + G.add_node("n4", label="d", file_type="code", source_file="d.py", community=1) + G.add_edge("n1", "n2", relation="calls", confidence="EXTRACTED", weight=1.0) + G.add_edge("n3", "n4", relation="calls", confidence="EXTRACTED", weight=1.0) + G.add_edge("n1", "n3", relation="references", confidence="INFERRED", weight=1.0) + communities = {0: ["n1", "n2"], 1: ["n3", "n4"]} + labels = {0: "Core", 1: "Core"} # identical, not just case-variant + to_wiki(G, communities, tmp_path, community_labels=labels) + index_targets = [t for _, t in _inline_links((tmp_path / "index.md").read_text())] + assert sorted(index_targets) == ["Core.md", "Core_2.md"], index_targets + for t in index_targets: + assert (tmp_path / t).exists(), t + + +def test_index_links_same_labelled_god_nodes_to_own_articles(tmp_path): + """Same for god nodes: two nodes sharing a label get one article each, and + the index must link each row to its own. Resolving by label sends both rows + to the first article and leaves `GameState_2.md` unreachable (#3033).""" + G = nx.Graph() + G.add_node("n1", label="GameState", file_type="code", source_file="a.py", community=0) + G.add_node("n2", label="GameState", file_type="code", source_file="b.py", community=0) + G.add_node("n3", label="helper", file_type="code", source_file="c.py", community=0) + G.add_edge("n1", "n3", relation="calls", confidence="EXTRACTED", weight=1.0) + G.add_edge("n2", "n3", relation="calls", confidence="EXTRACTED", weight=1.0) + communities = {0: ["n1", "n2", "n3"]} + labels = {0: "Only Community"} + god_nodes = [ + {"id": "n1", "label": "GameState", "degree": 1}, + {"id": "n2", "label": "GameState", "degree": 1}, + ] + to_wiki(G, communities, tmp_path, community_labels=labels, god_nodes_data=god_nodes) + index_targets = [t for _, t in _inline_links((tmp_path / "index.md").read_text())] + god_targets = [t for t in index_targets if t.startswith("GameState")] + assert sorted(god_targets) == ["GameState.md", "GameState_2.md"], index_targets + for t in index_targets: + assert (tmp_path / t).exists(), t + + +def test_index_links_every_article_written(tmp_path): + """Every article `to_wiki` writes must be reachable from the index. This is + the general form of #3032/#3033: any row that resolves its target by label + rather than by the slug the article was written under leaves a real file + with no way in.""" + G = nx.Graph() + G.add_node("n1", label="GameState", file_type="code", source_file="a.py", community=0) + G.add_node("n2", label="GameState", file_type="code", source_file="b.py", community=1) + G.add_edge("n1", "n2", relation="references", confidence="INFERRED", weight=1.0) + communities = {0: ["n1"], 1: ["n2"]} + labels = {0: "Core", 1: "Core"} + god_nodes = [ + {"id": "n1", "label": "GameState", "degree": 1}, + {"id": "n2", "label": "GameState", "degree": 1}, + ] + n = to_wiki(G, communities, tmp_path, community_labels=labels, god_nodes_data=god_nodes) + articles = {p.name for p in tmp_path.glob("*.md")} - {"index.md"} + assert len(articles) == n == 4, sorted(articles) + linked = {t for _, t in _inline_links((tmp_path / "index.md").read_text())} + assert articles <= linked, sorted(articles - linked) # no orphaned article