diff --git a/CHANGELOG.md b/CHANGELOG.md index 1bfdd30..45efa95 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -14,6 +14,7 @@ Full release notes with details on each version: [GitHub Releases](https://githu - Fix: `to_canvas` (Obsidian Canvas export) now lays out each community's node cards in the same `ceil(sqrt(n))`-column grid the group box is sized for. The box width assumed a roughly-square `sqrt(n)`-column layout, but the placement loop hardcoded 3 columns, so any community larger than ~9 members rendered as a cramped 3-wide strip in an over-wide, mostly-empty box. The column count is now computed once per community and reused for the box width, box height, and card placement, so the cards fill the box. Cosmetic, no data change (#1452, thanks @TPAteeq). - Fix: `to_obsidian` / `to_canvas` / `to_wiki` no longer silently overwrite notes whose labels differ only by case (e.g. a class `References` and a prose heading `references`). The filename dedup was keyed on the exact-case name, so two such labels counted as non-colliding and the second write clobbered the first on case-insensitive filesystems (macOS/APFS, Windows/NTFS) — no suffix, no warning. Dedup now folds case (keyed on the lowercased name) while still emitting the original-case filename, so any pair that would collide on disk gets a numeric suffix. The obsidian/canvas dedup is shared in one helper so they can't drift, `wiki`'s slug dedup gets the matching fix, the `_COMMUNITY_*.md` overview notes (which had no dedup) are covered, and a generated `base_1` is itself re-checked so it can't overwrite a node literally labelled `base_1` (#1453, thanks @TPAteeq). - Feat: the `kimi`, `gemini`, and `deepseek` semantic-extraction backends now honor `KIMI_BASE_URL`, `GEMINI_BASE_URL`, and `DEEPSEEK_BASE_URL` to point at any OpenAI-compatible endpoint (a proxy, gateway, or self-hosted relay), matching the existing `OLLAMA_BASE_URL` / `OPENAI_BASE_URL` overrides. Each falls back to its hardcoded official default when the variable is unset, so behavior is unchanged for everyone who doesn't set it (#1458, thanks @jc2shile). +- Fix: `to_wiki` (Wikipedia-style wiki export) now emits portable relative markdown links instead of Obsidian `[[wikilinks]]`, so navigation works in every renderer — VS Code preview, GitHub, GitLab, a plain browser — not just Obsidian. Two defects: (1) `[[Title]]` resolves by note title only inside Obsidian; everywhere else `[[Domain Data Models]]` points at a literal `Domain Data Models.md`, but the article file is `Domain_Data_Models.md` (the slug substitutes spaces and reserved characters), so nearly every community/god-node navigation link opened an empty page. (2) God-node articles linked every neighbor (`[[AwsHelper.py]]`, `[[.read_object_key()]]`), but only communities and god nodes get article files, so those node-level links were dead even inside Obsidian. Links are now standard `[display](slug.md)` with the target URL-encoded, so spaces, `&`, parentheses, and `#` survive intact in CommonMark renderers and Obsidian alike; any link whose target has no article is downgraded to plain text instead of left dangling. Each article's slug is computed up front (a `label -> slug` resolver built before any body is rendered) so a link to a community or god-node article points at the real on-disk filename, including the case-fold collision suffix (`parser_2.md`). Cosmetic, no graph/data change (#1444, thanks @restagner). ## 0.8.49 (2026-06-24) diff --git a/graphify/wiki.py b/graphify/wiki.py index 4212fb7..cb9c6cf 100644 --- a/graphify/wiki.py +++ b/graphify/wiki.py @@ -3,6 +3,7 @@ from __future__ import annotations from collections import Counter from pathlib import Path +from urllib.parse import quote import networkx as nx from graphify.build import edge_data @@ -23,6 +24,30 @@ def _safe_filename(name: str) -> str: return s[:200] if s else 'unnamed' +def _md_link(label: str, resolver: dict[str, str]) -> str: + """Render a link to another wiki article as a portable relative markdown link. + + ``resolver`` maps an article's display label to the slug (filename stem) it + was written under. When the label has an article, emit a standard + ``[label](slug.md)`` link, URL-encoding the target so any spaces, parens, & + or # in the slug survive every CommonMark renderer (GitHub, GitLab, VS Code + preview, a plain browser) and Obsidian alike. The old ``[[label]]`` form + only resolved inside Obsidian, because the on-disk filename differs from the + label — _safe_filename turns spaces into underscores and substitutes + reserved characters — so e.g. ``[[Domain Data Models]]`` pointed at a + non-existent ``Domain Data Models.md`` everywhere else. + + Labels with no article — most node-level links, since only communities and + god nodes get article files — render as plain text instead of a dead link + that points nowhere even inside Obsidian. + """ + text = label.replace("[", r"\[").replace("]", r"\]") + slug = resolver.get(label) + if slug is None: + return text + return f"[{text}]({quote(f'{slug}.md')})" + + def _cross_community_links(G: nx.Graph, nodes: list[str], own_cid: int, labels: dict[int, str], node_community: dict[str, int]) -> list[tuple[str, int]]: """Return (community_label, edge_count) pairs for cross-community connections, sorted descending.""" counts: dict[str, int] = Counter() @@ -42,7 +67,9 @@ def _community_article( labels: dict[int, str], cohesion: float | None, node_community: dict[str, int] | None = None, + resolver: dict[str, str] | None = None, ) -> str: + resolver = resolver or {} top_nodes = sorted(nodes, key=lambda n: G.degree(n), reverse=True)[:25] cross = _cross_community_links(G, nodes, cid, labels, node_community or {}) @@ -80,7 +107,7 @@ def _community_article( lines += ["## Relationships", ""] if cross: for other_label, count in cross[:12]: - lines.append(f"- [[{other_label}]] ({count} shared connections)") + lines.append(f"- {_md_link(other_label, resolver)} ({count} shared connections)") else: lines.append("- No strong cross-community connections detected") lines.append("") @@ -98,11 +125,12 @@ def _community_article( lines.append(f"- {conf}: {n} ({pct}%)") lines.append("") - lines += ["---", "", "*Part of the graphify knowledge wiki. See [[index]] to navigate.*"] + lines += ["---", "", f"*Part of the graphify knowledge wiki. See {_md_link('index', resolver)} to navigate.*"] return "\n".join(lines) -def _god_node_article(G: nx.Graph, nid: str, labels: dict[int, str], node_community: dict[str, int] | None = None) -> str: +def _god_node_article(G: nx.Graph, nid: str, labels: dict[int, str], node_community: dict[str, int] | None = None, resolver: dict[str, str] | None = None) -> str: + resolver = resolver or {} d = G.nodes[nid] node_label = d.get("label", nid) src = d.get("source_file", "") @@ -114,7 +142,7 @@ def _god_node_article(G: nx.Graph, nid: str, labels: dict[int, str], node_commun lines += [f"> God node · {G.degree(nid)} connections · `{src}`", ""] if community_name: - lines += [f"**Community:** [[{community_name}]]", ""] + lines += [f"**Community:** {_md_link(community_name, resolver)}", ""] # Group neighbors by relation type by_relation: dict[str, list[str]] = {} @@ -125,7 +153,7 @@ def _god_node_article(G: nx.Graph, nid: str, labels: dict[int, str], node_commun neighbor_label = nd.get("label", neighbor) conf = ed.get("confidence", "") conf_str = f" `{conf}`" if conf else "" - by_relation.setdefault(rel, []).append(f"[[{neighbor_label}]]{conf_str}") + by_relation.setdefault(rel, []).append(f"{_md_link(neighbor_label, resolver)}{conf_str}") lines += ["## Connections by Relation", ""] for rel, targets in sorted(by_relation.items()): @@ -134,7 +162,7 @@ def _god_node_article(G: nx.Graph, nid: str, labels: dict[int, str], node_commun lines.append(f"- {t}") lines.append("") - lines += ["---", "", "*Part of the graphify knowledge wiki. See [[index]] to navigate.*"] + lines += ["---", "", f"*Part of the graphify knowledge wiki. See {_md_link('index', resolver)} to navigate.*"] return "\n".join(lines) @@ -144,7 +172,9 @@ def _index_md( god_nodes_data: list[dict], total_nodes: int, total_edges: int, + resolver: dict[str, str] | None = None, ) -> str: + resolver = resolver or {} lines: list[str] = [ "# Knowledge Graph Index", "", @@ -161,13 +191,13 @@ def _index_md( for cid, nodes in sorted(communities.items(), key=lambda x: -len(x[1])): label = labels.get(cid, f"Community {cid}") - lines.append(f"- [[{label}]] — {len(nodes)} nodes") + lines.append(f"- {_md_link(label, resolver)} — {len(nodes)} nodes") lines.append("") if god_nodes_data: lines += ["## God Nodes", "(most connected concepts — the load-bearing abstractions)", ""] for node in god_nodes_data: - lines.append(f"- [[{node['label']}]] — {node['degree']} connections") + lines.append(f"- {_md_link(node['label'], resolver)} — {node['degree']} connections") lines.append("") lines += [ @@ -260,26 +290,47 @@ def to_wiki( used_slugs.add(slug.lower()) return slug - # Community articles - for cid, nodes in communities.items(): - label = labels.get(cid, f"Community {cid}") - article = _community_article(G, cid, nodes, label, labels, cohesion.get(cid), node_community) - slug = _unique_slug(_safe_filename(label)) - (out / f"{slug}.md").write_text(article, encoding="utf-8") - count += 1 + # First pass: assign every article its slug before rendering any body, so the + # bodies can link to one another. A link's target is the on-disk filename (the + # slug), which differs from the label — _safe_filename turns spaces into + # underscores and substitutes reserved chars, and a slug may pick up a numeric + # suffix from collision dedup — so the final slug must be known up front. + # resolver maps display label -> slug; labels with no article are absent, so + # _md_link renders them as plain text. Communities are slugged before god nodes + # (and setdefault keeps the first), preserving the filename-assignment order + # the case-collision dedup relies on. + resolver: dict[str, str] = {"index": "index"} - # God node articles + community_slugs: dict[int, str] = {} + for cid in communities: + label = labels.get(cid, f"Community {cid}") + slug = _unique_slug(_safe_filename(label)) + community_slugs[cid] = slug + resolver.setdefault(label, slug) + + god_articles: list[tuple[str, str]] = [] # (node_id, slug) for node_data in god_nodes_data: nid = node_data.get("id") if nid and nid in G: - article = _god_node_article(G, nid, labels, node_community) slug = _unique_slug(_safe_filename(node_data['label'])) - (out / f"{slug}.md").write_text(article, encoding="utf-8") - count += 1 + god_articles.append((nid, slug)) + resolver.setdefault(node_data['label'], slug) + + # Second pass: render and write each article with the full resolver in hand. + for cid, nodes in communities.items(): + label = labels.get(cid, f"Community {cid}") + article = _community_article(G, cid, nodes, label, labels, cohesion.get(cid), node_community, resolver) + (out / f"{community_slugs[cid]}.md").write_text(article, encoding="utf-8") + count += 1 + + for nid, slug in god_articles: + article = _god_node_article(G, nid, labels, node_community, resolver) + (out / f"{slug}.md").write_text(article, encoding="utf-8") + count += 1 # Index (out / "index.md").write_text( - _index_md(communities, labels, god_nodes_data, G.number_of_nodes(), G.number_of_edges()), + _index_md(communities, labels, god_nodes_data, G.number_of_nodes(), G.number_of_edges(), resolver), encoding="utf-8", ) diff --git a/tests/test_wiki.py b/tests/test_wiki.py index b4fc97c..289a0a2 100644 --- a/tests/test_wiki.py +++ b/tests/test_wiki.py @@ -1,9 +1,24 @@ """Tests for graphify.wiki — Wikipedia-style article generation.""" +import re +import urllib.parse import pytest from pathlib import Path import networkx as nx from graphify.wiki import to_wiki, _index_md, _community_article, _god_node_article +_MD_LINK = re.compile(r"\[([^\]]+)\]\(([^)]+)\)") + + +def _inline_links(text): + """Yield (display, decoded_target) for each inline markdown link, skipping + external URLs. Targets are URL-decoded so they can be checked against the + on-disk filename. (Display text with an escaped `]` isn't matched, but the + generated labels used in link position never contain brackets.)""" + for display, target in _MD_LINK.findall(text): + if "://" in target: + continue + yield display, urllib.parse.unquote(target) + def _make_graph(): G = nx.Graph() @@ -53,15 +68,15 @@ def test_index_links_all_communities(tmp_path): G = _make_graph() to_wiki(G, COMMUNITIES, tmp_path, community_labels=LABELS) index = (tmp_path / "index.md").read_text() - assert "[[Parsing Layer]]" in index - assert "[[Rendering Layer]]" in index + assert "[Parsing Layer](Parsing_Layer.md)" in index + assert "[Rendering Layer](Rendering_Layer.md)" in index def test_index_lists_god_nodes(tmp_path): G = _make_graph() to_wiki(G, COMMUNITIES, tmp_path, community_labels=LABELS, god_nodes_data=GOD_NODES) index = (tmp_path / "index.md").read_text() - assert "[[parse]]" in index + assert "[parse](parse.md)" in index assert "2 connections" in index @@ -70,7 +85,7 @@ def test_community_article_has_cross_links(tmp_path): to_wiki(G, COMMUNITIES, tmp_path, community_labels=LABELS) parsing = (tmp_path / "Parsing_Layer.md").read_text() # n1 (parsing) references n3 (rendering) → cross-community link - assert "[[Rendering Layer]]" in parsing + assert "[Rendering Layer](Rendering_Layer.md)" in parsing def test_community_article_shows_cohesion(tmp_path): @@ -92,14 +107,18 @@ def test_god_node_article_has_connections(tmp_path): G = _make_graph() to_wiki(G, COMMUNITIES, tmp_path, community_labels=LABELS, god_nodes_data=GOD_NODES) article = (tmp_path / "parse.md").read_text() - assert "[[validate]]" in article or "[[render]]" in article + # parse's neighbours (validate, render) have no article of their own, so the + # connections list shows them as plain text rather than as links. + assert "validate" in article and "render" in article + assert "[[" not in article + assert "](validate.md)" not in article and "](render.md)" not in article def test_god_node_article_links_community(tmp_path): G = _make_graph() to_wiki(G, COMMUNITIES, tmp_path, community_labels=LABELS, god_nodes_data=GOD_NODES) article = (tmp_path / "parse.md").read_text() - assert "[[Parsing Layer]]" in article + assert "[Parsing Layer](Parsing_Layer.md)" in article def test_to_wiki_skips_missing_god_node_ids(tmp_path): @@ -116,13 +135,16 @@ def test_to_wiki_no_labels_uses_fallback(tmp_path): to_wiki(G, COMMUNITIES, tmp_path) # no labels assert (tmp_path / "Community_0.md").exists() assert (tmp_path / "Community_1.md").exists() + # fallback "Community N" labels still produce links that resolve to the file + targets = [t for _, t in _inline_links((tmp_path / "index.md").read_text())] + assert "Community_0.md" in targets and (tmp_path / "Community_0.md").exists() def test_article_navigation_footer(tmp_path): G = _make_graph() to_wiki(G, COMMUNITIES, tmp_path, community_labels=LABELS) article = (tmp_path / "Parsing_Layer.md").read_text() - assert "[[index]]" in article + assert "[index](index.md)" in article def test_community_article_truncation_notice(tmp_path): @@ -150,7 +172,7 @@ def test_cross_community_links_without_node_community_attrs(tmp_path): labels = {0: "Parsing", 1: "Rendering"} to_wiki(G, communities, tmp_path, community_labels=labels) article = (tmp_path / "Parsing.md").read_text() - assert "[[Rendering]]" in article + assert "[Rendering](Rendering.md)" in article def test_god_node_article_community_without_node_attr(tmp_path): @@ -164,7 +186,7 @@ def test_god_node_article_community_without_node_attr(tmp_path): god_nodes = [{"id": "n1", "label": "parse", "degree": 1}] to_wiki(G, communities, tmp_path, community_labels=labels, god_nodes_data=god_nodes) article = (tmp_path / "parse.md").read_text() - assert "[[Core Logic]]" in article + assert "[Core Logic](Core_Logic.md)" in article # Regression tests for #936 - stale community node IDs crash to_wiki after dedup/re-extract @@ -248,3 +270,122 @@ def test_to_wiki_god_node_label_case_collides_with_community(tmp_path): assert len(articles) == n == 2, [p.name for p in articles] lowered = [p.stem.lower() for p in articles] assert len(set(lowered)) == len(lowered), [p.name for p in articles] + + +# Regression tests for portable wiki links - Obsidian [[wikilinks]] break in +# every non-Obsidian renderer (VS Code preview, GitHub, GitLab, plain browsers). + + +def test_wiki_emits_no_obsidian_wikilinks(tmp_path): + """No generated file may contain Obsidian [[...]] syntax. Those links resolve + only inside Obsidian (by note title); everywhere else [[Domain Data Models]] + points at a literal `Domain Data Models.md` that doesn't exist.""" + G = _make_graph() + to_wiki(G, COMMUNITIES, tmp_path, community_labels=LABELS, cohesion=COHESION, god_nodes_data=GOD_NODES) + for md in tmp_path.glob("*.md"): + assert "[[" not in md.read_text(), md.name + + +def test_wiki_links_resolve_to_real_files(tmp_path): + """Every inline markdown link target across the whole wiki must point at a + file that actually exists on disk. The display text may keep spaces/special + characters, but the target is the URL-encoded slug, so it has to round-trip + back to a real filename in any renderer.""" + G = _make_graph() + to_wiki(G, COMMUNITIES, tmp_path, community_labels=LABELS, cohesion=COHESION, god_nodes_data=GOD_NODES) + seen_link = False + for md in tmp_path.glob("*.md"): + for display, target in _inline_links(md.read_text()): + seen_link = True + assert (tmp_path / target).exists(), f"{md.name}: [{display}] -> {target} is dead" + # guard against the test passing vacuously if links ever stop being emitted + assert seen_link, "expected the wiki to contain inline markdown links" + + +def test_wiki_link_display_keeps_label_but_target_is_filename(tmp_path): + """The fix's whole point: a link's display text is the human label (with + spaces) while its target is the on-disk slug (underscores). This is what + [[Domain Data Models]] could never express portably.""" + G = _make_graph() + to_wiki(G, COMMUNITIES, tmp_path, community_labels=LABELS) + index = (tmp_path / "index.md").read_text() + assert "[Parsing Layer](Parsing_Layer.md)" in index + assert "Parsing Layer.md" not in index # the broken Obsidian-only target + + +def test_wiki_special_characters_in_label_resolve(tmp_path): + """Labels with spaces, &, #, and parentheses must still produce a link whose + URL-encoded target decodes back to the real (underscored) filename, so it + works in CommonMark renderers and Obsidian alike. # is the dangerous one — + left raw in a relative link it would be misread as a fragment.""" + G = nx.Graph() + G.add_node("n1", label="a", file_type="code", source_file="a.py", community=0) + G.add_node("n2", label="b", file_type="code", source_file="b.py", community=1) + G.add_edge("n1", "n2", relation="references", confidence="INFERRED", weight=1.0) + communities = {0: ["n1"], 1: ["n2"]} + labels = {0: "C# & Auth (v2)", 1: "Other"} + to_wiki(G, communities, tmp_path, community_labels=labels) + article = (tmp_path / "Other.md").read_text() + # the cross-link to the special-char community resolves to its real file + targets = [t for _, t in _inline_links(article)] + assert "C#_&_Auth_(v2).md" in targets + assert (tmp_path / "C#_&_Auth_(v2).md").exists() + # the raw target is fully percent-encoded — no bare ( ) that would terminate + # the link early, no bare # that would be misread as a fragment + assert "C%23_%26_Auth_%28v2%29.md" in article + + +def test_wiki_link_with_bracketed_label_resolves(tmp_path): + """A label containing `[` / `]` (e.g. a generic like `Array[T]`) still + produces a resolvable link: the brackets are escaped in the display text so + they don't break the markdown, and percent-encoded in the target so it + decodes back to the real file. (`_safe_filename` keeps brackets in the slug, + so they reach the link target.)""" + G = nx.Graph() + G.add_node("n1", label="a", file_type="code", source_file="a.py", community=0) + G.add_node("n2", label="b", file_type="code", source_file="b.py", community=1) + G.add_edge("n1", "n2", relation="references", confidence="INFERRED", weight=1.0) + communities = {0: ["n1"], 1: ["n2"]} + labels = {0: "Array[T] Models", 1: "Other"} + to_wiki(G, communities, tmp_path, community_labels=labels) + article = (tmp_path / "Other.md").read_text() + assert r"[Array\[T\] Models](Array%5BT%5D_Models.md)" in article + assert (tmp_path / "Array[T]_Models.md").exists() + + +def test_wiki_links_to_nodes_without_articles_are_plain_text(tmp_path): + """A god node links its neighbours, but only communities and god nodes get + article files — neighbours without one must render as plain text, not as a + link (dead even inside Obsidian).""" + G = _make_graph() + # only `parse` (n1) is a god node; its neighbours validate/render are not, + # and have no article of their own + to_wiki(G, COMMUNITIES, tmp_path, community_labels=LABELS, god_nodes_data=GOD_NODES) + article = (tmp_path / "parse.md").read_text() + assert "validate" in article and "render" in article + # they appear as plain list items, not links + assert "- validate" in article and "- render" in article + # not wrapped in an Obsidian wikilink (the old form — dead even in Obsidian + # since validate/render have no article)... + assert "[[validate]]" not in article and "[[render]]" not in article + # ...nor in a standard link to a non-existent article file + for _, target in _inline_links(article): + assert target not in ("validate.md", "render.md"), target + + +def test_wiki_links_use_collision_suffixed_slug(tmp_path): + """When two labels collide on disk and the second article gets a numeric + suffix (`parser_2.md`), links to it must target the suffixed slug, not the + bare label. The resolver records the exact filename each article was written + under, so the link target tracks the collision suffix.""" + G = nx.Graph() + G.add_node("n1", label="a", file_type="code", source_file="a.py", community=0) + G.add_node("n2", label="b", file_type="code", source_file="b.py", community=1) + G.add_edge("n1", "n2", relation="references", confidence="INFERRED", weight=1.0) + communities = {0: ["n1"], 1: ["n2"]} + labels = {0: "Parser", 1: "parser"} # collide case-insensitively + to_wiki(G, communities, tmp_path, community_labels=labels) + index_targets = [t for _, t in _inline_links((tmp_path / "index.md").read_text())] + assert "parser_2.md" in index_targets # link points at the suffixed file... + for t in index_targets: + assert (tmp_path / t).exists(), t # ...and every target is a real file