diff --git a/CHANGELOG.md b/CHANGELOG.md index 86e6de3..2f2cdcf 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,7 @@ Full release notes with details on each version: [GitHub Releases](https://githu ## Unreleased +- Fix: the `get_community` MCP tool now shows the community name in its header (`Community 12 — Auth & Sessions (8 nodes)`), matching `get_node` and the query-traversal output, which already read the `community_name` attribute `to_json` writes onto every node. `get_community` was the only graph tool still returning a bare numeric id. The name is read from the community's member nodes (they share it), sanitised like every other LLM-derived field, and skipped when it is just the `Community N` placeholder so the header never doubles to `Community 12 — Community 12` (#1448, thanks @rmart1308). - Security: floor `starlette` at `>=1.3.1` to pick up the fixes for CVE-2026-48818 and CVE-2026-54283 (both resolved by 1.3.1). starlette underpins the HTTP MCP transport (`graphify-mcp` over HTTP / `serve_http`); the stdio transport and CLI are unaffected. It was an undeclared transitive dependency (via `mcp`) that `graphify/serve.py` imports directly, so it is now declared in the `mcp` (and `all`) extras and floored, which protects end users installing `graphifyy[mcp]`, not just the locked dev/CI environment. Lockfile bumped 1.0.0 -> 1.3.1; serve/MCP/HTTP tests pass on the new version (#1391, #1396, thanks @orbisai0security). - Refactor: begin splitting the monolithic `extract.py` into per-language modules under `graphify/extractors/` (#1212). The `blade`, `elixir`, `razor`, and `zig` extractors plus the shared primitives (`_make_id`, `_file_stem`, `_read_text`, `_LANGUAGE_BUILTIN_GLOBALS`) move into their own files, with `graphify/extractors/base.py` holding the shared pieces and a strict one-way import direction (`extract.py` -> `extractors/`, never the reverse). `extract.py` re-exports the moved names, so every `from graphify.extract import ...` caller and the dispatch table are unchanged. Behavior-neutral lift-and-shift (verified byte-identical), groundwork for moving the remaining languages out. See `graphify/extractors/MIGRATION.md`. - Feat: community labeling can now run in parallel (#1390). `graphify cluster-only` and `graphify label` accept `--max-concurrency N` (default 4) to fan labeling batches out across a thread pool, and `--batch-size N` (default 100) to tune communities per LLM call. A large graph that previously needed hundreds of sequential calls now runs them in rounds. Mirrors the existing `extract` parallelism, including the safety guards: `ollama` and `claude-cli` are forced serial (set `GRAPHIFY_OLLAMA_PARALLEL=1` / `GRAPHIFY_CLAUDE_CLI_PARALLEL=1` to override). Output is unchanged and deterministic regardless of concurrency, since results are keyed by community id and merged on the main thread. diff --git a/graphify/serve.py b/graphify/serve.py index 06e8cf7..117e70a 100644 --- a/graphify/serve.py +++ b/graphify/serve.py @@ -631,6 +631,21 @@ def _filter_blank_stdin() -> None: sys.stdin = open(0, "r", closefd=False) +def _community_header(cid: int, community_name) -> str: + # Header for get_community: "Community N — Name", matching get_node / query + # output which read the community_name attribute to_json writes onto nodes. + # Skip the name when it is just the "Community N" placeholder (written for + # unnamed communities) so the header never reads "Community 12 — Community 12"; + # also falls back to the bare id when there is no name. Name is sanitised + # (F-010) like every other LLM-derived field. + base = f"Community {cid}" + if community_name: + clean = sanitize_label(str(community_name)) + if clean and clean != base: + return f"{base} — {clean}" + return base + + def _build_server(graph_path: str): """Build the configured low-level MCP Server (shared by every transport). @@ -898,7 +913,8 @@ def _build_server(graph_path: str): nodes = communities.get(cid, []) if not nodes: return f"Community {cid} not found." - lines = [f"Community {cid} ({len(nodes)} nodes):"] + header = _community_header(cid, G.nodes[nodes[0]].get("community_name")) + lines = [f"{header} ({len(nodes)} nodes):"] for n in nodes: d = G.nodes[n] # Sanitise label and source_file (F-010). diff --git a/tests/test_serve.py b/tests/test_serve.py index a0432a4..4ffa9ae 100644 --- a/tests/test_serve.py +++ b/tests/test_serve.py @@ -23,6 +23,7 @@ from graphify.serve import ( _resolve_context_filters, _subgraph_to_text, _load_graph, + _community_header, ) @@ -682,3 +683,27 @@ def test_query_text_chinese_finds_routing_nodes(): text = _query_graph_text(G, "页面路由", mode="bfs", depth=2) assert "No matching nodes found." not in text assert "路由" in text + + +# --- get_community header (#1448): show the community name, no placeholder doubling --- + +def test_community_header_shows_real_name(): + assert _community_header(12, "Auth & Sessions") == "Community 12 — Auth & Sessions" + + +def test_community_header_skips_placeholder_name(): + # community_name is written as the "Community N" placeholder for unnamed + # communities; the header must not read "Community 12 — Community 12". + assert _community_header(12, "Community 12") == "Community 12" + + +def test_community_header_falls_back_when_no_name(): + assert _community_header(7, None) == "Community 7" + assert _community_header(7, "") == "Community 7" + + +def test_community_header_sanitizes_name(): + # control characters in an LLM-derived name are stripped (F-010) + out = _community_header(3, "Pay\x00ments\x1b[31m") + assert out.startswith("Community 3 — ") + assert "\x00" not in out and "\x1b" not in out