diff --git a/CHANGELOG.md b/CHANGELOG.md index 1255f61..9a2df0c 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,7 @@ Full release notes with details on each version: [GitHub Releases](https://githu ## Unreleased +- Feat: the MCP server can serve many projects from one process via an optional `project_path` on every tool (#1594, thanks @joanfgarcia). Omit it and nothing changes — the server answers against the graph it was started with. Pass an absolute `project_path` and that call is routed to `//graph.json` instead, with its own mtime+size hot-reload, so one stdio/HTTP server backs a whole workspace of repos. Graphs load lazily and cache per resolved path; a missing/corrupt project graph is a tool error, not a process exit, and the server starts even when its default graph is absent. Backward-compatible and additive. - Fix: Swift singleton cached into a local var now resolves later calls (#1604, thanks @jerryliurui). `let x = NetworkManager.shared` followed by `x.fetchData()` on a subsequent line produced zero call edges — local `let`/`var` bindings inside method bodies weren't typed (only class-level properties and params were), and a static-member init (`Type.shared`, a navigation expression) wasn't recognized even where locals were typed. Method-body locals are now typed from both constructor (`Type()`) and static-member (`Type.shared`) initializers, so `x.method()` resolves to the receiver type via the existing single-definition guard. This singleton-into-local idiom is one of the most common Swift call patterns. - Fix: the skill's Python-interpreter detection now accepts Homebrew `python@3.x` paths (#1586, thanks @SUDARSHANCHAUDHARI). The shebang allowlist rejected any path with a character outside `[a-zA-Z0-9/_.-]`, but Homebrew installs versioned Python under `python@3.13`, so a valid interpreter containing `@` was skipped and detection fell through to a bare `python3` that lacked graphify (every step then failed with `ModuleNotFoundError`). `@` is now allowed across all skill variants (matching the #473 hooks.py fix); injection characters are still rejected. - Fix: `graphify merge-graphs` no longer crashes on inputs that disagree on graph type (#1606, thanks @AdrianRusan). Per-repo `graph.json` files don't always share the same `directed` / `multigraph` flags, and `compose` requires one uniform type, so a mixed set raised an unhandled `NetworkXError`. All inputs are now normalized to a plain undirected graph (which the cross-repo merged view already is) before composing. diff --git a/graphify/serve.py b/graphify/serve.py index c1c16a3..2203c1a 100644 --- a/graphify/serve.py +++ b/graphify/serve.py @@ -158,7 +158,7 @@ def _compute_idf(G: nx.Graph, terms: list[str]) -> dict[str, float]: Common terms like 'error' or 'exception' that match hundreds of nodes get low weights; rare identifiers like 'FooBarService' get high weights. Cache is stored on the graph object itself so it auto-invalidates when - _maybe_reload() replaces G with a new object. + a hot-reload replaces G with a new object. """ cache: dict[str, float] = G.graph.setdefault("_idf_cache", {}) N = G.number_of_nodes() or 1 @@ -210,7 +210,7 @@ def _node_search_text(data: dict, nid: str) -> str: def _get_trigram_index(G: nx.Graph) -> dict: """Lazily build and cache a trigram -> node-position postings map on the graph. - Cached on `G.graph` so it auto-invalidates when _maybe_reload() swaps in a + Cached on `G.graph` so it auto-invalidates when a hot-reload swaps in a fresh graph object, exactly like `_idf_cache`. `set_cache` memoizes per-trigram id-sets across queries within one graph generation. """ @@ -737,56 +737,81 @@ def _build_server(graph_path: str): except ImportError as e: raise ImportError('mcp not installed. Run: pip install "graphifyy[mcp]"') from e - G = _load_graph(graph_path) - communities = _communities_from_graph(G) - # Build the trigram query index eagerly so the first query doesn't pay the - # one-time build (a few seconds on a large graph). serve exists to answer - # queries, so absorbing the cost at startup beats spiking a random first - # request; it's cached on G.graph, so queries just reuse it. - _get_trigram_index(G) + from graphify import paths as _paths - # Hot-reload state: mtime+size key lets us detect graph.json changes without - # polling. Initialised from the file stat at startup so the first tool call - # never triggers a redundant reload. - _reload_lock = threading.Lock() - try: - _s = Path(graph_path).stat() - _reload_state: dict = {"mtime_ns": _s.st_mtime_ns, "size": _s.st_size} - except FileNotFoundError: - _reload_state = {"mtime_ns": 0, "size": -1} + # Per-graph context cache: resolved graph.json path -> {key, G, communities}. + # The server's default graph is just the first entry; a tool call carrying a + # project_path adds its own. Routing every graph through one cache means the + # eager trigram index and the mtime+size hot-reload behave identically for + # the default graph and for any project graph. + _default_graph_path = graph_path + _ctx_lock = threading.Lock() + _ctx_cache: dict[str, dict] = {} - def _maybe_reload() -> None: - nonlocal G, communities + def _load_ctx(path: str): + """Return (G, communities) for a graph.json path, reusing a cached + context until the file's (mtime, size) changes and then transparently + rebuilding it. Unlike ``_load_graph`` it never exits the process on a + missing/corrupt file — it raises, so a bad project_path surfaces as a + tool error instead of killing a server that is happily serving other + projects.""" try: - s = Path(graph_path).stat() + s = Path(path).stat() key = (s.st_mtime_ns, s.st_size) except FileNotFoundError: - return - if key == (_reload_state["mtime_ns"], _reload_state["size"]): - return - with _reload_lock: + raise FileNotFoundError(f"graph.json not found: {path}") + ent = _ctx_cache.get(path) + if ent is not None and ent["key"] == key: + return ent["G"], ent["communities"] + with _ctx_lock: + ent = _ctx_cache.get(path) + if ent is not None and ent["key"] == key: + return ent["G"], ent["communities"] # another thread built it try: - s = Path(graph_path).stat() - key = (s.st_mtime_ns, s.st_size) - except FileNotFoundError: - return - if key == (_reload_state["mtime_ns"], _reload_state["size"]): - return # another thread already reloaded - try: - new_G = _load_graph(graph_path) - except SystemExit: - return # keep serving stale graph on transient read error - _get_trigram_index(new_G) # warm before exposing, so the first - # post-reload query is fast too (same rationale as startup) - G = new_G - communities = _communities_from_graph(new_G) - _reload_state["mtime_ns"], _reload_state["size"] = key + new_G = _load_graph(path) + except SystemExit as e: # _load_graph exits on missing/corrupt file + raise RuntimeError(f"could not load graph.json at {path}") from e + # Warm the trigram index before exposing the graph so the first query + # against it is fast (same rationale as the original startup warm-up). + _get_trigram_index(new_G) + comm = _communities_from_graph(new_G) + _ctx_cache[path] = {"key": key, "G": new_G, "communities": comm} + return new_G, comm + + def _resolve_graph_path(project_path) -> str: + """Map an optional project_path to a concrete graph.json path. ``None`` + keeps the server's default graph (backward-compatible); a project_path + resolves to ``//graph.json``, honouring the + GRAPHIFY_OUT override so worktree/shared-output setups keep working.""" + if not project_path: + return _default_graph_path + return str(Path(project_path) / _paths.GRAPHIFY_OUT / "graph.json") + + # Active per-request context, rebound by _select_graph() and read by the tool + # handlers below. No lock needed on the hot path: _select_graph and the + # handler run in one synchronous stretch of each call_tool coroutine (no + # await between them), so a concurrent call never observes a half-applied + # swap. + active_graph_path = _default_graph_path + try: + G, communities = _load_ctx(_default_graph_path) + except (FileNotFoundError, RuntimeError): + # No default graph at startup → run as a pure multi-project server. Tools + # then require project_path; a call without one gets a clear error rather + # than the process refusing to start (which is what _load_graph would do). + G, communities = None, {} + + def _select_graph(project_path) -> None: + nonlocal G, communities, active_graph_path + path = _resolve_graph_path(project_path) + G, communities = _load_ctx(path) + active_graph_path = path server = Server("graphify") @server.list_tools() async def list_tools() -> list[types.Tool]: - return [ + _tools = [ types.Tool( name="query_graph", description="Search the knowledge graph using BFS or DFS. Returns relevant nodes and edges as text context.", @@ -907,6 +932,20 @@ def _build_server(graph_path: str): }, ), ] + # Multi-project support: every tool accepts an optional project_path. + # Injected here (rather than repeated in 11 literal schemas) so the set + # stays in lockstep as tools are added. Omitting it keeps the historical + # single-graph behaviour, so this is purely additive for existing callers. + for _t in _tools: + _t.inputSchema.setdefault("properties", {})["project_path"] = { + "type": "string", + "description": ( + "Absolute path to a project directory containing " + "graphify-out/graph.json. Optional — defaults to the graph " + "this server was started with." + ), + } + return _tools def _tool_query_graph(arguments: dict) -> str: import time as _time @@ -928,7 +967,7 @@ def _build_server(graph_path: str): querylog.log_query( kind="mcp_query", question=question, - corpus=str(graph_path), + corpus=str(active_graph_path), result=result, mode=mode, depth=depth, @@ -1171,7 +1210,7 @@ def _build_server(graph_path: str): } def _load_community_labels() -> dict[int, str]: - labels_path = Path(graph_path).parent / ".graphify_labels.json" + labels_path = Path(active_graph_path).parent / ".graphify_labels.json" if labels_path.exists(): try: return {int(k): v for k, v in json.loads(labels_path.read_text(encoding="utf-8")).items()} @@ -1192,10 +1231,10 @@ def _build_server(graph_path: str): @server.read_resource() async def read_resource(uri: AnyUrl) -> str: - _maybe_reload() + _select_graph(None) # resources read the server's default graph uri_str = str(uri) if uri_str == "graphify://report": - report_path = Path(graph_path).parent / "GRAPH_REPORT.md" + report_path = Path(active_graph_path).parent / "GRAPH_REPORT.md" if report_path.exists(): return report_path.read_text(encoding="utf-8") return "GRAPH_REPORT.md not found. Run graphify extract first." @@ -1244,11 +1283,13 @@ def _build_server(graph_path: str): @server.call_tool() async def call_tool(name: str, arguments: dict) -> list[types.TextContent]: - _maybe_reload() + arguments = dict(arguments or {}) + project_path = arguments.pop("project_path", None) handler = _handlers.get(name) if not handler: return [types.TextContent(type="text", text=f"Unknown tool: {name}")] try: + _select_graph(project_path) # bind G/communities to the target graph return [types.TextContent(type="text", text=handler(arguments))] except Exception as exc: return [types.TextContent(type="text", text=f"Error executing {name}: {exc}")] diff --git a/tests/test_serve_http.py b/tests/test_serve_http.py index e582053..7d54290 100644 --- a/tests/test_serve_http.py +++ b/tests/test_serve_http.py @@ -170,6 +170,75 @@ def test_tools_list_over_http(tmp_path): assert {"query_graph", "get_node", "graph_stats"} <= names +def _project_with_graph(tmp_path, node_count: int) -> str: + """Create ``/graphify-out/graph.json`` and return the project dir.""" + proj = tmp_path / "proj" + (proj / "graphify-out").mkdir(parents=True) + graph = { + "directed": True, + "nodes": [{"id": f"n{i}", "label": f"N{i}", "community": 0} for i in range(node_count)], + "edges": [], + } + (proj / "graphify-out" / "graph.json").write_text(json.dumps(graph), encoding="utf-8") + return str(proj) + + +def _init_session(client) -> dict: + init = client.post("/mcp", headers=_MCP_HEADERS, json=_INIT_BODY) + assert init.status_code == 200 + headers = {**_MCP_HEADERS, "mcp-session-id": init.headers.get("mcp-session-id")} + client.post("/mcp", headers=headers, json={"jsonrpc": "2.0", "method": "notifications/initialized"}) + return headers + + +def _call_tool(client, headers, name, arguments, rid) -> str: + resp = client.post("/mcp", headers=headers, json={ + "jsonrpc": "2.0", "id": rid, "method": "tools/call", + "params": {"name": name, "arguments": arguments}, + }) + assert resp.status_code == 200 + return resp.json()["result"]["content"][0]["text"] + + +def test_project_path_is_optional_on_every_tool(tmp_path): + """Multi-project support is additive: every tool gains an optional + project_path, and none of them makes it required.""" + app = serve_mod._build_http_app(_graph_file(tmp_path), json_response=True) + with _client(app) as client: + headers = _init_session(client) + resp = client.post("/mcp", headers=headers, + json={"jsonrpc": "2.0", "id": 2, "method": "tools/list", "params": {}}) + for tool in resp.json()["result"]["tools"]: + props = tool["inputSchema"].get("properties", {}) + assert "project_path" in props, f"{tool['name']} missing project_path" + assert "project_path" not in tool["inputSchema"].get("required", []) + + +def test_project_path_routes_to_that_projects_graph(tmp_path): + """One running server answers against the default graph when project_path is + omitted, and against a project's own graph when it is supplied.""" + proj = _project_with_graph(tmp_path, node_count=3) # default graph has 2 nodes + app = serve_mod._build_http_app(_graph_file(tmp_path), json_response=True) + with _client(app) as client: + headers = _init_session(client) + assert "Nodes: 2" in _call_tool(client, headers, "graph_stats", {}, rid=2) + assert "Nodes: 3" in _call_tool(client, headers, "graph_stats", {"project_path": proj}, rid=3) + # Falling back to the default afterwards still works (no state leak). + assert "Nodes: 2" in _call_tool(client, headers, "graph_stats", {}, rid=4) + + +def test_bad_project_path_errors_without_killing_server(tmp_path): + """A missing project graph is a tool error, not a process exit — the server + keeps serving the default graph.""" + app = serve_mod._build_http_app(_graph_file(tmp_path), json_response=True) + with _client(app) as client: + headers = _init_session(client) + bad = _call_tool(client, headers, "graph_stats", + {"project_path": str(tmp_path / "does-not-exist")}, rid=2) + assert "not found" in bad.lower() + assert "Nodes: 2" in _call_tool(client, headers, "graph_stats", {}, rid=3) + + def test_stateless_mode_initialize(tmp_path): app = serve_mod._build_http_app(_graph_file(tmp_path), stateless=True, json_response=True) with _client(app) as client: