diff --git a/CHANGELOG.md b/CHANGELOG.md index 1ec1cf1..42fe175 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,7 @@ Full release notes with details on each version: [GitHub Releases](https://githu ## Unreleased +- Fix: OpenAI-compatible backends (`ollama`, `openai`, `deepseek`, `kimi`) now honour their configured `16384` output-token cap instead of silently falling back to `8192`. The dispatch read a `max_completion_tokens` config key that only `gemini` defines; the others set `max_tokens`, so their advertised cap was dead and deep-mode JSON truncated mid-string (recovered by the adaptive bisect, but noisy and slower). The dispatch now reads either key, and the `openai` config gained an explicit cap. `GRAPHIFY_MAX_OUTPUT_TOKENS` still overrides. (#1365) - Fix: fuzzy dedup no longer over-merges distinct nodes in three cases. (1) Numbered/versioned siblings whose embedded digit runs differ as zero-padding-insensitive multisets (`ADR 0011` vs `ADR 0013`, `3.1 Product Goals` vs `1.1 Product Goals`, `40%+ …` vs `<20% …`) never merge. (2) `rationale`/`document` nodes are file-anchored like code (#1205's reasoning): near-identical docstring/heading boilerplate in parallel files no longer collapses across files, while same-file duplicates still merge. (3) Cross-file labels that share a long prefix but diverge in a distinguishing token (`testing-library jest-native` vs `react-native`) are scored on plain Jaro instead of Jaro-Winkler, so the leading-prefix bonus can no longer fabricate a merge; genuine cross-file duplicates still clear the bar on Jaro alone, and same-file near-duplicates keep Jaro-Winkler. Guards are mirrored into the `--dedup-llm` ambiguous-pair collection. (#1284 thanks @van4oza, #1243) - Fix: every platform's query skill now ships **both** the vocab/IDF query-expansion step and the inline NetworkX fallback. Previously the two capabilities were split across `cli.md` / `cli-inline.md` so no platform got both — Claude had the superior expansion but no CLI-down fallback, while all other platforms had the fallback but the weaker raw-question matcher. The two fragments are merged into one unified `query` reference (and stub) shipped to all hosts; the `query_variant` enum and its coverage-audit exemption are removed (#1325; thanks @LeanderBlume). - Fix: cross-file Java `implements`/`inherits`/`imports` edges no longer orphan onto bare "shadow" nodes when two packages define a same-named type. The referencing file's `import` statement now disambiguates by exact package (FQN) and re-points the edge to the real definition, dropping the orphan stub. Previously `_rewire_unique_stub_nodes` could only repair the globally-unique case, so same-named interfaces (common in large Java codebases — `Handler`, `Service`, interface+impl pairs) left the real definition isolated in its own community (#1318). diff --git a/graphify/llm.py b/graphify/llm.py index c0d294a..391ed28 100644 --- a/graphify/llm.py +++ b/graphify/llm.py @@ -100,6 +100,7 @@ BACKENDS: dict[str, dict] = { "default_model": os.environ.get("OPENAI_MODEL", "gpt-4.1-mini"), "env_key": "OPENAI_API_KEY", "model_env_key": "GRAPHIFY_OPENAI_MODEL", + "max_tokens": 16384, "pricing": {"input": 0.40, "output": 1.60}, # USD per 1M tokens # Default (gpt-4.1-mini) accepts temperature=0. Reasoning models # (o1/o3/o4/gpt-5) reject any explicit temperature and have it omitted @@ -1355,7 +1356,13 @@ def extract_files_direct( user_msg, temperature=_resolve_temperature(cfg.get("temperature", 0), mdl), reasoning_effort=cfg.get("reasoning_effort"), - max_completion_tokens=_resolve_max_tokens(cfg.get("max_completion_tokens", 8192)), + # Honour max_completion_tokens (gemini) or the older max_tokens key + # (ollama/deepseek/kimi/openai) -- most openai-compat configs define the + # latter, so reading only max_completion_tokens silently capped their + # output at the 8192 fallback and truncated deep-mode JSON (#1365). + max_completion_tokens=_resolve_max_tokens( + cfg.get("max_completion_tokens") or cfg.get("max_tokens", 8192) + ), backend=backend, deep_mode=deep_mode, images=image_refs, diff --git a/tests/test_llm_backends.py b/tests/test_llm_backends.py index cc0556a..888545e 100644 --- a/tests/test_llm_backends.py +++ b/tests/test_llm_backends.py @@ -82,6 +82,33 @@ def test_extract_files_direct_routes_gemini_through_openai_compat(tmp_path, monk assert call.call_args.kwargs["max_completion_tokens"] == 16384 +@pytest.mark.parametrize( + "backend, env_key", + [ + ("ollama", "OLLAMA_API_KEY"), + ("deepseek", "DEEPSEEK_API_KEY"), + ("openai", "OPENAI_API_KEY"), + ("kimi", "MOONSHOT_API_KEY"), + ], +) +def test_openai_compat_backends_resolve_full_output_cap(tmp_path, monkeypatch, backend, env_key): + # #1365: these configs define `max_tokens: 16384`, but the dispatch used to + # read only the `max_completion_tokens` key (which only gemini sets), so the + # output cap silently fell back to 8192 and truncated deep-mode JSON. The + # dispatch must resolve their configured 16384. + _clear_backend_env(monkeypatch) + monkeypatch.delenv("GRAPHIFY_MAX_OUTPUT_TOKENS", raising=False) + monkeypatch.setenv(env_key, "test-key") + source = tmp_path / "note.md" + source.write_text("# Architecture\n") + result = {"nodes": [], "edges": [], "hyperedges": [], "input_tokens": 1, "output_tokens": 1} + + with patch("graphify.llm._call_openai_compat", return_value=result) as call: + llm.extract_files_direct([source], backend=backend, root=tmp_path) + + assert call.call_args.kwargs["max_completion_tokens"] == 16384 + + def test_gemini_model_can_be_overridden_by_env(tmp_path, monkeypatch): _clear_backend_env(monkeypatch) monkeypatch.setenv("GOOGLE_API_KEY", "google-key")