fix(llm): force non-streaming on OpenAI-compatible calls (#1223)

Some OpenAI-compatible gateways default to SSE streaming when `stream` is omitted,
but graphify always reads the result as a single resp.choices[0]. The call would
then fail against those gateways. Pass `stream: False` explicitly.

Ported from PR #1482 by @jiangyq9 (covers the extraction dispatch path,
_call_openai_compat). Maintainer fix on top: applied the same `stream: False` to
the second OpenAI-compatible call site, _call_llm, which feeds the --dedup-llm
tiebreaker and had the identical bug.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
jiangyq9
2026-06-27 10:19:15 +01:00
committed by safishamsi
co-authored by Claude Opus 4.8
parent 9b49bfd9eb
commit ff47316b8a
2 changed files with 17 additions and 0 deletions
+6
View File
@@ -912,6 +912,7 @@ def _call_openai_compat(
{"role": "user", "content": _openai_content(user_message, images or [])},
],
"max_completion_tokens": max_completion_tokens,
"stream": False,
}
if temperature is not None:
kwargs["temperature"] = temperature
@@ -1968,6 +1969,11 @@ def _call_llm(
"model": mdl,
"messages": [{"role": "user", "content": prompt}],
"max_completion_tokens": max_tokens,
# Force a single non-streamed response: some OpenAI-compatible gateways
# default to SSE streaming when `stream` is omitted, but the result here
# is always read as resp.choices[0]. Same fix as _call_openai_compat
# (#1223) — this path feeds the --dedup-llm tiebreaker.
"stream": False,
}
temperature = _resolve_temperature(cfg.get("temperature", 0), mdl)
if temperature is not None:
+11
View File
@@ -533,6 +533,17 @@ def test_non_ollama_backend_gets_no_num_ctx_extra_body(monkeypatch):
assert eb is None or "options" not in eb, "non-ollama backends must not get num_ctx injection"
def test_openai_compat_forces_non_streaming_response(monkeypatch):
captured = _install_capturing_openai(monkeypatch)
llm._call_openai_compat(
"https://gateway.example/v1", "sk-test", "gpt-4.1-mini",
"u", temperature=0, max_completion_tokens=8192, backend="openai",
)
assert captured["stream"] is False
# ---------------------------------------------------------------------------
# Custom-provider extra_body: lets providers.json route around the moonshot-only
# default. Self-hosted Qwen3 served by vLLM needs