fix(llm): force non-streaming on OpenAI-compatible calls (#1223)
Some OpenAI-compatible gateways default to SSE streaming when `stream` is omitted, but graphify always reads the result as a single resp.choices[0]. The call would then fail against those gateways. Pass `stream: False` explicitly. Ported from PR #1482 by @jiangyq9 (covers the extraction dispatch path, _call_openai_compat). Maintainer fix on top: applied the same `stream: False` to the second OpenAI-compatible call site, _call_llm, which feeds the --dedup-llm tiebreaker and had the identical bug. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
committed by
safishamsi
co-authored by
Claude Opus 4.8
parent
9b49bfd9eb
commit
ff47316b8a
@@ -912,6 +912,7 @@ def _call_openai_compat(
|
||||
{"role": "user", "content": _openai_content(user_message, images or [])},
|
||||
],
|
||||
"max_completion_tokens": max_completion_tokens,
|
||||
"stream": False,
|
||||
}
|
||||
if temperature is not None:
|
||||
kwargs["temperature"] = temperature
|
||||
@@ -1968,6 +1969,11 @@ def _call_llm(
|
||||
"model": mdl,
|
||||
"messages": [{"role": "user", "content": prompt}],
|
||||
"max_completion_tokens": max_tokens,
|
||||
# Force a single non-streamed response: some OpenAI-compatible gateways
|
||||
# default to SSE streaming when `stream` is omitted, but the result here
|
||||
# is always read as resp.choices[0]. Same fix as _call_openai_compat
|
||||
# (#1223) — this path feeds the --dedup-llm tiebreaker.
|
||||
"stream": False,
|
||||
}
|
||||
temperature = _resolve_temperature(cfg.get("temperature", 0), mdl)
|
||||
if temperature is not None:
|
||||
|
||||
@@ -533,6 +533,17 @@ def test_non_ollama_backend_gets_no_num_ctx_extra_body(monkeypatch):
|
||||
assert eb is None or "options" not in eb, "non-ollama backends must not get num_ctx injection"
|
||||
|
||||
|
||||
def test_openai_compat_forces_non_streaming_response(monkeypatch):
|
||||
captured = _install_capturing_openai(monkeypatch)
|
||||
|
||||
llm._call_openai_compat(
|
||||
"https://gateway.example/v1", "sk-test", "gpt-4.1-mini",
|
||||
"u", temperature=0, max_completion_tokens=8192, backend="openai",
|
||||
)
|
||||
|
||||
assert captured["stream"] is False
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Custom-provider extra_body: lets providers.json route around the moonshot-only
|
||||
# default. Self-hosted Qwen3 served by vLLM needs
|
||||
|
||||
Reference in New Issue
Block a user