Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
18 changes: 18 additions & 0 deletions scripts/generate-openclaw-config.py
Original file line number Diff line number Diff line change
Expand Up @@ -423,6 +423,24 @@ def build_config(env: dict | None = None) -> dict:
setup, inference_compat, openclaw_plugins, openclaw_plugin_ids
)

# Ollama's OpenAI-compatible /v1/chat/completions stream omits the
# `usage` chunk by default; OpenAI clients have to send
# `stream_options.include_usage: true` to receive it. OpenClaw gates
# that request flag on `model.compat.supportsUsageInStreaming`
# (src/agents/openai-transport-stream.ts) and its Ollama extension
# only opts in when its own detector recognises the endpoint as
# Ollama. NemoClaw routes ollama-local traffic via the standardised
# `https://inference.local/v1` URL through the OpenShell gateway, so
# the upstream detector misses it and the TUI token counter stays
# `?` indefinitely (#2747). Set the flag here so the request is sent
# with `stream_options.include_usage: true` regardless of how
# OpenClaw resolves the provider id. Mirrors the LM Studio extension
# workaround (`withLmstudioUsageCompat` in
# extensions/lmstudio/src/stream.ts). Keep the set of provider keys
# in sync with `_bundled_provider_plugins["ollama"]` below.
if provider_key in {"ollama", "ollama-local"}:
inference_compat.setdefault("supportsUsageInStreaming", True)

msg_channels = json.loads(
base64.b64decode(
env.get("NEMOCLAW_MESSAGING_CHANNELS_B64", "W10=") or "W10="
Expand Down
64 changes: 64 additions & 0 deletions test/generate-openclaw-config.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -414,6 +414,70 @@ describe("generate-openclaw-config.py: config generation", () => {
});
});

// #2747: Ollama's OpenAI-compatible streaming API omits the usage chunk
// unless `stream_options.include_usage` is set on the request. OpenClaw
// gates that on `model.compat.supportsUsageInStreaming`. NemoClaw routes
// ollama-local through the standardised `inference.local` URL, which
// OpenClaw's own Ollama detector does not recognise — so we force the
// flag here. Cloud providers and other local backends must not be
// affected.
it("enables supportsUsageInStreaming for Ollama provider keys (#2747)", () => {
for (const providerKey of ["ollama", "ollama-local"]) {
const config = runConfigScript({
NEMOCLAW_MODEL: "qwen2.5:7b",
NEMOCLAW_PROVIDER_KEY: providerKey,
NEMOCLAW_PRIMARY_MODEL_REF: "qwen2.5:7b",
NEMOCLAW_INFERENCE_BASE_URL: "https://inference.local/v1",
NEMOCLAW_INFERENCE_API: "openai-completions",
});
const model = config.models.providers[providerKey].models[0];
expect(model.compat?.supportsUsageInStreaming).toBe(true);
}
});

it("does not enable supportsUsageInStreaming for non-Ollama providers (#2747)", () => {
const cases = [
{ NEMOCLAW_PROVIDER_KEY: "openai", NEMOCLAW_INFERENCE_BASE_URL: "https://api.openai.com/v1" },
{
NEMOCLAW_PROVIDER_KEY: "anthropic",
NEMOCLAW_INFERENCE_BASE_URL: "https://api.anthropic.com",
},
{ NEMOCLAW_PROVIDER_KEY: "vllm", NEMOCLAW_INFERENCE_BASE_URL: "https://inference.local/v1" },
{
NEMOCLAW_PROVIDER_KEY: "nim-local",
NEMOCLAW_INFERENCE_BASE_URL: "https://inference.local/v1",
},
];

for (const envCase of cases) {
const config = runConfigScript({
NEMOCLAW_MODEL: "test-model",
NEMOCLAW_PRIMARY_MODEL_REF: "test-ref",
NEMOCLAW_INFERENCE_API: "openai-completions",
...envCase,
});
const model = config.models.providers[envCase.NEMOCLAW_PROVIDER_KEY].models[0];
expect(model.compat?.supportsUsageInStreaming).toBeUndefined();
}
});

// If a future model-specific-setup manifest declares
// supportsUsageInStreaming explicitly, that decision should win over our
// ollama-keyed default — including when a manifest opts the flag *off*.
it("respects existing supportsUsageInStreaming from inference compat (#2747)", () => {
const config = runConfigScript({
NEMOCLAW_MODEL: "qwen2.5:7b",
NEMOCLAW_PROVIDER_KEY: "ollama",
NEMOCLAW_PRIMARY_MODEL_REF: "qwen2.5:7b",
NEMOCLAW_INFERENCE_BASE_URL: "https://inference.local/v1",
NEMOCLAW_INFERENCE_API: "openai-completions",
NEMOCLAW_INFERENCE_COMPAT_B64: Buffer.from(
JSON.stringify({ supportsUsageInStreaming: false }),
).toString("base64"),
});
expect(config.models.providers.ollama.models[0].compat.supportsUsageInStreaming).toBe(false);
});

it("does not activate the OpenClaw Kimi setup for non-matching routes", () => {
const cases = [
{ NEMOCLAW_MODEL: "deepseek-ai/DeepSeek-V4-Flash" },
Expand Down
Loading