Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
243 changes: 243 additions & 0 deletions src/services/api/openaiShim.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -879,6 +879,249 @@ test('preserves usage from final OpenAI stream chunk with empty choices', async
expect(usageEvent?.usage?.output_tokens).toBe(45)
})

test('preserves usage from early stream chunk when finish_reason chunk has no usage', async () => {
// Some providers emit usage in an early chunk (before the finish_reason chunk).
// lastSeenUsage accumulation ensures this is captured and forwarded in message_delta.
globalThis.fetch = (async (_input, init) => {
const chunks = makeStreamChunks([
{
id: 'chatcmpl-early-usage',
object: 'chat.completion.chunk',
model: 'fake-model',
choices: [
{
index: 0,
delta: { role: 'assistant', content: 'hi' },
finish_reason: null,
},
],
usage: {
prompt_tokens: 77,
completion_tokens: 11,
total_tokens: 88,
},
},
// Stop chunk — no usage field
{
id: 'chatcmpl-early-usage',
object: 'chat.completion.chunk',
model: 'fake-model',
choices: [
{
index: 0,
delta: {},
finish_reason: 'stop',
},
],
},
])
return makeSseResponse(chunks)
}) as FetchType

const client = createOpenAIShimClient({}) as OpenAIShimClient

const result = await client.beta.messages
.create({
model: 'fake-model',
messages: [{ role: 'user', content: 'hello' }],
max_tokens: 64,
stream: true,
})
.withResponse()

const events: Array<Record<string, unknown>> = []
for await (const event of result.data) {
events.push(event)
}

const usageEvent = events.find(
event => event.type === 'message_delta' && typeof event.usage === 'object' && event.usage !== null,
) as { usage?: { input_tokens?: number; output_tokens?: number } } | undefined

// The early-chunk usage should be carried forward to the message_delta
expect(usageEvent).toBeDefined()
expect(usageEvent?.usage?.input_tokens).toBe(77)
expect(usageEvent?.usage?.output_tokens).toBe(11)
})

test('trailing empty-choices chunk fires when stop chunk used lastSeenUsage fallback', async () => {
// Regression test: previously the stop chunk with lastSeenUsage incorrectly set
// hasEmittedFinalUsage = true, suppressing the trailing chunk. Now only a stop
// chunk with real chunkUsage should set hasEmittedFinalUsage.
globalThis.fetch = (async () => {
const chunks = makeStreamChunks([
// Early chunk with usage
{
id: 'chatcmpl-trailing-fires',
object: 'chat.completion.chunk',
model: 'fake-model',
choices: [{ index: 0, delta: { content: 'hi' }, finish_reason: null }],
usage: { prompt_tokens: 55, completion_tokens: 9, total_tokens: 64 },
},
// Stop chunk — no chunkUsage, will use lastSeenUsage fallback
{
id: 'chatcmpl-trailing-fires',
object: 'chat.completion.chunk',
model: 'fake-model',
choices: [{ index: 0, delta: {}, finish_reason: 'stop' }],
},
// Trailing empty-choices chunk with real definitive counts
{
id: 'chatcmpl-trailing-fires',
object: 'chat.completion.chunk',
model: 'fake-model',
choices: [],
usage: { prompt_tokens: 55, completion_tokens: 9, total_tokens: 64 },
},
])
return makeSseResponse(chunks)
}) as FetchType

const client = createOpenAIShimClient({}) as OpenAIShimClient

const result = await client.beta.messages
.create({
model: 'fake-model',
messages: [{ role: 'user', content: 'hi' }],
max_tokens: 32,
stream: true,
})
.withResponse()

const events: Array<Record<string, unknown>> = []
for await (const event of result.data) {
events.push(event)
}

const messageDeltaEvents = events.filter(e => e.type === 'message_delta') as Array<{
usage?: { input_tokens?: number; output_tokens?: number }
}>
// Two message_delta events: stop chunk (lastSeenUsage fallback) + trailing chunk (real counts).
expect(messageDeltaEvents).toHaveLength(2)
// The last (trailing) message_delta carries the definitive final counts.
const lastDelta = messageDeltaEvents[messageDeltaEvents.length - 1]
expect(lastDelta.usage?.input_tokens).toBe(55)
expect(lastDelta.usage?.output_tokens).toBe(9)
})

test('trailing empty-choices chunk with different real usage supersedes lastSeenUsage fallback', async () => {
// Confirms the trailing chunk's values are used as the definitive answer when
// the stop chunk only had provisional lastSeenUsage (e.g. completion_tokens grew).
globalThis.fetch = (async () => {
const chunks = makeStreamChunks([
// Early chunk with provisional usage
{
id: 'chatcmpl-trailing-real',
object: 'chat.completion.chunk',
model: 'fake-model',
choices: [{ index: 0, delta: { content: 'hi' }, finish_reason: null }],
usage: { prompt_tokens: 10, completion_tokens: 5, total_tokens: 15 },
},
// Stop chunk — no chunkUsage, uses lastSeenUsage (10/5) as fallback
{
id: 'chatcmpl-trailing-real',
object: 'chat.completion.chunk',
model: 'fake-model',
choices: [{ index: 0, delta: {}, finish_reason: 'stop' }],
},
// Trailing empty-choices chunk with REAL definitive counts (completion grew to 20)
{
id: 'chatcmpl-trailing-real',
object: 'chat.completion.chunk',
model: 'fake-model',
choices: [],
usage: { prompt_tokens: 10, completion_tokens: 20, total_tokens: 30 },
},
])
return makeSseResponse(chunks)
}) as FetchType

const client = createOpenAIShimClient({}) as OpenAIShimClient

const result = await client.beta.messages
.create({
model: 'fake-model',
messages: [{ role: 'user', content: 'hello' }],
max_tokens: 64,
stream: true,
})
.withResponse()

const events: Array<Record<string, unknown>> = []
for await (const event of result.data) {
events.push(event)
}

const messageDeltaEvents = events.filter(e => e.type === 'message_delta') as Array<{
usage?: { input_tokens?: number; output_tokens?: number }
}>

// Two message_delta events: stop chunk (fallback 10/5) + trailing chunk (real 10/20).
expect(messageDeltaEvents).toHaveLength(2)
// The last message_delta carries the definitive final counts from the trailing chunk.
const lastDelta = messageDeltaEvents[messageDeltaEvents.length - 1]
expect(lastDelta.usage?.input_tokens).toBe(10)
expect(lastDelta.usage?.output_tokens).toBe(20)
})

test('does not emit a second message_delta when stop chunk has real chunkUsage', async () => {
// When the stop chunk itself carries chunkUsage, hasEmittedFinalUsage is set to true
// and any subsequent trailing empty-choices chunk must be suppressed.
globalThis.fetch = (async () => {
const chunks = makeStreamChunks([
{
id: 'chatcmpl-no-double',
object: 'chat.completion.chunk',
model: 'fake-model',
choices: [{ index: 0, delta: { content: 'hi' }, finish_reason: null }],
},
// Stop chunk WITH real chunkUsage — hasEmittedFinalUsage should be set
{
id: 'chatcmpl-no-double',
object: 'chat.completion.chunk',
model: 'fake-model',
choices: [{ index: 0, delta: {}, finish_reason: 'stop' }],
usage: { prompt_tokens: 55, completion_tokens: 9, total_tokens: 64 },
},
// Trailing empty-choices chunk — must be suppressed (stop chunk had real usage)
{
id: 'chatcmpl-no-double',
object: 'chat.completion.chunk',
model: 'fake-model',
choices: [],
usage: { prompt_tokens: 55, completion_tokens: 9, total_tokens: 64 },
},
])
return makeSseResponse(chunks)
}) as FetchType

const client = createOpenAIShimClient({}) as OpenAIShimClient

const result = await client.beta.messages
.create({
model: 'fake-model',
messages: [{ role: 'user', content: 'hi' }],
max_tokens: 32,
stream: true,
})
.withResponse()

const events: Array<Record<string, unknown>> = []
for await (const event of result.data) {
events.push(event)
}

const messageDeltaEvents = events.filter(e => e.type === 'message_delta') as Array<{
usage?: { input_tokens?: number; output_tokens?: number }
}>

// Only one message_delta: the stop chunk had real chunkUsage, so hasEmittedFinalUsage
// is set and the trailing chunk is correctly suppressed.
expect(messageDeltaEvents).toHaveLength(1)
expect(messageDeltaEvents[0].usage?.input_tokens).toBe(55)
expect(messageDeltaEvents[0].usage?.output_tokens).toBe(9)
})

test('uses max_tokens instead of max_completion_tokens for local providers', async () => {
process.env.OPENAI_BASE_URL = 'http://localhost:11434/v1'

Expand Down
33 changes: 31 additions & 2 deletions src/services/api/openaiShim.ts
Original file line number Diff line number Diff line change
Expand Up @@ -1274,6 +1274,10 @@ async function* openaiStreamToAnthropic(
let lastStopReason: 'tool_use' | 'max_tokens' | 'end_turn' | null = null
let hasEmittedFinalUsage = false
let hasProcessedFinishReason = false
// Accumulate the most recent non-undefined usage seen across all chunks.
// Some providers send usage in an early chunk before the finish_reason chunk;
// without this accumulator the message_delta would be emitted without usage.
let lastSeenUsage: Partial<AnthropicUsage> | undefined
const streamState = createStreamState()
let bufferedRawToolCallsText: string | null = null

Expand Down Expand Up @@ -1476,6 +1480,11 @@ async function* openaiStreamToAnthropic(
}

const chunkUsage = convertChunkUsage(chunk.usage)
// Keep a running record of the most recent usage seen across all chunks.
// Some providers emit usage in an early chunk before the finish_reason
// chunk arrives, so chunkUsage may be undefined at stop time even though
// we already observed real usage data earlier in the stream.
if (chunkUsage) lastSeenUsage = chunkUsage

for (const choice of chunk.choices ?? []) {
const delta = choice.delta
Expand Down Expand Up @@ -1755,6 +1764,14 @@ async function* openaiStreamToAnthropic(
}
lastStopReason = stopReason

// Only attach usage when this stop chunk actually carries it.
// Do NOT fall back to lastSeenUsage here: some providers emit an
// early/provisional usage chunk followed by a separate trailing
// empty-choices chunk with the real final totals. Emitting
// lastSeenUsage at stop time AND again at the trailing chunk would
// double-count both in addToTotalSessionCost. The in-loop post check
// below handles trailing chunks; the post-stream fallback handles
// providers that send no trailing chunk at all.
yield {
type: 'message_delta',
delta: { stop_reason: stopReason, stop_sequence: null },
Expand All @@ -1768,14 +1785,14 @@ async function* openaiStreamToAnthropic(

if (
!hasEmittedFinalUsage &&
chunkUsage &&
(chunkUsage ?? lastSeenUsage) &&
(chunk.choices?.length ?? 0) === 0 &&
lastStopReason !== null
) {
yield {
type: 'message_delta',
delta: { stop_reason: lastStopReason, stop_sequence: null },
usage: chunkUsage,
usage: (chunkUsage ?? lastSeenUsage)!,
}
hasEmittedFinalUsage = true
}
Expand All @@ -1785,6 +1802,18 @@ async function* openaiStreamToAnthropic(
reader.releaseLock()
}

// Post-stream fallback: if a provider sent usage only in an early chunk
// (not the stop chunk, not a trailing empty-choices chunk), we still need
// to emit exactly one message_delta with that accumulated usage so the
// status-line and session-cost accounting see it.
if (!hasEmittedFinalUsage && lastSeenUsage && lastStopReason !== null) {
yield {
type: 'message_delta',
delta: { stop_reason: lastStopReason, stop_sequence: null },
usage: lastSeenUsage,
}
}

const stats = getStreamStats(streamState)
if (stats.totalChunks > 0) {
logForDebugging(
Expand Down
11 changes: 8 additions & 3 deletions src/tools/AgentTool/built-in/statuslineSetup.ts
Original file line number Diff line number Diff line change
Expand Up @@ -55,14 +55,19 @@ How to use the statusLine command:
"total_input_tokens": number, // Total input tokens used in session (cumulative)
"total_output_tokens": number, // Total output tokens used in session (cumulative)
"context_window_size": number, // Context window size for current model (e.g., 200000)
"current_usage": { // Token usage from last API call (null if no messages yet)
"current_usage": { // Token usage from last API call.
// null when: no messages yet, OR the active provider does not
// report token usage (e.g. providers that strip stream_options
// such as MiMo/OpenGateway).
"input_tokens": number, // Input tokens for current context
"output_tokens": number, // Output tokens generated
"cache_creation_input_tokens": number, // Tokens written to cache
"cache_read_input_tokens": number // Tokens read from cache
} | null,
"used_percentage": number | null, // Pre-calculated: % of context used (0-100), null if no messages yet
"remaining_percentage": number | null // Pre-calculated: % of context remaining (0-100), null if no messages yet
"used_percentage": number | null, // Pre-calculated: % of context used (0-100).
// null when: no messages yet, OR provider does not report usage.
"remaining_percentage": number | null // Pre-calculated: % of context remaining (0-100).
// null when: no messages yet, OR provider does not report usage.
},
"rate_limits": { // Optional: Claude.ai subscription usage limits. Only present for subscribers after first API response.
"five_hour": { // Optional: 5-hour session limit (may be absent)
Expand Down
Loading