Skip to content
Merged
Show file tree
Hide file tree
Changes from 4 commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
150 changes: 150 additions & 0 deletions src/api/providers/__tests__/lite-llm.spec.ts
Original file line number Diff line number Diff line change
Expand Up @@ -719,6 +719,156 @@ describe("LiteLLMHandler", () => {
})
})

describe("reasoning field handling", () => {
it("should yield reasoning chunks from reasoning_content delta", async () => {
const mockStream = {
async *[Symbol.asyncIterator]() {
yield {
choices: [{ delta: { reasoning_content: "Let me think..." } }],
usage: null,
}
yield {
choices: [{ delta: { content: "The answer is 42." } }],
usage: { prompt_tokens: 20, completion_tokens: 10 },
}
},
}

mockCreate.mockReturnValue({
withResponse: vi.fn().mockResolvedValue({ data: mockStream }),
})

const generator = handler.createMessage("system", [{ role: "user", content: "What is the answer?" }])
const results = []
for await (const chunk of generator) {
results.push(chunk)
}

const reasoningChunk = results.find((c) => c.type === "reasoning")
expect(reasoningChunk).toBeDefined()
expect(reasoningChunk).toMatchObject({ type: "reasoning", text: "Let me think..." })

const textChunk = results.find((c) => c.type === "text")
expect(textChunk).toMatchObject({ type: "text", text: "The answer is 42." })
})

it("should yield reasoning chunks from reasoning delta field", async () => {
const mockStream = {
async *[Symbol.asyncIterator]() {
yield {
choices: [{ delta: { reasoning: "Analyzing the problem..." } }],
usage: null,
}
yield {
choices: [{ delta: { content: "Done." } }],
usage: { prompt_tokens: 10, completion_tokens: 5 },
}
},
}

mockCreate.mockReturnValue({
withResponse: vi.fn().mockResolvedValue({ data: mockStream }),
})

const generator = handler.createMessage("system", [{ role: "user", content: "Solve this." }])
const results = []
for await (const chunk of generator) {
results.push(chunk)
}

const reasoningChunk = results.find((c) => c.type === "reasoning")
expect(reasoningChunk).toBeDefined()
expect(reasoningChunk).toMatchObject({ type: "reasoning", text: "Analyzing the problem..." })
})

it("should prefer reasoning_content over reasoning when both are present", async () => {
const mockStream = {
async *[Symbol.asyncIterator]() {
yield {
choices: [
{ delta: { reasoning_content: "from_reasoning_content", reasoning: "from_reasoning" } },
],
usage: { prompt_tokens: 5, completion_tokens: 5 },
}
},
}

mockCreate.mockReturnValue({
withResponse: vi.fn().mockResolvedValue({ data: mockStream }),
})

const generator = handler.createMessage("system", [{ role: "user", content: "Test." }])
const results = []
for await (const chunk of generator) {
results.push(chunk)
}

const reasoningChunks = results.filter((c) => c.type === "reasoning")
expect(reasoningChunks).toHaveLength(1)
expect(reasoningChunks[0]).toMatchObject({ type: "reasoning", text: "from_reasoning_content" })
})

it("should not yield reasoning chunk when reasoning field is present but falsy", async () => {
const mockStream = {
async *[Symbol.asyncIterator]() {
yield {
choices: [{ delta: { reasoning_content: undefined } }],
usage: null,
}
yield {
choices: [{ delta: { reasoning: "" } }],
usage: null,
}
yield {
choices: [{ delta: { content: "Hello" } }],
usage: { prompt_tokens: 5, completion_tokens: 5 },
}
},
}

mockCreate.mockReturnValue({
withResponse: vi.fn().mockResolvedValue({ data: mockStream }),
})

const generator = handler.createMessage("system", [{ role: "user", content: "Hi" }])
const results = []
for await (const chunk of generator) {
results.push(chunk)
}

const reasoningChunks = results.filter((c) => c.type === "reasoning")
expect(reasoningChunks).toHaveLength(0)
})

it("should not yield reasoning chunk for empty or whitespace-only reasoning", async () => {
const mockStream = {
async *[Symbol.asyncIterator]() {
yield {
choices: [{ delta: { reasoning_content: " " } }],
usage: null,
}
yield {
choices: [{ delta: { content: "Hello" } }],
usage: { prompt_tokens: 5, completion_tokens: 5 },
}
},
}

mockCreate.mockReturnValue({
withResponse: vi.fn().mockResolvedValue({ data: mockStream }),
})

const generator = handler.createMessage("system", [{ role: "user", content: "Hi" }])
const results = []
for await (const chunk of generator) {
results.push(chunk)
}

const reasoningChunks = results.filter((c) => c.type === "reasoning")
expect(reasoningChunks).toHaveLength(0)
})
})

describe("tool ID normalization", () => {
it("should truncate tool IDs longer than 64 characters", async () => {
const optionsWithBedrock: ApiHandlerOptions = {
Expand Down
12 changes: 12 additions & 0 deletions src/api/providers/lite-llm.ts
Original file line number Diff line number Diff line change
Expand Up @@ -235,6 +235,18 @@ export class LiteLLMHandler extends RouterProvider implements SingleCompletionHa
yield { type: "text", text: delta.content }
}

if (delta) {
for (const key of ["reasoning_content", "reasoning"] as const) {

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I noticed openai.ts, requesty.ts, unbound.ts, and opencode-go.ts all only check reasoning_content — they don't have the reasoning fallback this adds. Was that a deliberate scope choice for this PR, or worth a follow-up to bring those in line too?

Copy link
Copy Markdown
Contributor Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

For openai.ts, requesty.ts, unbound.ts, and opencode-go.ts, I intentionally kept them out of this PR’s scope. This PR is focused on the LiteLLM issue from #447, with the base provider included only because it had the same duplicated logic. I agree those standalone providers should probably be aligned with the helper too, but I’d prefer to handle that in a follow-up PR to keep this change focused.

if (key in delta) {
const reasoningText = ((delta as any)[key] as string | undefined) || ""
if (reasoningText?.trim()) {
yield { type: "reasoning", text: reasoningText }
}
break
}
}
}
Comment thread
daewoongoh marked this conversation as resolved.
Outdated

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

This block is an exact copy of base-openai-compatible-provider.ts:150-160. If the break/fallback bug gets fixed in one place, the other will silently stay broken. Would it make sense to extract a shared helper (something like extractReasoningFromDelta(delta)) so both providers stay in sync automatically?

Copy link
Copy Markdown
Contributor Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I agreed with the helper suggestion and updated the PR. lite-llm.ts and base-openai-compatible-provider.ts now both use a shared extractReasoningFromDelta(delta) helper, so the fallback behavior stays in sync.

The helper also fixes the fallback bug: reasoning_content: null, non-string, or "" now falls through to reasoning, and I added tests for that case. I also preserved whitespace-only chunks like " " and "\n\n" because streamed reasoning deltas can use those as word/paragraph boundaries.


// Handle tool calls in stream - emit partial chunks for NativeToolCallParser
if (delta?.tool_calls) {
for (const toolCall of delta.tool_calls) {
Expand Down
Loading