Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 2 additions & 1 deletion package.json
Original file line number Diff line number Diff line change
Expand Up @@ -134,6 +134,7 @@
"test:autoresearch": "npx tsx test/continuous-test-suite-autoresearch.ts",
"test:autoresearch:redis": "npx tsx test/continuous-test-suite-autoresearch-redis.ts",
"test:anthropic-tools-policy": "npx tsx test/continuous-test-suite-anthropic-tools-policy.ts",
"test:anthropic-structured": "npx tsx test/continuous-test-suite-anthropic-structured-tools.ts",
"test:sagemaker-tools": "npx tsx test/continuous-test-suite-sagemaker-tools.ts",
"test:anthropic-multimodal": "npx tsx test/continuous-test-suite-anthropic-multimodal.ts",
"test:excel-interop": "npx tsx test/continuous-test-suite-excel-interop.ts",
Expand All @@ -153,7 +154,7 @@
"test:system-messages": "npx tsx test/continuous-test-suite-system-messages.ts",
"test:test-stubs": "npx tsx test/continuous-test-suite-test-stubs.ts",
"test:tool-routing-semantic": "npx tsx test/continuous-test-suite-tool-routing-semantic.ts",
"test:unit": "pnpm run test:envguard && pnpm run test:bugfixes && pnpm run test:file-detector-extension && pnpm run test:file-detector-magic-bytes && pnpm run test:mcp:infra && pnpm run test:mcp:bash && pnpm run test:mcp:limits && pnpm run test:mcp:spans && pnpm run test:autoresearch:redis && pnpm run test:tool-routing && pnpm run test:tool-routing-cli && pnpm run test:tool-dedup && pnpm run test:model-pool && pnpm run test:litellm-context && pnpm run test:dedup-execute-map && pnpm run test:step-budget-guard && pnpm run test:agent-plumbing && pnpm run test:tool-execution-recorder && pnpm run test:proxy-terminal-errors && pnpm run test:system-messages && pnpm run test:tool-routing-semantic && pnpm run test:anthropic-tools-policy && pnpm run test:sagemaker-tools && pnpm run test:anthropic-multimodal && pnpm run test:excel-interop && pnpm run test:model-capabilities && pnpm run test:agent-runtime:vitest && pnpm run test:agent-delegation && pnpm run test:retry-after:vitest && pnpm run test:sampling-params && pnpm run test:structured-recovery && pnpm run test:prompt-redaction && pnpm run test:mcp-result-cache && pnpm run test:test-stubs && pnpm run test:model-not-found-retryable && pnpm run test:websearch-grounding",
"test:unit": "pnpm run test:envguard && pnpm run test:bugfixes && pnpm run test:file-detector-extension && pnpm run test:file-detector-magic-bytes && pnpm run test:mcp:infra && pnpm run test:mcp:bash && pnpm run test:mcp:limits && pnpm run test:mcp:spans && pnpm run test:autoresearch:redis && pnpm run test:tool-routing && pnpm run test:tool-routing-cli && pnpm run test:tool-dedup && pnpm run test:model-pool && pnpm run test:litellm-context && pnpm run test:dedup-execute-map && pnpm run test:step-budget-guard && pnpm run test:agent-plumbing && pnpm run test:tool-execution-recorder && pnpm run test:proxy-terminal-errors && pnpm run test:system-messages && pnpm run test:tool-routing-semantic && pnpm run test:anthropic-tools-policy && pnpm run test:anthropic-structured && pnpm run test:sagemaker-tools && pnpm run test:anthropic-multimodal && pnpm run test:excel-interop && pnpm run test:model-capabilities && pnpm run test:agent-runtime:vitest && pnpm run test:agent-delegation && pnpm run test:retry-after:vitest && pnpm run test:sampling-params && pnpm run test:structured-recovery && pnpm run test:prompt-redaction && pnpm run test:mcp-result-cache && pnpm run test:test-stubs && pnpm run test:model-not-found-retryable && pnpm run test:websearch-grounding",
"// CI tier — live providers, runs only when API keys are present (test:credentials and test:dynamic make real provider calls when keys are set, so they live here, not in test:unit)": "",
"test:live": "pnpm run test:providers && pnpm run test:mcp:http && pnpm run test:mcp:sdk && pnpm run test:mcp:cli && pnpm run test:observability && pnpm run test:context && pnpm run test:memory && pnpm run test:tool-reliability && pnpm run test:evaluation && pnpm run test:autoresearch && pnpm run test:credentials && pnpm run test:dynamic",
"// CI tier — product output (image/video/TTS/PPT) — costs $$ per run": "",
Expand Down
26 changes: 26 additions & 0 deletions src/lib/core/modules/GenerationHandler.ts
Original file line number Diff line number Diff line change
Expand Up @@ -58,11 +58,13 @@ import {
isToolsSchemaExclusionInForce,
} from "./structuredOutputPolicy.js";
import { coerceJsonToSchema } from "../../utils/json/coerce.js";
import { convertZodToJsonSchema } from "../../utils/schemaConversion.js";
import type {
LanguageModel,
ModelMessage,
PrepareStepFunction,
Tool,
ZodUnknownSchema,
} from "../../types/index.js";
import { NoObjectGeneratedError } from "../../utils/generationErrors.js";
import { Output, stepCountIs } from "../../utils/tool.js";
Expand Down Expand Up @@ -165,11 +167,15 @@ function buildProviderOptions(
options: TextGenerationOptions,
isGoogleProvider: boolean,
callerTimeoutMs: number | undefined,
finalResultSchema: Record<string, unknown> | undefined,
): Parameters<typeof generateText>[0]["providerOptions"] {
const providerOptions: Record<string, Record<string, unknown>> = {};
if (callerTimeoutMs !== undefined) {
providerOptions.neurolink = { timeoutMs: callerTimeoutMs };
}
if (finalResultSchema) {
providerOptions.anthropic = { finalResultSchema };
}
if (options.thinkingConfig?.enabled && isGoogleProvider) {
// Gemini 3 uses thinkingLevel; Gemini 2.5 uses thinkingBudget.
providerOptions.google = {
Expand Down Expand Up @@ -349,10 +355,30 @@ export class GenerationHandler {
const { callerTimeoutMs, turnBudgetMs, wrapupLeadMs, turnDeadline } =
resolveTurnBudget(options, turnStartMs);
let wrapupForced = false;
// The native Anthropic Messages surface cannot combine AI-SDK structured
// output with tools (see structuredOutputPolicy — experimental_output
// replaces the tools array), so `useStructuredOutput` is false above and
// the schema would simply be dropped for every agent/MCP turn. Hand the
// JSON Schema to the provider instead: it appends an additive
// `final_result` tool and returns the answer as that tool's arguments,
// keeping the real tools callable. Bedrock is deliberately excluded — it
// runs on the third-party @ai-sdk/amazon-bedrock model, which has no such
// handling.
const finalResultSchema =
this.providerName === "anthropic" &&
!!options.schema &&
shouldUseTools &&
Object.keys(tools).length > 0
? (convertZodToJsonSchema(options.schema as ZodUnknownSchema) as Record<
string,
unknown
>)
: undefined;
const providerOptions = buildProviderOptions(
options,
isGoogleProvider,
callerTimeoutMs,
finalResultSchema,
);

// Hoist system-role messages into generateText's top-level `system` option
Expand Down
8 changes: 8 additions & 0 deletions src/lib/core/modules/structuredOutputPolicy.ts
Original file line number Diff line number Diff line change
Expand Up @@ -38,6 +38,14 @@ export function isGeminiProvider(
* experimental_output + tools silently drops tool_use blocks on this surface, so
* structured output must be disabled when tools are active. Vertex+Claude is NOT
* matched here (different transport, no conflict).
*
* Being excluded here no longer means the schema is LOST for provider
* "anthropic": GenerationHandler forwards the JSON Schema to the provider via
* `providerOptions.anthropic.finalResultSchema`, and the provider appends an
* additive `final_result` tool (see providers/anthropic/structuredOutput.ts) —
* schema enforcement without giving up tool calling. "bedrock" has no such
* handling (it runs on the third-party @ai-sdk/amazon-bedrock model) and still
* falls back to text-mode coercion.
*/
export function isNativeAnthropicProvider(providerName: string): boolean {
return providerName === "anthropic" || providerName === "bedrock";
Expand Down
133 changes: 130 additions & 3 deletions src/lib/providers/anthropic/client.ts
Original file line number Diff line number Diff line change
Expand Up @@ -55,6 +55,7 @@ import type {
ClaudeSubscriptionTier,
ClaudeUsageInfo,
OAuthToken,
ZodUnknownSchema,
} from "../../types/index.js";
import {
AuthenticationError,
Expand Down Expand Up @@ -108,6 +109,12 @@ import {
} from "../openaiChatCompletionsClient.js";

import { ANTHROPIC_BETA_HEADERS } from "./constants.js";
import {
appendFinalResultInstruction,
appendFinalResultTool,
FINAL_RESULT_TOOL_NAME,
stringifyFinalResultInput,
} from "./structuredOutput.js";

// AnthropicProviderConfig is imported from types/providers.ts
// Re-export for backward compatibility
Expand Down Expand Up @@ -1382,9 +1389,13 @@ export class AnthropicProvider extends BaseProvider {
} & Record<string, unknown>,
) => {
await refreshAuth();
const { system, messages } = messagesToAnthropic(
const built = messagesToAnthropic(
options.prompt as Array<{ role: string; content: unknown }>,
);
const messages = built.messages;
// `let`: the additive structured-output path below appends the
// final_result instruction to the system prompt.
let system = built.system;

let tools: Anthropic.Messages.Tool[] | undefined = (options.tools ?? [])
.filter((t) => t.type === "function")
Expand Down Expand Up @@ -1434,6 +1445,25 @@ export class AnthropicProvider extends BaseProvider {
toolChoice = { type: "tool", name: jsonTool };
}

// Additive structured output: when the caller wants a schema AND real
// tools, the forced-json path above cannot be used (it replaces the
// tools array), and the AI-SDK experimental_output path is excluded
// for this surface by structuredOutputPolicy. GenerationHandler hands
// the JSON Schema down here instead, and we APPEND a `final_result`
// tool — tool_choice stays auto, so every real tool keeps working and

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

💬 MINOR: Bare type assertion - Line 1453 uses a bare type assertion (as Record<string, unknown> | undefined) when accessing options.providerOptions?.anthropic?.finalResultSchema. While this is safe given the current code structure, using proper type narrowing would be cleaner and more maintainable. Consider restructuring the property access or adding a type guard instead of relying on the assertion.

// the model self-selects final_result when it is ready to answer.
const finalResultSchema = options.providerOptions?.anthropic
?.finalResultSchema as Record<string, unknown> | undefined;
let finalResultActive = false;
if (!jsonTool && finalResultSchema) {
const appended = appendFinalResultTool(tools, finalResultSchema);
tools = appended.tools;
finalResultActive = appended.applied;
if (appended.applied) {
system = appendFinalResultInstruction(system);
}
}

// Extended thinking passthrough (providerOptions.anthropic.thinking).
const thinking = options.providerOptions?.anthropic?.thinking as
| { type: "enabled"; budget_tokens: number }
Expand Down Expand Up @@ -1528,6 +1558,7 @@ export class AnthropicProvider extends BaseProvider {
}

const content: Array<{ type: string } & Record<string, unknown>> = [];
let finalResultText: string | undefined;
for (const block of response.content) {
if (block.type === "thinking") {
content.push({ type: "reasoning", text: block.thinking });
Expand All @@ -1544,6 +1575,13 @@ export class AnthropicProvider extends BaseProvider {
type: "text",
text: stringifyToolInput(block.input),
});
} else if (
finalResultActive &&
block.name === FINAL_RESULT_TOOL_NAME
) {
// Internal pattern: never surfaced as a tool call. Its arguments
// ARE the structured answer.
finalResultText = stringifyToolInput(block.input);
} else {
content.push({
type: "tool-call",
Expand All @@ -1555,12 +1593,34 @@ export class AnthropicProvider extends BaseProvider {
}
}

// final_result is terminal — parity with the native Claude-on-Vertex
// and Gemini loops, which break out of the tool loop the moment it
// arrives. Reasoning blocks are kept; any prose preamble and any tool
// calls issued alongside it are dropped so `text` is exactly the
// structured payload and the AI-SDK loop stops here.
if (finalResultText !== undefined) {
const reasoning = content.filter((part) => part.type === "reasoning");
content.length = 0;
content.push(...reasoning, { type: "text", text: finalResultText });
logger.debug(
"[Anthropic] Extracted structured output from final_result tool (generate)",
{ chars: finalResultText.length },
);
}

const cacheRead = response.usage.cache_read_input_tokens ?? 0;
const cacheWrite = response.usage.cache_creation_input_tokens ?? 0;
return {
content,
finishReason: {
unified: mapAnthropicStopReason(response.stop_reason),
// A final_result call ends the turn: the provider reports
// stop_reason "tool_use", but no tool call is surfaced, so
// reporting "tool-calls" would misread as a step-capped turn.
// `raw` still carries the provider's verbatim stop_reason.
unified:
finalResultText !== undefined
? ("stop" as const)
: mapAnthropicStopReason(response.stop_reason),
raw: response.stop_reason ?? "stop",
},
usage: {
Expand Down Expand Up @@ -1745,6 +1805,9 @@ export class AnthropicProvider extends BaseProvider {
messages: Anthropic.Messages.MessageParam[];
};
let shouldUseTools: boolean;
// True once the additive `final_result` tool is in the request — the
// streaming twin of the doGenerate path above.
let finalResultActive = false;
try {
// options.tools is pre-merged by BaseProvider.stream() with base tools
// (MCP/built-in) + user-provided tools (RAG, etc.)
Expand All @@ -1761,6 +1824,24 @@ export class AnthropicProvider extends BaseProvider {
payload = messagesToAnthropic(
built as Array<{ role: string; content: unknown }>,
);
// Schema + tools: append final_result rather than pinning tool_choice to
// a json tool, so the real tools stay callable for the whole turn.
// Unlike generate, no plumbing is needed — this is a native loop, so the
// caller's Zod/JSON schema is right here on the options.
if (options.schema && anthropicTools && anthropicTools.length > 0) {
const appended = appendFinalResultTool(
anthropicTools,
convertZodToJsonSchema(options.schema as ZodUnknownSchema) as Record<
string,
unknown
>,
);
anthropicTools = appended.tools;
finalResultActive = appended.applied;
if (appended.applied) {
payload.system = appendFinalResultInstruction(payload.system);
}
}
Comment on lines +1827 to +1844

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

🚀 Performance & Scalability | 🟡 Minor | ⚡ Quick win

Mid-turn tool hydration appends after final_result and breaks the LAST-position invariant.

appendFinalResultTool places final_result last on purpose. The comment in src/lib/providers/anthropic/structuredOutput.ts line 81 states that this keeps an upstream cache_control breakpoint marking the same prefix boundary.

The discovery block at line 1976 then runs anthropicTools.push(...) on every step. When tools.discovery hydrates a tool mid-turn, that tool lands after final_result. The tool prefix changes shape, and the invariant no longer holds for the rest of the turn.

The suite does not catch this. In the multi-step test the hydrated set is empty, so the push never runs.

Re-insert final_result at the end after hydration.

♻️ Proposed fix in the discovery block (around line 1975)
           if (Object.keys(hydrated).length > 0) {
-            anthropicTools.push(...(toolsToAnthropic(hydrated) ?? []));
+            const hydratedTools = toolsToAnthropic(hydrated) ?? [];
+            // Keep final_result LAST: hydrated tools must be spliced in before
+            // it so the cache_control prefix boundary is unchanged.
+            const finalIndex = anthropicTools.findIndex(
+              (t) => t.name === FINAL_RESULT_TOOL_NAME,
+            );
+            if (finalResultActive && finalIndex !== -1) {
+              anthropicTools.splice(finalIndex, 0, ...hydratedTools);
+            } else {
+              anthropicTools.push(...hydratedTools);
+            }
             logger.info(
🤖 Prompt for AI Agents
Verify each finding against current code. Fix only still-valid issues, skip the
rest with a brief reason, keep changes minimal, and validate.

In `@src/lib/providers/anthropic/client.ts` around lines 1827 - 1844, Update the
tool discovery hydration block around the existing anthropicTools.push call so
any newly hydrated tools are inserted before final_result rather than after it.
Preserve final_result as the last tool whenever finalResultActive is enabled,
while retaining the current append behavior when it is not active.

} catch (setupErr) {
timeoutController?.cleanup();
throw this.handleProviderError(setupErr);
Expand Down Expand Up @@ -1869,6 +1950,15 @@ export class AnthropicProvider extends BaseProvider {
...(totalCacheWrite > 0 ? { cacheCreationTokens: totalCacheWrite } : {}),
});

// Structured-output turns are delivered as ONE chunk, not incrementally:
// a caller that passed a schema needs parseable JSON, and text deltas
// emitted before the model calls final_result would prefix the payload
// with prose and break every JSON.parse on the consumer side. Same
// contract as the native Vertex loops. Non-schema streams are untouched
// and stay fully incremental.
let bufferedText = "";
let finalResultText: string | undefined;

const runLoop = async (): Promise<void> => {
const conversation = payload.messages.slice();

Expand Down Expand Up @@ -1984,7 +2074,11 @@ export class AnthropicProvider extends BaseProvider {
event.index,
(textAcc.get(event.index) ?? "") + delta.text,
);
pushChunk({ content: delta.text });
if (finalResultActive) {
bufferedText += delta.text;
} else {
pushChunk({ content: delta.text });
}
} else if (delta.type === "thinking_delta") {
const acc = thinkingAcc.get(event.index) ?? {
text: "",
Expand Down Expand Up @@ -2018,6 +2112,26 @@ export class AnthropicProvider extends BaseProvider {
}
lastStop = stopReason;

// final_result is terminal: its arguments ARE the answer, so the turn
// ends here and any tool calls issued alongside it are not executed
// (parity with the native Vertex loops). It is never executed as a
// tool, never recorded in toolsUsed, and never stored as a tool
// execution — the pattern stays invisible to callers.
if (finalResultActive) {
const finalCall = [...toolAcc.values()].find(
(acc) => acc.name === FINAL_RESULT_TOOL_NAME,
);
if (finalCall) {
finalResultText = stringifyFinalResultInput(finalCall.inputJson);
lastStop = "end_turn";
logger.debug(
"[Anthropic] Extracted structured output from final_result tool (stream)",
{ chars: finalResultText.length },
);
break;
}
}

if (stopReason !== "tool_use" || toolAcc.size === 0) {
break;
}
Expand Down Expand Up @@ -2191,6 +2305,19 @@ export class AnthropicProvider extends BaseProvider {
throw this.formatProviderError(error);
})
.finally(() => {
// Deliver the buffered structured-output turn: `finalResultText` when
// the model called final_result, otherwise the prose it produced
// instead — never nothing, so a model that ignores the instruction
// degrades to today's plain-text behaviour rather than an empty
// stream. In `finally` so a turn that dies mid-loop still surfaces
// the text it had already buffered, exactly as the unbuffered path
// surfaces its partial deltas.
if (finalResultActive) {
const output = finalResultText ?? bufferedText;
if (output.length > 0) {
pushChunk({ content: output });
}
}
timeoutController?.cleanup();
pushChunk({ done: true });
});
Expand Down
Loading
Loading