Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -25,3 +25,4 @@ plan/
.tmp
.tmp-*
temp_reference/
.codelite/
9 changes: 9 additions & 0 deletions src/constants/messages.ts
Original file line number Diff line number Diff line change
@@ -1 +1,10 @@
export const NO_CONTENT_MESSAGE = '(no content)'

// Semantic assistant boundary injected by the OpenAI shim when a 'tool' role
// message must be followed by an 'assistant' message (Mistral / Devstral
// strict role sequence). The query loop detects this text to decide whether
// to continue the tool-execution path or treat the turn as stalled.
//
// IMPORTANT: if you change this string, update the detection in query.ts as
// well — both places must stay in sync.
export const TOOL_RESULTS_RECEIVED_MARKER = '[Tool results received]'
246 changes: 238 additions & 8 deletions src/query.ts
Original file line number Diff line number Diff line change
Expand Up @@ -65,6 +65,7 @@ import {
createMicrocompactBoundaryMessage,
} from './utils/messages.js'
import { analyzeContinuationIntent } from './utils/continuation.js'
import { TOOL_RESULTS_RECEIVED_MARKER } from './constants/messages.js'
import { generateToolUseSummary } from './services/toolUseSummary/toolUseSummaryGenerator.js'
import { prependUserContext, appendSystemContext } from './utils/api.js'
import {
Expand Down Expand Up @@ -995,6 +996,19 @@ async function* queryLoop(
// loop-exit signal. If false after streaming, we're done (modulo stop-hook retry).
const toolUseBlocks: ToolUseBlock[] = []
let needsFollowUp = false
// Tracks whether the streaming loop encountered a marker-only (shim
// artifact) assistant response this iteration — used by the
// continuation-nudge block below to force a nudge when the stripped
// text is empty (model effectively said nothing).
let markerOnlyStall = false
// Provenance flag: true when the API request that produced the current
// assistant response contained a tool-result boundary at the end
// (last message in messagesForQuery is a tool result). This is the
// condition the OpenAI shim checks (prev.role === 'tool' &&
// msg.role === 'user') before injecting TOOL_RESULTS_RECEIVED_MARKER.
// Scoping marker suppression to this boundary prevents suppressing
// legitimate user output that happens to contain the marker text.
let shimToolResultBoundary = false

queryCheckpoint('query_setup_start')
const useStreamingToolExecution =
Expand Down Expand Up @@ -1254,6 +1268,54 @@ async function* queryLoop(
attemptWithFallback = false
try {
let streamingFallbackOccured = false
// Set shim provenance for this loop iteration: the OpenAI shim
// injects TOOL_RESULTS_RECEIVED_MARKER as a standalone assistant
// message only when the last message in the request is a tool
// result (prev.role === 'tool' && msg.role === 'user'). Gate
// marker suppression on this boundary so the model can
// legitimately say "[Tool results received]" without being
// silently withheld.
// We scan backwards from the end for the shim boundary pattern:
// the last message must be a user message and some earlier message
// must contain a tool_result block (meaning the shim injected the
// marker for this turn). Continuation nudges keep appending
// user messages after the tool results, so we skip over trailing
// assistant messages and keep scanning past them until we find a
// user message or exhaust the array.
// Initialize false, scan messages backward from the end.
// Stop at the most recent tool_use so results from earlier turns
// cannot qualify. Set boundary true only when a matching
// tool_result is found within that scope.
shimToolResultBoundary = false
const lastMsg = messagesForQuery.at(-1)
if (lastMsg?.type === 'user') {
// Find the index of the most recent tool_use message (scanning backward)
let mostRecentToolUseIndex = -1
for (let i = messagesForQuery.length - 2; i >= 0; i--) {
const msg = messagesForQuery[i]
if (msg?.type === 'assistant') {
const content = msg.message.content as { type: string }[] | undefined
if (content?.some(c => c.type === 'tool_use')) {
mostRecentToolUseIndex = i
break
}
}
}
// Scan backward from end, looking for user message with tool_result
for (let i = messagesForQuery.length - 1; i >= 0; i--) {
const msg = messagesForQuery[i]
if (i < mostRecentToolUseIndex && mostRecentToolUseIndex !== -1) {
break // stopped at most recent tool_use
}
if (msg?.type === 'user') {
const content = msg.message.content as { type: string }[] | string | undefined
if (Array.isArray(content) && content.some(c => c.type === 'tool_result')) {
shimToolResultBoundary = true
break
}
}
}
}
Comment thread
coderabbitai[bot] marked this conversation as resolved.
queryCheckpoint('query_api_streaming_start')
for await (const message of deps.callModel({
messages: prependUserContext(messagesForQuery, userContext),
Expand Down Expand Up @@ -1455,19 +1517,139 @@ async function* queryLoop(
) {
withheld = true
}
if (!withheld) {
yield yieldMessage
}
if (message.type === 'assistant') {
assistantMessages.push(message)

const msgToolUseBlocks = message.message.content.filter(
content => content.type === 'tool_use',
) as ToolUseBlock[]
if (msgToolUseBlocks.length > 0) {
// Detect the semantic boundary the OpenAI shim injects so
// self-hosted providers (e.g. llama-server) which echo the
// shim's assistant-injection marker back in their response are
// handled correctly. Use the same whitespace-tolerant pattern
// as the strip logic below so detection and removal stay in
// sync even when the server wraps the marker in newlines.
// Detection pattern: \s* greedily matches surrounding
// whitespace/newlines so we find the marker even when the
// provider wraps it in formatting.
const markerPattern = new RegExp(
`\\s*${TOOL_RESULTS_RECEIVED_MARKER
.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}\\s*`,
'g',
)
// Strip pattern: matches only the escaped literal marker,
// without consuming adjacent whitespace. This prevents
// mid-sentence corruption when the marker is embedded in
// larger text (e.g. "continue [Marker] and finish" →
// "continue and finish" instead of "continueand finish").
// Strip pattern: matches only the escaped literal marker,
// without consuming adjacent whitespace. This prevents
// mid-sentence corruption when the marker is embedded in
// larger text (e.g. "continue [Marker] and finish" →
// "continue and finish" instead of "continueand finish").
const stripPattern = new RegExp(
TOOL_RESULTS_RECEIVED_MARKER.replace(
/[.*+?^${}()|[\]\\]/g,
'\\$&',
),
'g',
)
const hasToolResultsMarker = message.message.content.some(
content =>
content.type === 'text' &&
typeof content.text === 'string' &&
markerPattern.test(content.text),
)
Comment thread
coderabbitai[bot] marked this conversation as resolved.
// Marker-only response (no tool_use blocks): this is a shim
// artifact, not a user-visible message. Withhold it from the
// UI and strip it from the state so the continuation-nudge
// block below sees a clean text tail and the capped nudge
// path handles the stall (max 20 retries).
// Only suppress when shim provenance is established
// (shimToolResultBoundary) so the model can legitimately say
// "[Tool results received]" without being silenced.
let remainingText = ''
if (hasToolResultsMarker && msgToolUseBlocks.length === 0 && shimToolResultBoundary) {
// Compute remaining text after removing the marker. Only
// withhold / stall when there is no substantive text left;
// a message containing real continuation text should stay
// visible even without tool_use blocks.
const remaining = message.message.content
.map((content: unknown) => {
if (
typeof content === 'object' &&
content !== null &&
'type' in content &&
'text' in content &&
(content as { type: string }).type === 'text'
) {
const textBlock = content as { text: string }
const cleaned = textBlock.text.replace(
stripPattern,
'',
)
return cleaned.trim().length > 0
? { ...textBlock, text: cleaned }
: null
}
return content
})
.filter(Boolean) as typeof message.message.content
remainingText = remaining
.filter(b => typeof b === 'object' && b !== null && 'text' in b && (b as { text: string }).text.trim().length > 0)
.map(b => (b as { text: string }).text)
.join(' ')
if (remainingText.trim().length === 0) {
withheld = true
markerOnlyStall = true
}
if (remaining.length === 0) {
// Strip the marker from message.message so subsequent
// query iterations (after a continuation nudge) don't
// re-inject the shim marker — the model already echoed
// it, and re-injecting causes an infinite echo loop.
// The yielded placeholder is empty to keep transcript clean.
message.message = {
...message.message,
content: [{ type: 'text' as const, text: '' }] as typeof message.message.content,
} as typeof message.message
yieldMessage = {
...yieldMessage,
message: message.message,
}
} else {
yieldMessage = {
...yieldMessage,
message: { ...yieldMessage.message, content: remaining },
}
// Also strip the marker from message.message to prevent
// the shim from re-injecting it on the next iteration.
message.message = {
...message.message,
content: remaining,
} as typeof message.message
}
} else if (msgToolUseBlocks.length > 0) {
toolUseBlocks.push(...msgToolUseBlocks)
needsFollowUp = true
}
// Clear markerOnlyStall when a substantive message arrives
// after a marker-only one — a prior empty block shouldn't
// cause an unnecessary continuation nudge. Also clear it when
// a marker-containing block has real text after stripping
// (the old condition only checked tool_use blocks or absence
// of the marker, missing the case where the message has both
// the marker AND substantive remaining text).
if (
msgToolUseBlocks.length > 0 ||
!hasToolResultsMarker ||
remainingText.trim().length > 0 ||
!shimToolResultBoundary
) {
Comment thread
coderabbitai[bot] marked this conversation as resolved.
markerOnlyStall = false
}
assistantMessages.push(message)
if (!withheld) {
yield yieldMessage
}

if (
streamingToolExecutor &&
Expand All @@ -1479,6 +1661,10 @@ async function* queryLoop(
}
}

if (message.type !== 'assistant') {
yield message
}

if (
streamingToolExecutor &&
!toolUseContext.abortController.signal.aborted
Expand Down Expand Up @@ -2211,6 +2397,13 @@ async function* queryLoop(
continue
}

// Capture the current stop-hook result for use in the continuation-nudge
// check below. The stale `stopHookActive` flag describes the PREVIOUS
// turn's result — after a successful stop-hook pass (no blocking errors),
// the hook already ran again and cleared itself, so using the stale flag
// forces an unnecessary extra model request.
let currentStopHookActive = stopHookResult.stopHookActive

if (feature('TOKEN_BUDGET')) {
const decision = checkTokenBudget(
budgetTracker!,
Expand Down Expand Up @@ -2288,9 +2481,30 @@ async function* queryLoop(
.join(' ')
.toLowerCase()

const { shouldNudge, reason: nudgeReason } = analyzeContinuationIntent(
let { shouldNudge, reason: nudgeReason } = analyzeContinuationIntent(
lastText,
)
) as { shouldNudge: boolean; reason?: string }

// Marker-only stall: when a self-hosted LLM echoes the shim's
// assistant-injection marker and the remaining text is empty (or
// whitespace), the model effectively said nothing. Force a nudge
// so the turn doesn't complete prematurely.
if (!shouldNudge && markerOnlyStall) {
shouldNudge = true
nudgeReason = 'marker_only_stall'
}

// Stop-hook-active: the model was already told the task is
// incomplete (via goal evaluation or hook blocking errors).
// Force a nudge even if the text itself has no continuation
// signal — the model is being prompted to continue from the
// previous turn's tool results. Use the CURRENT stop-hook
// result (currentStopHookActive) rather than the stale state
// flag, which may have been cleared by this pass's handleStopHooks.
if (!shouldNudge && currentStopHookActive) {
shouldNudge = true
nudgeReason = 'stop_hook_active'
}

if (shouldNudge) {
logForDebugging(
Expand Down Expand Up @@ -2333,6 +2547,22 @@ async function* queryLoop(
}
}

// Nudge cap hit and last message was marker-only: the model produced
// zero real output but the cap guard skipped the nudge block. Report
// a distinct reason so a stuck self-hosted endpoint isn't masked as
// a successful completion. Yield a visible system message so REPL
// consumers see the failure instead of silently calling onTurnComplete.
if (markerOnlyStall && state.continuationNudgeCount >= MAX_CONTINUATION_NUDGES) {
logForDebugging(
`Marker-only stall exhausted (${MAX_CONTINUATION_NUDGES} nudges) — model produced no real output`,
)
yield createSystemMessage(
`Marker-only stall exhausted after ${MAX_CONTINUATION_NUDGES} retries — the model kept echoing the internal tool-result marker without producing real output.`,
'error',
)
return { reason: 'marker_stall_exhausted' as const }
}

return { reason: 'completed' }
}

Expand Down
Loading