From b6726d57e4d0b91e78d9b57831c25cf70b89dfc2 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 17:33:49 -0700 Subject: [PATCH 001/376] feat(desktop): per-profile scope selector in the Capabilities view (#86548) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * feat(desktop): per-profile scope selector in the Capabilities view Adds a 'Configuring:' profile selector above the Tools and MCP tabs in the Capabilities (Skills) view, so a user can configure ANY profile's toolsets and MCP servers without switching the whole app into that profile. - hermes.ts: every capability fetcher (getToolsets, setToolsetEnabled, getToolsetConfig/Models, selectToolsetModel/Provider, runToolsetPostSetup, getMcpCatalog, installMcpCatalogEntry, testMcpServer, saveMcpServers, auth/oauth flow, setEnvVar/deleteEnvVar/revealEnvVar, startOAuthLogin/ pollOAuthSession, getActionStatus, getHermesConfigRecord) takes an optional trailing profile? that forwards to profileScoped(profile). Omitting it preserves exact app-wide behavior (profileScoped(undefined) → _apiProfile). - use-config-record.ts: hermesConfigKey(profile)/useHermesConfigRecord(profile)/ hermesConfigCacheWriter(profile) — per-profile RQ keys (scope-in-key). - toolset-config-panel.tsx + mcp-tab.tsx: thread profile through every fetch and their nested children (EnvVarField, PostSetupRunner, ModelCatalogPicker), keyed/remounted per selected profile so switching never shows stale state. - skills/index.tsx: the selector (seeded from profiles.list, default→'Hermes', shown only with >1 profile), defaulting to ; toolsets query + toggles keyed and scoped to the selection; McpTab/ToolsetDetail remounted per scope. - i18n: skills.configuringProfile (en + zh; others fall back). When the selected profile equals the active one (the default), behavior is identical to before — the selector is a pure override layered on top. Tests: index.test.tsx — new case asserts picking a non-active profile in the selector refetches toolsets scoped to that profile; existing single-profile cases still pass (selector hidden with one profile). 5/5. Full-project tsc clean. * Fix CI: command-palette getHermesConfigRecord call, panel test mock, lint - command-palette/index.tsx: getHermesConfigRecord now takes an optional profile; passing the bare fn as queryFn fed it react-query's context object (TS2769 + mcp_servers on {}). Wrap in an arrow. - toolset-config-panel.test.tsx: use-config-record now imports normalizeProfileKey from @/store/profile, which calls setApiRequestProfile at module-init; the full-replacement @/hermes mock must provide it (+ getApiRequestProfile). - index.test.tsx selector test: stub Element.prototype.scrollIntoView (Radix Select calls it on open; jsdom lacks it). - toolset-config-panel.tsx: PostSetupRunner useCallback missing 'profile' dep (stale-closure correctness); jsx-prop sort order. Verified in a full-dep checkout: tsc 0 errors, eslint clean, panel 28/28 + skills 5/5. --------- Co-authored-by: Teknium --- .../desktop/src/app/command-palette/index.tsx | 2 +- .../src/app/hooks/use-config-record.ts | 30 ++++- .../settings/toolset-config-panel.test.tsx | 7 +- .../src/app/settings/toolset-config-panel.tsx | 48 +++++--- apps/desktop/src/app/skills/index.test.tsx | 61 ++++++++-- apps/desktop/src/app/skills/index.tsx | 107 ++++++++++++++--- apps/desktop/src/app/skills/mcp-tab.tsx | 60 ++++++---- apps/desktop/src/hermes.ts | 111 +++++++++++------- apps/desktop/src/i18n/en.ts | 1 + apps/desktop/src/i18n/types.ts | 1 + apps/desktop/src/i18n/zh.ts | 1 + 11 files changed, 320 insertions(+), 109 deletions(-) diff --git a/apps/desktop/src/app/command-palette/index.tsx b/apps/desktop/src/app/command-palette/index.tsx index f9a640000aba7..e367e8058260d 100644 --- a/apps/desktop/src/app/command-palette/index.tsx +++ b/apps/desktop/src/app/command-palette/index.tsx @@ -605,7 +605,7 @@ function CommandPaletteBody({ onExited }: { onExited: () => void }) { // reopen paints from cache and revalidates in the background. const configQuery = useQuery({ queryKey: ['command-palette', 'config'], - queryFn: getHermesConfigRecord + queryFn: () => getHermesConfigRecord() }) const sessionsQuery = useQuery({ diff --git a/apps/desktop/src/app/hooks/use-config-record.ts b/apps/desktop/src/app/hooks/use-config-record.ts index ca4f00cb2a59a..baec1ca630224 100644 --- a/apps/desktop/src/app/hooks/use-config-record.ts +++ b/apps/desktop/src/app/hooks/use-config-record.ts @@ -2,6 +2,7 @@ import { useQuery } from '@tanstack/react-query' import { getHermesConfigRecord } from '@/hermes' import { queryClient, writeCache } from '@/lib/query-client' +import { normalizeProfileKey } from '@/store/profile' import type { HermesConfigRecord } from '@/types/hermes' // One shared cache for the whole profile config record (`GET /api/config`). @@ -13,10 +14,33 @@ import type { HermesConfigRecord } from '@/types/hermes' // it pushes personality/cwd/voice/… into the session stores for live chat. export const HERMES_CONFIG_KEY = ['hermes-config-record'] as const +// Per-profile cache key. The base key (no profile suffix) is the app-wide +// active profile, unchanged for every caller that passes nothing. An explicit +// profile — the Capabilities profile-scope selector configuring ANOTHER +// profile — gets its own suffixed key so switching the selector refetches and +// never paints stale cross-profile config (the AGENTS.md scope-in-key rule). +export const hermesConfigKey = (profile?: null | string) => + profile == null ? HERMES_CONFIG_KEY : ([...HERMES_CONFIG_KEY, normalizeProfileKey(profile)] as const) + // staleTime 0 → serve cache instantly, background-revalidate on every mount. -export const useHermesConfigRecord = () => - useQuery({ queryKey: HERMES_CONFIG_KEY, queryFn: getHermesConfigRecord, staleTime: 0 }) +// `profile` scopes both the query key and the fetch; omitting it preserves the +// exact app-wide behavior (base key, `profileScoped(undefined)` fallback). +export const useHermesConfigRecord = (profile?: null | string) => + useQuery({ + queryKey: hermesConfigKey(profile), + // null/undefined both mean "no override" → fetch with undefined so + // profileScoped falls back to the app-wide active profile (passing null + // would wrongly target the primary backend). + queryFn: () => getHermesConfigRecord(profile ?? undefined), + staleTime: 0 + }) +// setHermesConfigCache writes the app-wide (base-key) record. Pass a profile to +// write the suffixed per-profile cache instead — keeps the selector's optimistic +// write-through landing on the same key its query reads. export const setHermesConfigCache = writeCache(HERMES_CONFIG_KEY) +export const hermesConfigCacheWriter = (profile?: null | string) => + writeCache(hermesConfigKey(profile)) -export const invalidateHermesConfig = () => queryClient.invalidateQueries({ queryKey: HERMES_CONFIG_KEY }) +export const invalidateHermesConfig = (profile?: null | string) => + queryClient.invalidateQueries({ queryKey: hermesConfigKey(profile) }) diff --git a/apps/desktop/src/app/settings/toolset-config-panel.test.tsx b/apps/desktop/src/app/settings/toolset-config-panel.test.tsx index df509c198e524..908144bc84921 100644 --- a/apps/desktop/src/app/settings/toolset-config-panel.test.tsx +++ b/apps/desktop/src/app/settings/toolset-config-panel.test.tsx @@ -62,7 +62,12 @@ vi.mock('@/hermes', () => ({ getHermesConfigRecord: () => getHermesConfigRecord(), getHermesConfigSchema: () => getHermesConfigSchema(), saveHermesConfig: (config: unknown) => saveHermesConfig(config), - getElevenLabsVoices: () => getElevenLabsVoices() + getElevenLabsVoices: () => getElevenLabsVoices(), + // @/store/profile (pulled in transitively via use-config-record's + // normalizeProfileKey import) calls this at module-init; the full-replacement + // mock must provide it or the module graph throws on load. + setApiRequestProfile: () => undefined, + getApiRequestProfile: () => null })) vi.mock('@/store/notifications', () => ({ diff --git a/apps/desktop/src/app/settings/toolset-config-panel.tsx b/apps/desktop/src/app/settings/toolset-config-panel.tsx index bc9c150aadb09..40571042eb152 100644 --- a/apps/desktop/src/app/settings/toolset-config-panel.tsx +++ b/apps/desktop/src/app/settings/toolset-config-panel.tsx @@ -40,6 +40,10 @@ interface ToolsetConfigPanelProps { /** Called after a key is saved/cleared or a provider chosen, so the parent * can refresh the "Configured / Needs keys" pill. */ onConfiguredChange?: () => void + /** Capabilities profile-scope override: configure THIS profile instead of the + * app-wide active one. Omitted (every other caller) → app-wide active + * profile, so behavior is unchanged. Threaded into every fetch below. */ + profile?: null | string } /** Toolsets whose backends expose a selectable model catalog (mirrors the @@ -81,9 +85,10 @@ interface EnvVarFieldProps { isSet: boolean onSaved: (key: string) => void onCleared: (key: string) => void + profile?: null | string } -function EnvVarField({ envVar, isSet, onSaved, onCleared }: EnvVarFieldProps) { +function EnvVarField({ envVar, isSet, onSaved, onCleared, profile }: EnvVarFieldProps) { const { t } = useI18n() const copy = t.settings.toolsets const navigate = useNavigate() @@ -104,7 +109,7 @@ function EnvVarField({ envVar, isSet, onSaved, onCleared }: EnvVarFieldProps) { setBusy(true) try { - await setEnvVar(envVar.key, value) + await setEnvVar(envVar.key, value, profile) setEditing(false) setValue('') onSaved(envVar.key) @@ -124,7 +129,7 @@ function EnvVarField({ envVar, isSet, onSaved, onCleared }: EnvVarFieldProps) { setBusy(true) try { - await deleteEnvVar(envVar.key) + await deleteEnvVar(envVar.key, profile) setRevealed(null) onCleared(envVar.key) notify({ kind: 'success', title: copy.removedTitle, message: copy.removedMessage(envVar.key) }) @@ -143,7 +148,7 @@ function EnvVarField({ envVar, isSet, onSaved, onCleared }: EnvVarFieldProps) { } try { - const result = await revealEnvVar(envVar.key) + const result = await revealEnvVar(envVar.key, profile) setRevealed(result.value) } catch (err) { notifyError(err, copy.failedReveal(envVar.key)) @@ -226,6 +231,7 @@ interface PostSetupRunnerProps { /** Refresh the parent config after the install finishes (a backend may now * report itself configured). */ onComplete?: () => void + profile?: null | string } /** @@ -239,7 +245,7 @@ interface PostSetupRunnerProps { * "Installed" pill plus a small "Re-run setup" text button, so clicking * around the panel doesn't look like it keeps reinstalling. */ -function PostSetupRunner({ toolset, postSetupKey, installed = false, onComplete }: PostSetupRunnerProps) { +function PostSetupRunner({ toolset, postSetupKey, installed = false, onComplete, profile }: PostSetupRunnerProps) { const { t } = useI18n() const copy = t.settings.toolsets const [running, setRunning] = useState(false) @@ -260,7 +266,7 @@ function PostSetupRunner({ toolset, postSetupKey, installed = false, onComplete activeRef.current = true try { - const started = await runToolsetPostSetup(toolset, postSetupKey) + const started = await runToolsetPostSetup(toolset, postSetupKey, profile) // The spawn endpoint reports ok:false if it couldn't launch the action // (e.g. unknown key, server-side spawn failure). Don't poll a status @@ -283,7 +289,7 @@ function PostSetupRunner({ toolset, postSetupKey, installed = false, onComplete break } - const polled = await getActionStatus(started.name, 300) + const polled = await getActionStatus(started.name, 300, profile) last = polled setStatus(polled) upsertDesktopActionTask(polled) @@ -316,7 +322,7 @@ function PostSetupRunner({ toolset, postSetupKey, installed = false, onComplete setRunning(false) } } - }, [toolset, postSetupKey, onComplete, copy]) + }, [toolset, postSetupKey, onComplete, copy, profile]) return (
@@ -364,6 +370,7 @@ interface ModelCatalogPickerProps { /** True when this provider is the one written to config — selecting a model * only makes sense for the active backend. */ isActiveBackend: boolean + profile?: null | string } /** @@ -373,7 +380,7 @@ interface ModelCatalogPickerProps { * radio-card list and persists the choice to `image_gen.model` / * `video_gen.model`. */ -function ModelCatalogPicker({ toolset, providerName, isActiveBackend }: ModelCatalogPickerProps) { +function ModelCatalogPicker({ toolset, providerName, isActiveBackend, profile }: ModelCatalogPickerProps) { const { t } = useI18n() const copy = t.settings.toolsets const [catalog, setCatalog] = useState(null) @@ -384,7 +391,7 @@ function ModelCatalogPicker({ toolset, providerName, isActiveBackend }: ModelCat let cancelled = false setLoading(true) - getToolsetModels(toolset, providerName) + getToolsetModels(toolset, providerName, profile) .then(next => { if (!cancelled) { setCatalog(next) @@ -404,13 +411,13 @@ function ModelCatalogPicker({ toolset, providerName, isActiveBackend }: ModelCat }) return () => void (cancelled = true) - }, [toolset, providerName]) + }, [toolset, providerName, profile]) const pick = async (modelId: string) => { setSaving(modelId) try { - await selectToolsetModel(toolset, modelId, providerName) + await selectToolsetModel(toolset, modelId, providerName, profile) setCatalog(current => (current ? { ...current, current: modelId } : current)) notify({ kind: 'success', title: copy.modelSelectedTitle, message: copy.modelSelectedMessage(modelId) }) } catch (err) { @@ -486,7 +493,7 @@ function ModelCatalogPicker({ toolset, providerName, isActiveBackend }: ModelCat ) } -export function ToolsetConfigPanel({ toolset, onConfiguredChange }: ToolsetConfigPanelProps) { +export function ToolsetConfigPanel({ toolset, onConfiguredChange, profile }: ToolsetConfigPanelProps) { const { t } = useI18n() const copy = t.settings.toolsets const [cfg, setCfg] = useState(null) @@ -516,7 +523,7 @@ export function ToolsetConfigPanel({ toolset, onConfiguredChange }: ToolsetConfi setLoading(true) try { - const next = await getToolsetConfig(toolset) + const next = await getToolsetConfig(toolset, profile) setCfg(next) const seeded: Record = {} @@ -532,7 +539,7 @@ export function ToolsetConfigPanel({ toolset, onConfiguredChange }: ToolsetConfi } finally { setLoading(false) } - }, [copy.failedLoad, toolset]) + }, [copy.failedLoad, toolset, profile]) useEffect(() => { void refresh() @@ -574,7 +581,7 @@ export function ToolsetConfigPanel({ toolset, onConfiguredChange }: ToolsetConfi setSelecting(provider.name) try { - const result = await selectToolsetProvider(toolset, provider.name) + const result = await selectToolsetProvider(toolset, provider.name, undefined, profile) // Mirror the backend write locally so dependent UI (model catalog // enablement) tracks the new active backend without a refetch. setCfg(current => @@ -616,7 +623,7 @@ export function ToolsetConfigPanel({ toolset, onConfiguredChange }: ToolsetConfi // refetch the toolset config so is_active / status flip once entitled. async function signInToNousPortal() { try { - const start = await startOAuthLogin('nous') + const start = await startOAuthLogin('nous', profile) if (start.flow !== 'device_code') { notifyError(new Error(`unexpected flow: ${start.flow}`), copy.nousAuthFailed) @@ -644,7 +651,7 @@ export function ToolsetConfigPanel({ toolset, onConfiguredChange }: ToolsetConfi return } - const polled = await pollOAuthSession('nous', start.session_id) + const polled = await pollOAuthSession('nous', start.session_id, profile) if (polled.status === 'approved') { notify({ kind: 'success', title: copy.nousAuthDoneTitle, message: copy.nousAuthDoneMessage }) @@ -676,7 +683,7 @@ export function ToolsetConfigPanel({ toolset, onConfiguredChange }: ToolsetConfi setSelecting(provider.name) try { - await selectToolsetProvider(toolset, provider.name, capability) + await selectToolsetProvider(toolset, provider.name, capability, profile) // Mirror the backend write locally so the Search:/Extract: badges track // the new per-capability backend without a refetch. setCfg(current => @@ -846,6 +853,7 @@ export function ToolsetConfigPanel({ toolset, onConfiguredChange }: ToolsetConfi key={ev.key} onCleared={key => patchEnv(key, false)} onSaved={key => patchEnv(key, true)} + profile={profile} /> )) )} @@ -854,6 +862,7 @@ export function ToolsetConfigPanel({ toolset, onConfiguredChange }: ToolsetConfi installed={provider.status === 'ready'} onComplete={() => void refresh()} postSetupKey={provider.post_setup} + profile={profile} toolset={toolset} /> )} @@ -866,6 +875,7 @@ export function ToolsetConfigPanel({ toolset, onConfiguredChange }: ToolsetConfi {MODEL_CATALOG_TOOLSETS.has(toolset) && ( diff --git a/apps/desktop/src/app/skills/index.test.tsx b/apps/desktop/src/app/skills/index.test.tsx index a22d9aa521299..0982ea1cd7615 100644 --- a/apps/desktop/src/app/skills/index.test.tsx +++ b/apps/desktop/src/app/skills/index.test.tsx @@ -15,19 +15,23 @@ const setToolsetEnabled = vi.fn() const getToolsetConfig = vi.fn() const selectToolsetProvider = vi.fn() const getUsageAnalytics = vi.fn() +const getProfiles = vi.fn() // Partial mock: keep the real module (SkillsView pulls in @/store/profile, // whose import-time subscription calls setApiRequestProfile) and stub only the -// calls we assert on. +// calls we assert on. Args are forwarded so the per-profile scope arg is +// observable. vi.mock('@/hermes', async importOriginal => ({ ...(await importOriginal()), getSkills: () => getSkills(), - getToolsets: () => getToolsets(), + getToolsets: (profile?: null | string) => getToolsets(profile), setSkillEnabled: (name: string, enabled: boolean) => setSkillEnabled(name, enabled), - setToolsetEnabled: (name: string, enabled: boolean) => setToolsetEnabled(name, enabled), - getToolsetConfig: (name: string) => getToolsetConfig(name), + setToolsetEnabled: (name: string, enabled: boolean, profile?: null | string) => + setToolsetEnabled(name, enabled, profile), + getToolsetConfig: (name: string, profile?: null | string) => getToolsetConfig(name, profile), selectToolsetProvider: (toolset: string, provider: string) => selectToolsetProvider(toolset, provider), - getUsageAnalytics: (days: number) => getUsageAnalytics(days) + getUsageAnalytics: (days: number) => getUsageAnalytics(days), + getProfiles: () => getProfiles() })) // Notifications hit nanostores/timers we don't care about here. @@ -81,6 +85,9 @@ beforeEach(() => { setToolsetEnabled.mockResolvedValue({ ok: true, name: 'web', enabled: false }) getToolsetConfig.mockResolvedValue({ has_category: true, active_provider: null, providers: [] }) getUsageAnalytics.mockResolvedValue({ tools: [] }) + // Single profile by default → the scope selector stays hidden (>1 gate), + // so existing tests see unchanged single-profile behavior. + getProfiles.mockResolvedValue({ profiles: [{ name: 'default', is_default: true }] }) }) afterEach(() => { @@ -102,7 +109,8 @@ describe('SkillsView toolset management', () => { fireEvent.click(sw) }) - await waitFor(() => expect(setToolsetEnabled).toHaveBeenCalledWith('web', false)) + await waitFor(() => expect(setToolsetEnabled).toHaveBeenCalled()) + expect(setToolsetEnabled.mock.calls[0].slice(0, 2)).toEqual(['web', false]) }) it('renders toolset titles without leading emoji', async () => { @@ -124,7 +132,46 @@ describe('SkillsView toolset management', () => { await renderSkills() await screen.findByRole('switch', { name: 'Turn Web Search toolset off' }) - await waitFor(() => expect(getToolsetConfig).toHaveBeenCalledWith('web')) + await waitFor(() => expect(getToolsetConfig).toHaveBeenCalled()) + expect(getToolsetConfig.mock.calls[0][0]).toBe('web') + }) + + it('scopes Tools config to the profile chosen in the selector', async () => { + // Two profiles → the "Configuring:" selector renders. Picking a non-active + // profile must re-fetch toolsets scoped to THAT profile. + // jsdom's scrollIntoView is missing/non-functional; Radix Select calls it + // on open. Force a stub so the dropdown can render in the test env. + Element.prototype.scrollIntoView = vi.fn() + getProfiles.mockResolvedValue({ + profiles: [ + { name: 'default', is_default: true }, + { name: 'researcher', is_default: false } + ] + }) + + const { SkillsView } = await import('./index') + await act(async () => { + render( + + + + + + ) + }) + + // The selector appears with >1 profile. + const trigger = await screen.findByRole('combobox') + await act(async () => { + fireEvent.click(trigger) + }) + const option = await screen.findByRole('option', { name: 'researcher' }) + await act(async () => { + fireEvent.click(option) + }) + + // Toolsets refetch scoped to the picked profile. + await waitFor(() => expect(getToolsets).toHaveBeenCalledWith('researcher')) }) it('shows a vision explainer that deep-links to Settings → Models', async () => { diff --git a/apps/desktop/src/app/skills/index.tsx b/apps/desktop/src/app/skills/index.tsx index 9335d293dc384..4dff598d01411 100644 --- a/apps/desktop/src/app/skills/index.tsx +++ b/apps/desktop/src/app/skills/index.tsx @@ -9,10 +9,18 @@ import { CodeEditor } from '@/components/chat/code-editor' import { PageLoader } from '@/components/page-loader' import { Badge } from '@/components/ui/badge' import { Button } from '@/components/ui/button' +import { + Select, + SelectContent, + SelectItem, + SelectTrigger, + SelectValue +} from '@/components/ui/select' import { CountSkeleton } from '@/components/ui/skeleton' import { editLearningNode, getLearningNode, + getProfiles, getSkills, getToolsets, getUsageAnalytics, @@ -68,10 +76,10 @@ const SKILLS_MODES = ['skills', 'toolsets', 'mcp', 'hub'] as const const SKILLS_QUERY_KEY = ['skills-list'] as const const TOOLSETS_QUERY_KEY = ['toolsets-list'] as const -// Optimistic write-through: toggles/bulk/archive repaint instantly; the next -// background refetch reconciles with the backend. +// Optimistic write-through: skill toggles/bulk/archive repaint instantly; the +// next background refetch reconciles with the backend. (Toolsets write through +// the profile-scoped query key directly — see handleToggleToolset.) const setSkills = writeCache(SKILLS_QUERY_KEY) -const setToolsets = writeCache(TOOLSETS_QUERY_KEY) // Per-tool call counts come from a 365-day message scan — heavy, and purely // cosmetic (Toolsets usage badges). Cache the result module-wide with a TTL so @@ -191,6 +199,23 @@ export function SkillsView({ setStatusbarItemGroup: _setStatusbarItemGroup, ...p const [query, setQuery] = useState('') + // Capabilities profile-scope selector: which profile's Tools/MCP config we're + // editing. Defaults to the app-wide active profile; overriding it here lets + // the user configure ANY profile's toolsets/MCP without switching the whole + // app into that profile. null = the active profile (unchanged behavior). + const activeProfile = useStore($activeGatewayProfile) + const [scopeOverride, setScopeOverride] = useState(null) + const scopeProfile = scopeOverride ?? activeProfile ?? null + const scopeKey = normalizeProfileKey(scopeProfile) + + const { data: profilesData } = useQuery({ + queryKey: ['capabilities-profiles'], + queryFn: getProfiles, + staleTime: 60_000 + }) + + const profiles = profilesData?.profiles ?? [] + const { data: skills, isError: skillsFailed, @@ -202,8 +227,8 @@ export function SkillsView({ setStatusbarItemGroup: _setStatusbarItemGroup, ...p }) const { data: toolsets, isError: toolsetsFailed } = useQuery({ - queryKey: TOOLSETS_QUERY_KEY, - queryFn: getToolsets, + queryKey: [...TOOLSETS_QUERY_KEY, scopeKey], + queryFn: () => getToolsets(scopeProfile), staleTime: 0 }) @@ -353,15 +378,20 @@ export function SkillsView({ setStatusbarItemGroup: _setStatusbarItemGroup, ...p } async function handleToggleToolset(toolset: ToolsetInfo, enabled: boolean) { - setToolsets( + const scopedToolsetKey = [...TOOLSETS_QUERY_KEY, scopeKey] + + const writeScoped = (fn: (cur: ToolsetInfo[] | undefined) => ToolsetInfo[] | undefined) => + queryClient.setQueryData(scopedToolsetKey, prev => fn(prev) ?? prev) + + writeScoped( current => current?.map(row => (row.name === toolset.name ? { ...row, enabled, available: enabled } : row)) ?? current ) try { - await setToolsetEnabled(toolset.name, enabled) + await setToolsetEnabled(toolset.name, enabled, scopeProfile) } catch (err) { - setToolsets( + writeScoped( current => current?.map(row => (row.name === toolset.name ? { ...row, enabled: !enabled, available: !enabled } : row)) ?? current @@ -389,8 +419,11 @@ export function SkillsView({ setStatusbarItemGroup: _setStatusbarItemGroup, ...p } for (const row of toolsetTargets) { - await setToolsetEnabled(row.name, enabled) - setToolsets(cur => cur?.map(r => (r.name === row.name ? { ...r, enabled, available: enabled } : r)) ?? cur) + await setToolsetEnabled(row.name, enabled, scopeProfile) + queryClient.setQueryData( + [...TOOLSETS_QUERY_KEY, scopeKey], + cur => cur?.map(r => (r.name === row.name ? { ...r, enabled, available: enabled } : r)) ?? cur + ) done += 1 } @@ -540,6 +573,28 @@ export function SkillsView({ setStatusbarItemGroup: _setStatusbarItemGroup, ...p ) + // Profile-scope selector, shown above the Tools and MCP tabs. Lets the user + // configure ANY profile's capabilities without switching the whole app. + // Only meaningful with >1 profile; hidden otherwise to avoid clutter. + const profileScopeSelector = + profiles.length > 1 ? ( +
+ {t.skills.configuringProfile} + +
+ ) : null + return ( ) : mode === 'mcp' ? ( - +
+ {profileScopeSelector} +
+ +
+
) : (skillsFailed || toolsetsFailed) && (!skills || !toolsets) ? ( +
+ {profileScopeSelector} +
+ {activeToolset && ( - + )} +
+
)} {archiveTarget && ( void; onEd function ToolsetDetail({ toolset, toolCalls, - onConfiguredChange + onConfiguredChange, + profile }: { toolset: ToolsetInfo toolCalls: Record onConfiguredChange: () => void + profile?: null | string }) { const { t } = useI18n() const navigate = useNavigate() @@ -818,7 +890,12 @@ function ToolsetDetail({ )} {toolset.name === 'computer_use' && } {toolset.name === 'terminal' && } - + ) } diff --git a/apps/desktop/src/app/skills/mcp-tab.tsx b/apps/desktop/src/app/skills/mcp-tab.tsx index 59b8a126bae3b..560fd1a0eb61a 100644 --- a/apps/desktop/src/app/skills/mcp-tab.tsx +++ b/apps/desktop/src/app/skills/mcp-tab.tsx @@ -36,7 +36,7 @@ import { $activeGatewayProfile, normalizeProfileKey } from '@/store/profile' import { $activeSessionId } from '@/store/session' import type { HermesConfigRecord } from '@/types/hermes' -import { setHermesConfigCache, useHermesConfigRecord } from '../hooks/use-config-record' +import { hermesConfigCacheWriter, useHermesConfigRecord } from '../hooks/use-config-record' import { useOnProfileSwitch } from '../hooks/use-on-profile-switch' import { DetailPane, ICON_BUTTON, MASTER_DETAIL_WIDE_COLS } from '../master-detail' import { PanelAddButton, PanelEmpty } from '../overlays/panel' @@ -120,8 +120,8 @@ const probeCache = new Map() const serverFingerprint = (server: Record): string => JSON.stringify([server.url, server.command, server.args, server.env, server.headers, server.transport, server.auth]) -const probeKey = (name: string, server: Record | undefined): string => - `${normalizeProfileKey($activeGatewayProfile.get())}::${name}::${serverFingerprint(server ?? {})}` +const probeKey = (name: string, server: Record | undefined, profileKey: string): string => + `${profileKey}::${name}::${serverFingerprint(server ?? {})}` type Probe = McpTestResult | 'probing' @@ -330,11 +330,20 @@ function scanServerBlocks(text: string): ServerBlock[] { return blocks } -export function McpTab({ gateway }: { gateway: HermesGateway | null }) { +export function McpTab({ gateway, profile }: { gateway: HermesGateway | null; profile?: null | string }) { const { t } = useI18n() const m = t.settings.mcp const activeSessionId = useStore($activeSessionId) + // The profile this tab configures: the Capabilities profile-scope selector's + // choice (`profile`) when set, otherwise the app-wide active profile. Every + // fetch/save below is scoped to it, and it keys the config/catalog/probe + // caches so switching the selector refetches and never shows another + // profile's servers (AGENTS.md scope-in-key). When no override is passed this + // resolves to $activeGatewayProfile, so behavior is identical to before. + const appProfile = useStore($activeGatewayProfile) + const scopeProfileKey = normalizeProfileKey(profile ?? appProfile) + // Shared config cache (see use-config-record): revisiting the tab paints the // cached record instantly; mutations write through `setConfig` and stay // visible to the other settings surfaces. @@ -346,9 +355,9 @@ export function McpTab({ gateway }: { gateway: HermesGateway | null }) { refetch: refetchConfig, dataUpdatedAt: configUpdatedAt, errorUpdatedAt: configErroredAt - } = useHermesConfigRecord() + } = useHermesConfigRecord(profile) - const setConfig = setHermesConfigCache + const setConfig = hermesConfigCacheWriter(profile) // True from a profile switch until the config query resettles for the new // profile. Until then `config` (and thus `servers`) still holds profile A's @@ -407,11 +416,13 @@ export function McpTab({ gateway }: { gateway: HermesGateway | null }) { // enrichment below), so switching between them never re-requests. const [leftView, setLeftView] = useState<'catalog' | 'servers'>('servers') - // Key by active profile — installed/enabled badges are per-profile, so sharing - // one cache across profiles would flash the previous profile's state on switch. + // Key by the SCOPED profile — installed/enabled badges are per-profile, so + // sharing one cache across profiles would flash the previous profile's state + // on switch. When no selector override is set this is the active profile, + // identical to before. const catalogQuery = useQuery({ - queryKey: [...MCP_CATALOG_KEY, normalizeProfileKey(useStore($activeGatewayProfile))], - queryFn: getMcpCatalog, + queryKey: [...MCP_CATALOG_KEY, scopeProfileKey], + queryFn: () => getMcpCatalog(profile ?? undefined), staleTime: 5 * 60_000 }) @@ -538,11 +549,11 @@ export function McpTab({ gateway }: { gateway: HermesGateway | null }) { const runProbe = async (serverName: string) => { const epoch = profileEpoch.current - const key = probeKey(serverName, servers[serverName]) + const key = probeKey(serverName, servers[serverName], scopeProfileKey) setProbes(current => ({ ...current, [serverName]: 'probing' })) try { - const result = await testMcpServer(serverName) + const result = await testMcpServer(serverName, profile ?? undefined) // Drop the result if the profile changed mid-probe — it belongs to A. if (profileEpoch.current !== epoch) { @@ -573,8 +584,8 @@ export function McpTab({ gateway }: { gateway: HermesGateway | null }) { try { const flow = await completeMcpDesktopOAuth({ serverName, - start: authMcpServer, - status: getMcpOAuthFlow, + start: name => authMcpServer(name, profile ?? undefined), + status: flowId => getMcpOAuthFlow(flowId, profile ?? undefined), openExternal: url => window.hermesDesktop.openExternal(url) }) @@ -589,7 +600,7 @@ export function McpTab({ gateway }: { gateway: HermesGateway | null }) { // Cache under the POST-auth fingerprint (auth: oauth) on success — that's // the config the mount effect will read back, so it hits this entry. const probedConfig = result.ok ? { ...servers[serverName], auth: 'oauth' } : servers[serverName] - probeCache.set(probeKey(serverName, probedConfig), { at: Date.now(), result }) + probeCache.set(probeKey(serverName, probedConfig, scopeProfileKey), { at: Date.now(), result }) if (result.ok) { // The endpoint persisted `auth: oauth` — mirror it locally. @@ -640,7 +651,7 @@ export function McpTab({ gateway }: { gateway: HermesGateway | null }) { continue } - const cached = probeCache.get(probeKey(serverName, server)) + const cached = probeCache.get(probeKey(serverName, server, scopeProfileKey)) if (cached && Date.now() - cached.at < PROBE_TTL_MS) { setProbes(current => ({ ...current, [serverName]: cached.result })) @@ -674,7 +685,7 @@ export function McpTab({ gateway }: { gateway: HermesGateway | null }) { // caller must skip its post-await writes. const persist = async (nextServers: McpServers): Promise => { const epoch = profileEpoch.current - await saveMcpServers(nextServers) + await saveMcpServers(nextServers, profile ?? undefined) if (profileEpoch.current !== epoch) { return false @@ -991,7 +1002,12 @@ export function McpTab({ gateway }: { gateway: HermesGateway | null }) {
{leftView === 'catalog' ? ( - + ) : ( <> {names.map(serverName => { @@ -1329,11 +1345,13 @@ function CatalogTag({ children }: { children: string }) { function McpCatalog({ entries, loading, - onInstalled + onInstalled, + profile }: { entries: McpCatalogEntry[] loading: boolean onInstalled: () => void + profile?: null | string }) { const { t } = useI18n() const m = t.settings.mcp @@ -1361,7 +1379,7 @@ function McpCatalog({ setInstalling(entry.name) try { - const res = await installMcpCatalogEntry(entry.name, draft) + const res = await installMcpCatalogEntry(entry.name, draft, profile ?? undefined) // Git-backed entries clone in the background — keep the row busy and poll // the action to completion before refetching / re-enabling, so a re-click @@ -1369,7 +1387,7 @@ function McpCatalog({ // exit is a real failure — surface it instead of a false success. if (res.background && res.action) { for (;;) { - const status = await getActionStatus(res.action, 1) + const status = await getActionStatus(res.action, 1, profile ?? undefined) if (!status.running) { if (status.exit_code !== 0) { diff --git a/apps/desktop/src/hermes.ts b/apps/desktop/src/hermes.ts index b3722926d2786..240fdf01894b1 100644 --- a/apps/desktop/src/hermes.ts +++ b/apps/desktop/src/hermes.ts @@ -833,9 +833,9 @@ export function getHermesConfig(profile?: string): Promise { }) } -export function getHermesConfigRecord(): Promise { +export function getHermesConfigRecord(profile?: null | string): Promise { return window.hermesDesktop.api({ - ...profileScoped(), + ...profileScoped(profile), path: '/api/config' }) } @@ -888,9 +888,9 @@ export function getEnvVars(): Promise> { }) } -export function setEnvVar(key: string, value: string): Promise<{ ok: boolean }> { +export function setEnvVar(key: string, value: string, profile?: null | string): Promise<{ ok: boolean }> { return window.hermesDesktop.api<{ ok: boolean }>({ - ...profileScoped(), + ...profileScoped(profile), path: '/api/env', method: 'PUT', body: { key, value } @@ -950,18 +950,18 @@ export function deleteCustomEndpoint(id: string): Promise { +export function deleteEnvVar(key: string, profile?: null | string): Promise<{ ok: boolean }> { return window.hermesDesktop.api<{ ok: boolean }>({ - ...profileScoped(), + ...profileScoped(profile), path: '/api/env', method: 'DELETE', body: { key } }) } -export function revealEnvVar(key: string): Promise<{ key: string; value: string }> { +export function revealEnvVar(key: string, profile?: null | string): Promise<{ key: string; value: string }> { return window.hermesDesktop.api<{ key: string; value: string }>({ - ...profileScoped(), + ...profileScoped(profile), path: '/api/env/reveal', method: 'POST', body: { key } @@ -983,9 +983,9 @@ export function disconnectOAuthProvider(providerId: string): Promise<{ ok: boole }) } -export function startOAuthLogin(providerId: string): Promise { +export function startOAuthLogin(providerId: string, profile?: null | string): Promise { return window.hermesDesktop.api({ - ...profileScoped(), + ...profileScoped(profile), path: `/api/providers/oauth/${encodeURIComponent(providerId)}/start`, method: 'POST', body: {} @@ -1001,9 +1001,13 @@ export function submitOAuthCode(providerId: string, sessionId: string, code: str }) } -export function pollOAuthSession(providerId: string, sessionId: string): Promise { +export function pollOAuthSession( + providerId: string, + sessionId: string, + profile?: null | string +): Promise { return window.hermesDesktop.api({ - ...profileScoped(), + ...profileScoped(profile), path: `/api/providers/oauth/${encodeURIComponent(providerId)}/poll/${encodeURIComponent(sessionId)}` }) } @@ -1113,9 +1117,9 @@ export interface McpOAuthFlow { /** Connect to the server, list its tools, disconnect. Slow (spawns/handshakes * for real) — well past the 15s default fetch timeout. */ -export function testMcpServer(name: string): Promise { +export function testMcpServer(name: string, profile?: null | string): Promise { return window.hermesDesktop.api({ - ...profileScoped(), + ...profileScoped(profile), path: `/api/mcp/servers/${encodeURIComponent(name)}/test`, method: 'POST', timeoutMs: 60_000 @@ -1125,9 +1129,12 @@ export function testMcpServer(name: string): Promise { /** Replace the whole `mcp_servers` map (the mcp.json editor's save). Unlike * `saveHermesConfig`, this REPLACES rather than deep-merges, so deletes, * re-enables (dropping `enabled: false`), and removed nested fields persist. */ -export function saveMcpServers(servers: Record>): Promise<{ ok: boolean }> { +export function saveMcpServers( + servers: Record>, + profile?: null | string +): Promise<{ ok: boolean }> { return window.hermesDesktop.api<{ ok: boolean }>({ - ...profileScoped(), + ...profileScoped(profile), path: '/api/mcp/servers', method: 'PUT', body: { servers } @@ -1135,63 +1142,76 @@ export function saveMcpServers(servers: Record>) } /** Start an MCP OAuth flow and return the authorization URL. */ -export function authMcpServer(name: string): Promise { +export function authMcpServer(name: string, profile?: null | string): Promise { return window.hermesDesktop.api({ - ...profileScoped(), + ...profileScoped(profile), path: `/api/mcp/servers/${encodeURIComponent(name)}/auth`, method: 'POST', timeoutMs: 60_000 }) } -export function getMcpOAuthFlow(flowId: string): Promise { +export function getMcpOAuthFlow(flowId: string, profile?: null | string): Promise { return window.hermesDesktop.api({ - ...profileScoped(), + ...profileScoped(profile), path: `/api/mcp/oauth/flows/${encodeURIComponent(flowId)}` }) } /** Cancel an in-flight MCP OAuth flow server-side, freeing the per-server * "already in progress" slot so a retry doesn't 409. */ -export function cancelMcpOAuthFlow(flowId: string): Promise<{ ok: boolean; status: string }> { +export function cancelMcpOAuthFlow( + flowId: string, + profile?: null | string +): Promise<{ ok: boolean; status: string }> { return window.hermesDesktop.api<{ ok: boolean; status: string }>({ - ...profileScoped(), + ...profileScoped(profile), path: `/api/mcp/oauth/flows/${encodeURIComponent(flowId)}`, method: 'DELETE' }) } -export function getToolsets(): Promise { +// The optional trailing `profile` on every capability fetcher below is the +// Capabilities view's profile-scope override: it lets the Skills/Tools/MCP +// panels configure ANY profile without swapping the app-wide active profile. +// Omitting it (every pre-existing caller) means `profileScoped(undefined)` +// falls back to the app-wide `_apiProfile`, so behavior is byte-identical. +export function getToolsets(profile?: null | string): Promise { return window.hermesDesktop.api({ - ...profileScoped(), + ...profileScoped(profile), path: '/api/tools/toolsets' }) } export function setToolsetEnabled( name: string, - enabled: boolean + enabled: boolean, + profile?: null | string ): Promise<{ ok: boolean; name: string; enabled: boolean }> { return window.hermesDesktop.api<{ ok: boolean; name: string; enabled: boolean }>({ - ...profileScoped(), + ...profileScoped(profile), path: `/api/tools/toolsets/${encodeURIComponent(name)}`, method: 'PUT', body: { enabled } }) } -export function getToolsetConfig(name: string): Promise { +export function getToolsetConfig(name: string, profile?: null | string): Promise { return window.hermesDesktop.api({ - ...profileScoped(), + ...profileScoped(profile), path: `/api/tools/toolsets/${encodeURIComponent(name)}/config` }) } -export function getToolsetModels(name: string, provider?: string): Promise { +export function getToolsetModels( + name: string, + provider?: string, + profile?: null | string +): Promise { const suffix = provider ? `?provider=${encodeURIComponent(provider)}` : '' return window.hermesDesktop.api({ - ...profileScoped(), + ...profileScoped(profile), path: `/api/tools/toolsets/${encodeURIComponent(name)}/models${suffix}` }) } @@ -1199,10 +1219,11 @@ export function getToolsetModels(name: string, provider?: string): Promise { return window.hermesDesktop.api<{ ok: boolean; name: string; model: string }>({ - ...profileScoped(), + ...profileScoped(profile), path: `/api/tools/toolsets/${encodeURIComponent(name)}/model`, method: 'PUT', body: { model, provider } @@ -1226,19 +1247,24 @@ export interface SelectToolsetProviderResponse { export function selectToolsetProvider( name: string, provider: string, - capability?: 'search' | 'extract' + capability?: 'search' | 'extract', + profile?: null | string ): Promise { return window.hermesDesktop.api({ - ...profileScoped(), + ...profileScoped(profile), path: `/api/tools/toolsets/${encodeURIComponent(name)}/provider`, method: 'PUT', body: capability ? { provider, capability } : { provider } }) } -export function runToolsetPostSetup(name: string, key: string): Promise { +export function runToolsetPostSetup( + name: string, + key: string, + profile?: null | string +): Promise { return window.hermesDesktop.api({ - ...profileScoped(), + ...profileScoped(profile), path: `/api/tools/toolsets/${encodeURIComponent(name)}/post-setup`, method: 'POST', body: { key } @@ -1711,9 +1737,9 @@ export function checkHermesUpdate(force = false): Promise { +export function getActionStatus(name: string, lines = 200, profile?: null | string): Promise { return window.hermesDesktop.api({ - ...profileScoped(), + ...profileScoped(profile), path: `/api/actions/${encodeURIComponent(name)}/status?lines=${Math.max(1, lines)}` }) } @@ -1873,19 +1899,20 @@ export function setMcpServerEnabled(name: string, enabled: boolean): Promise<{ o }) } -export function getMcpCatalog(): Promise { +export function getMcpCatalog(profile?: null | string): Promise { return window.hermesDesktop.api({ - ...profileScoped(), + ...profileScoped(profile), path: '/api/mcp/catalog' }) } export function installMcpCatalogEntry( name: string, - env: Record = {} + env: Record = {}, + profile?: null | string ): Promise<{ ok: boolean; name?: string; pid?: number; action?: string; background?: boolean }> { return window.hermesDesktop.api<{ ok: boolean; name?: string; pid?: number; action?: string; background?: boolean }>({ - ...profileScoped(), + ...profileScoped(profile), path: '/api/mcp/catalog/install', method: 'POST', body: { name, env, enable: true }, diff --git a/apps/desktop/src/i18n/en.ts b/apps/desktop/src/i18n/en.ts index b38cfb88dcfce..d8bafbb7bbae0 100644 --- a/apps/desktop/src/i18n/en.ts +++ b/apps/desktop/src/i18n/en.ts @@ -1014,6 +1014,7 @@ export const en: Translations = { skills: { tabSkills: 'Skills', tabToolsets: 'Tools', + configuringProfile: 'Configuring:', tabMcp: 'MCP', tabHub: 'Browse Hub', all: 'All', diff --git a/apps/desktop/src/i18n/types.ts b/apps/desktop/src/i18n/types.ts index c314d85a9a0cb..fb994d6de9a5c 100644 --- a/apps/desktop/src/i18n/types.ts +++ b/apps/desktop/src/i18n/types.ts @@ -877,6 +877,7 @@ export interface Translations { skills: { tabSkills: string tabToolsets: string + configuringProfile: string tabMcp: string tabHub: string all: string diff --git a/apps/desktop/src/i18n/zh.ts b/apps/desktop/src/i18n/zh.ts index f5931a06adc7f..69c9c0680ad8d 100644 --- a/apps/desktop/src/i18n/zh.ts +++ b/apps/desktop/src/i18n/zh.ts @@ -1212,6 +1212,7 @@ export const zh: Translations = { skills: { tabSkills: '技能', tabToolsets: '工具集', + configuringProfile: '正在配置:', tabMcp: 'MCP', tabHub: '浏览技能中心', all: '全部', From b9fa46b5a57af8f21af5529df2ad342318641940 Mon Sep 17 00:00:00 2001 From: "hermes-seaeye[bot]" <307254004+hermes-seaeye[bot]@users.noreply.github.com> Date: Sat, 15 Aug 2026 00:39:34 +0000 Subject: [PATCH 002/376] fmt(js): `npm run fix` on merge (#86559) Co-authored-by: github-actions[bot] --- apps/desktop/src/app/skills/index.tsx | 102 ++++++++++++-------------- apps/desktop/src/hermes.ts | 5 +- 2 files changed, 49 insertions(+), 58 deletions(-) diff --git a/apps/desktop/src/app/skills/index.tsx b/apps/desktop/src/app/skills/index.tsx index 4dff598d01411..11917ddbff4dd 100644 --- a/apps/desktop/src/app/skills/index.tsx +++ b/apps/desktop/src/app/skills/index.tsx @@ -9,13 +9,7 @@ import { CodeEditor } from '@/components/chat/code-editor' import { PageLoader } from '@/components/page-loader' import { Badge } from '@/components/ui/badge' import { Button } from '@/components/ui/button' -import { - Select, - SelectContent, - SelectItem, - SelectTrigger, - SelectValue -} from '@/components/ui/select' +import { Select, SelectContent, SelectItem, SelectTrigger, SelectValue } from '@/components/ui/select' import { CountSkeleton } from '@/components/ui/skeleton' import { editLearningNode, @@ -696,53 +690,53 @@ export function SkillsView({ setStatusbarItemGroup: _setStatusbarItemGroup, ...p {profileScopeSelector}
- $toolsetsSortDesc.set(!$toolsetsSortDesc.get()))} - right={} - /> - } - > - {visibleToolsets.map(toolset => { - const label = toolsetDisplayLabel(toolset) - const calls = toolCalls ? toolsetCalls(toolset, toolCalls) : null - - return ( - - ) : calls > 0 ? ( - `×${compactNumber(calls)}` - ) : ( - `${toolNames(toolset).length} tools` - ) - } - onSelect={() => setSelectedToolset(toolset.name)} - onToggle={checked => void handleToggleToolset(toolset, checked)} - subtitle={asText(toolset.description)} - title={label} - toggleLabel={t.skills.toggleToolset(label, !toolset.enabled)} - /> - ) - })} - - - {activeToolset && ( - - )} - - + $toolsetsSortDesc.set(!$toolsetsSortDesc.get()))} + right={} + /> + } + > + {visibleToolsets.map(toolset => { + const label = toolsetDisplayLabel(toolset) + const calls = toolCalls ? toolsetCalls(toolset, toolCalls) : null + + return ( + + ) : calls > 0 ? ( + `×${compactNumber(calls)}` + ) : ( + `${toolNames(toolset).length} tools` + ) + } + onSelect={() => setSelectedToolset(toolset.name)} + onToggle={checked => void handleToggleToolset(toolset, checked)} + subtitle={asText(toolset.description)} + title={label} + toggleLabel={t.skills.toggleToolset(label, !toolset.enabled)} + /> + ) + })} + + + {activeToolset && ( + + )} + +
)} diff --git a/apps/desktop/src/hermes.ts b/apps/desktop/src/hermes.ts index 240fdf01894b1..d19b958603cca 100644 --- a/apps/desktop/src/hermes.ts +++ b/apps/desktop/src/hermes.ts @@ -1160,10 +1160,7 @@ export function getMcpOAuthFlow(flowId: string, profile?: null | string): Promis /** Cancel an in-flight MCP OAuth flow server-side, freeing the per-server * "already in progress" slot so a retry doesn't 409. */ -export function cancelMcpOAuthFlow( - flowId: string, - profile?: null | string -): Promise<{ ok: boolean; status: string }> { +export function cancelMcpOAuthFlow(flowId: string, profile?: null | string): Promise<{ ok: boolean; status: string }> { return window.hermesDesktop.api<{ ok: boolean; status: string }>({ ...profileScoped(profile), path: `/api/mcp/oauth/flows/${encodeURIComponent(flowId)}`, From ed4f50de51d3adfb8f99d96dc18881d09a9cb859 Mon Sep 17 00:00:00 2001 From: Adolanium <94890352+Adolanium@users.noreply.github.com> Date: Sat, 15 Aug 2026 03:09:52 +0530 Subject: [PATCH 003/376] fix(send_message): hand unresolved cron and react targets to the adapter again Restores pass-through behavior for cron delivery and react/unreact that was lost when d409f6748 routed them through resolve_send_target. Stored cron job targets the channel directory doesn't recognize (e.g. telegram:ops-room on a fresh install, photon group GUIDs) used to go to the adapter verbatim; after d409f6748 they were silently dropped. Same for react on platform-native ids. Adds an opt-in pass_unresolved_references flag to resolve_send_target, passed only by cron and react. Model-facing send tool stays strict. Plugin platforms with a parser stay strict for all callers. The optional validator still has the final say over passed-through ids. Follow-up fixes on salvage: - Update test_cron_relay_delivery_guards.py mock lambdas to accept **kw (file added to main after PR branch point; lambdas didn't accept the new keyword argument) - Consolidate duplicated pass-through blocks into _pass_through_unresolved local helper Fixes #85128 Co-authored-by: Adolanium --- cron/scheduler.py | 7 +- tests/cron/test_cron_relay_delivery_guards.py | 4 +- tests/cron/test_scheduler.py | 18 +++ tests/tools/test_send_message_target_parse.py | 104 ++++++++++++++++++ tools/send_message_tool.py | 44 +++++++- 5 files changed, 168 insertions(+), 9 deletions(-) diff --git a/cron/scheduler.py b/cron/scheduler.py index 79b5210cc1aa8..345a098b96d8a 100644 --- a/cron/scheduler.py +++ b/cron/scheduler.py @@ -1859,8 +1859,13 @@ def _resolve_single_delivery_target(job: dict, deliver_value: str) -> Optional[d ) prepare_send_message_platforms() + # pass_unresolved_references: stored jobs have no model in the loop to react + # to a resolution error, and a target the directory doesn't know + # (fresh install, platform-native id) used to be handed to the + # adapter as written. Dropping it here silently loses the job's + # output. chat_id, thread_id, resolution_error = resolve_send_target( - platform_key, rest + platform_key, rest, pass_unresolved_references=True ) if resolution_error: logger.warning( diff --git a/tests/cron/test_cron_relay_delivery_guards.py b/tests/cron/test_cron_relay_delivery_guards.py index 71a062195cb48..838e3cd33b1c9 100644 --- a/tests/cron/test_cron_relay_delivery_guards.py +++ b/tests/cron/test_cron_relay_delivery_guards.py @@ -79,7 +79,7 @@ def test_explicit_target_no_reattach_when_chat_is_home(self, monkeypatch): "tools.send_message_tool.prepare_send_message_platforms", lambda: None) monkeypatch.setattr( "tools.send_message_tool.resolve_send_target", - lambda platform, rest: (rest, None, None)) + lambda platform, rest, **kw: (rest, None, None)) job = {"origin": {"platform": "slack", "chat_id": "D0BJTDCSR7C", "thread_id": SYNTH}} target = _resolve_single_delivery_target(job, "slack:D0BJTDCSR7C") @@ -92,7 +92,7 @@ def test_explicit_target_reattach_kept_for_non_home_chat(self, monkeypatch): "tools.send_message_tool.prepare_send_message_platforms", lambda: None) monkeypatch.setattr( "tools.send_message_tool.resolve_send_target", - lambda platform, rest: (rest, None, None)) + lambda platform, rest, **kw: (rest, None, None)) job = {"origin": {"platform": "slack", "chat_id": "C0AGENERAL", "thread_id": "1755040000.000100"}} target = _resolve_single_delivery_target(job, "slack:C0AGENERAL") diff --git a/tests/cron/test_scheduler.py b/tests/cron/test_scheduler.py index 199c7b3892525..b8e5cce652636 100644 --- a/tests/cron/test_scheduler.py +++ b/tests/cron/test_scheduler.py @@ -254,6 +254,24 @@ def test_raw_id_not_mangled_when_directory_returns_none(self): "thread_id": None, } + def test_unresolved_target_still_delivered_as_written(self): + """A stored job's platform-native target keeps delivering when neither + parser nor directory recognizes it. Routing cron through + resolve_send_target turned these into a warning plus a silently + dropped delivery; pass_unresolved_references hands the raw id to the adapter + again.""" + job = {"deliver": "telegram:ops-room"} + with patch( + "gateway.channel_directory.resolve_channel_name", + return_value=None, + ): + result = _resolve_delivery_target(job) + assert result == { + "platform": "telegram", + "chat_id": "ops-room", + "thread_id": None, + } + def test_list_form_deliver_is_normalized(self, monkeypatch): """deliver=['telegram'] (Python list) should resolve like 'telegram' string. diff --git a/tests/tools/test_send_message_target_parse.py b/tests/tools/test_send_message_target_parse.py index 07e6ac28fba70..d698070fa6e9d 100644 --- a/tests/tools/test_send_message_target_parse.py +++ b/tests/tools/test_send_message_target_parse.py @@ -206,3 +206,107 @@ def test_unresolved_builtin_target_keeps_directory_error() -> None: } send_mock.assert_not_awaited() + + +def test_unresolved_builtin_target_passes_through_when_requested() -> None: + """Cron and react keep the old pass-through behavior for unresolved + built-in targets: with no model in the loop to react to an error, the + raw id must reach the adapter, as it did before resolve_send_target + took over these callers.""" + from tools.send_message_tool import resolve_send_target + + with patch("gateway.channel_directory.resolve_channel_name", return_value=None): + chat_id, thread_id, error = resolve_send_target( + "telegram", "ops-room", pass_unresolved_references=True + ) + + assert error is None + assert chat_id == "ops-room" + assert thread_id is None + + +def test_unresolved_builtin_target_still_errors_for_the_model_tool() -> None: + """The model-facing default stays strict: unresolved targets error with a hint.""" + from tools.send_message_tool import resolve_send_target + + with patch("gateway.channel_directory.resolve_channel_name", return_value=None): + chat_id, _thread_id, error = resolve_send_target("telegram", "ops-room") + + assert chat_id is None + assert error is not None + + +def test_photon_group_guid_passes_through_when_requested() -> None: + """The reported regression case: a photon group GUID matches no parser + pattern (only DM GUIDs have an explicit rule) and no directory entry. + Photon registers as a parser-less plugin platform, so the pass-through + applies once platforms are prepared.""" + from tools.send_message_tool import ( + prepare_send_message_platforms, + resolve_send_target, + ) + + prepare_send_message_platforms() + with patch("gateway.channel_directory.resolve_channel_name", return_value=None): + chat_id, thread_id, error = resolve_send_target( + "photon", "iMessage;+;chat527148912345", pass_unresolved_references=True + ) + + assert error is None + assert chat_id == "iMessage;+;chat527148912345" + assert thread_id is None + + +def test_parserless_plugin_target_passes_through_when_requested() -> None: + """A plugin platform that declares no parser has no explicit syntax at + all, so passing the raw id through is the only way cron can target it.""" + from gateway.platform_registry import PlatformEntry, platform_registry + from tools.send_message_tool import resolve_send_target + + platform_name = "opaque-cron-fallback-test" + entry = PlatformEntry( + name=platform_name, + label="Opaque cron fallback test", + adapter_factory=lambda cfg: None, + check_fn=lambda: True, + ) + platform_registry.register(entry) + try: + with patch("gateway.channel_directory.resolve_channel_name", return_value=None): + chat_id, thread_id, error = resolve_send_target( + platform_name, "dm:panyaozhen", pass_unresolved_references=True + ) + finally: + platform_registry.unregister(platform_name) + + assert error is None + assert chat_id == "dm:panyaozhen" + assert thread_id is None + + +def test_plugin_parser_stays_authoritative_despite_fallback() -> None: + """A plugin that DOES declare a parser stays strict for every caller: + its parser is the authority on native syntax, so an unrecognized + target errors even with pass_unresolved_references.""" + from gateway.platform_registry import PlatformEntry, platform_registry + from tools.send_message_tool import resolve_send_target + + platform_name = "opaque-parser-strict-test" + entry = PlatformEntry( + name=platform_name, + label="Opaque parser strict test", + adapter_factory=lambda cfg: None, + check_fn=lambda: True, + parse_target_ref_fn=lambda ref: None, + ) + platform_registry.register(entry) + try: + with patch("gateway.channel_directory.resolve_channel_name", return_value=None): + chat_id, _thread_id, error = resolve_send_target( + platform_name, "dm:panyaozhen", pass_unresolved_references=True + ) + finally: + platform_registry.unregister(platform_name) + + assert chat_id is None + assert error is not None diff --git a/tools/send_message_tool.py b/tools/send_message_tool.py index cf93756121c68..403d49fdcc541 100644 --- a/tools/send_message_tool.py +++ b/tools/send_message_tool.py @@ -296,8 +296,11 @@ def _handle_react(args, remove=False): chat_id = None prepare_send_message_platforms() if target_ref: + # Platform-native ids (e.g. photon space GUIDs like 'any;-;+1555...') + # match no parser pattern and no directory entry, so hand them to + # the adapter unchanged; it validates them. chat_id, _thread_id, resolution_error = resolve_send_target( - platform_name, target_ref + platform_name, target_ref, pass_unresolved_references=True ) if resolution_error: return tool_error(resolution_error) @@ -622,14 +625,26 @@ def _parse_target_ref(platform_name: str, target_ref: str): def resolve_send_target( - platform_name: str, target_ref: str + platform_name: str, target_ref: str, *, pass_unresolved_references: bool = False ) -> tuple[str | None, str | None, str | None]: - """Resolve one send target identically for model/CLI/cron surfaces. + """Resolve one send target the same way for every caller (model tool, CLI, cron). Channel-directory IDs are trusted. Plugin platforms must explicitly parse - native target syntax; unresolved strings never receive an opaque fallback. - The optional validator is the final authority over parser-normalized and - directory-resolved IDs. + native target syntax; for the model-facing send tool (the default), a + target that can't be resolved is an error — the model can read the error + and pick a listed target instead. + + ``pass_unresolved_references=True`` restores the old pass-through behavior for + callers that have no model in the loop (cron delivering a stored job's + output, react/unreact on platform-native message ids): if the target + can't be resolved and the platform is built in, or is a plugin platform + that declares no parser, the string is handed to the adapter exactly as + written and the adapter decides whether it's valid. A plugin platform + that DOES declare a parser stays strict for every caller — its parser is + the authority on native syntax. + + The optional validator has the final say over parser-normalized, + directory-resolved, and passed-through IDs alike. """ from gateway.config import Platform from gateway.platform_registry import platform_registry @@ -715,13 +730,30 @@ def _validate(candidate: str) -> str | None: is_builtin = platform_name in {member.value for member in Platform} if entry is None and not is_builtin: return None, None, f"Unknown or unregistered plugin platform: {platform_name}" + + def _pass_through_unresolved(): + """Hand the raw target to the adapter unchanged (it validates).""" + error = _validate(target_ref) + if error: + return None, None, error + logger.debug( + "Handing unresolved target '%s' to the %s adapter unchanged " + "(the adapter validates it)", + target_ref, platform_name, + ) + return target_ref, None, None + if entry is not None and entry.source == "plugin" and not is_builtin: + if pass_unresolved_references and entry.parse_target_ref_fn is None: + return _pass_through_unresolved() return ( None, None, f"Could not resolve '{target_ref}' on {platform_name}. " "The plugin parser did not recognize it and no channel-directory entry matched.", ) + if pass_unresolved_references: + return _pass_through_unresolved() hint = ( "Try using a numeric channel ID instead." if resolution_failed From b4da6b15e132e49b64131888f2f807a4d6f652b0 Mon Sep 17 00:00:00 2001 From: "Daniel V. Baecker" <295684751+dvbaecker@users.noreply.github.com> Date: Tue, 11 Aug 2026 12:46:17 +0100 Subject: [PATCH 004/376] fix(telegram): hold inbound messages across disconnect instead of destroying them MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The disconnect drop-guard (#55971) correctly prevents dispatch into a torn-down session. Destroying the event was wrong: by enqueue/flush time python-telegram-bot has already acked the update and advanced the polling offset, so Telegram never redelivers. Result: silent permanent loss, no log, no error. Hold inbound events (text/photo/media-group) when the drop-guard fires, salvage pending batch maps on teardown, cancel+await the redispatch task in the delivery cancel map (lifecycle-tracked), and redispatch from _mark_connected after reconnect. Cap the hold queue (default 64), dedupe by object identity, discard on non-retryable fatal. Cancel-after-pop in flush paths also holds. Distinct from #72037 (cancel-after-pop during follow-up supersession) and #81528 (boundary discard). Tests use delay=0 and entered/release Events — no wall-clock races; includes production terminal-step coverage. --- plugins/platforms/telegram/adapter.py | 187 +++++++++++++- tests/gateway/test_telegram_text_batching.py | 255 +++++++++++++++++++ 2 files changed, 431 insertions(+), 11 deletions(-) diff --git a/plugins/platforms/telegram/adapter.py b/plugins/platforms/telegram/adapter.py index f31cd85e6a5ca..af3578aa7132e 100644 --- a/plugins/platforms/telegram/adapter.py +++ b/plugins/platforms/telegram/adapter.py @@ -657,6 +657,9 @@ class TelegramAdapter(BasePlatformAdapter): # When a chunk is near this limit, a continuation is almost certain. _SPLIT_THRESHOLD = 4000 MEDIA_GROUP_WAIT_SECONDS = 0.8 + # Cap on inbound events held across a disconnect/reconnect window. + # Bounds memory during extended outages; oldest events are dropped first. + HELD_INBOUND_MAX = 64 _GENERAL_TOPIC_THREAD_ID = "1" # Telegram's edit_message applies MarkdownV2 formatting only on the @@ -789,6 +792,12 @@ def __init__(self, config: PlatformConfig): self._pending_text_batches: Dict[str, MessageEvent] = {} self._pending_text_batch_tasks: Dict[str, asyncio.Task] = {} self._drop_delayed_deliveries = False + # Inbound events held across disconnect. PTB advances the polling offset + # before our enqueue/flush drop-guard runs, so Telegram will not + # redeliver — destroying the event is silent permanent loss. Hold and + # redispatch on reconnect instead (see _hold_inbound_event). + self._held_inbound_events: List[MessageEvent] = [] + self._held_inbound_redispatch_task: Optional[asyncio.Task] = None self._polling_error_task: Optional[asyncio.Task] = None self._polling_conflict_count: int = 0 self._polling_conflict_recovery_generation: Optional[int] = None @@ -911,6 +920,20 @@ def __init__(self, config: PlatformConfig): def _mark_connected(self) -> None: self._drop_delayed_deliveries = False super()._mark_connected() + # Drain anything held while we were down. PTB will not redeliver — + # these events exist only in our hold queue now. + if not getattr(self, "_held_inbound_events", None): + return + try: + loop = asyncio.get_running_loop() + except RuntimeError: + return + prior = getattr(self, "_held_inbound_redispatch_task", None) + # Single tracked task so disconnect can cancel+await it (teknium + # lifecycle rule from #72037 review: no untracked dispatch). + self._held_inbound_redispatch_task = loop.create_task( + self._redispatch_held_inbound(prior=prior) + ) def _mark_disconnected(self) -> None: self._drop_delayed_deliveries = True @@ -919,16 +942,119 @@ def _mark_disconnected(self) -> None: def _set_fatal_error(self, code: str, message: str, *, retryable: bool) -> None: self._drop_delayed_deliveries = True super()._set_fatal_error(code, message, retryable=retryable) + # Permanent fatal: no reconnect will drain the queue. Surface the loss + # explicitly instead of letting held messages die silently with the process. + if not retryable and getattr(self, "_held_inbound_events", None): + n = len(self._held_inbound_events) + logger.warning( + "[Telegram] Non-retryable fatal (%s); discarding %d held inbound message(s)", + code, + n, + ) + self._held_inbound_events.clear() def _should_drop_delayed_delivery(self) -> bool: - """True once teardown/fatal-error started — delayed flushes must drop. + """True once teardown/fatal-error started — delayed flushes must not dispatch. Buffered text/photo/media-group flushes sit behind an asyncio.sleep(). If disconnect wins the race, dispatching them spawns an agent on a torn-down session, producing stale/duplicate deliveries. + + Callers must NOT destroy the event when this returns True: PTB has + already advanced the polling offset, so Telegram will never redeliver. + Use ``_hold_inbound_event`` and redispatch on reconnect. """ return bool(getattr(self, "_drop_delayed_deliveries", False)) + def _hold_inbound_event(self, event: "MessageEvent", *, where: str) -> None: + """Preserve an inbound event that cannot be dispatched right now. + + The disconnect drop-guard (#55971) correctly prevents dispatch into a + torn-down session. Destroying the event is wrong: by the time we reach + enqueue/flush, python-telegram-bot has already acked the update and + advanced the offset — silent permanent loss, no log, no error. + + Hold the event and redispatch it from ``_mark_connected``. Cap the + queue so a long outage cannot grow without bound. Dedup by object + identity so salvage-after-hold never double-queues the same event. + """ + # Dedup: same MessageEvent object already held (flush held, then + # teardown salvage races the same reference — shouldn't happen after + # pop, but identity guard is free and closes the class of bug). + held = getattr(self, "_held_inbound_events", None) + if held is None: + self._held_inbound_events = [] + held = self._held_inbound_events + for existing in held: + if existing is event: + return + + max_n = int(getattr(self, "HELD_INBOUND_MAX", 64) or 64) + while len(held) >= max_n: + dropped = held.pop(0) + logger.warning( + "[Telegram] Held-inbound queue full (%d); dropping oldest (%d chars)", + max_n, + len(getattr(dropped, "text", None) or ""), + ) + held.append(event) + logger.warning( + "[Telegram] Holding inbound during disconnect (%s, %d chars, queue=%d) " + "- will redispatch on reconnect", + where, + len(getattr(event, "text", None) or ""), + len(held), + ) + + async def _redispatch_held_inbound( + self, prior: Optional[asyncio.Task] = None + ) -> None: + """Drain the hold queue after reconnect. + + ``prior`` is the previous redispatch task, if any — awaited here so + ``_mark_connected`` stays synchronous while teardown can still + cancel+await the single tracked task via + ``_cancel_pending_delivery_tasks``. + """ + if prior is not None and prior is not asyncio.current_task() and not prior.done(): + prior.cancel() + try: + await prior + except asyncio.CancelledError: + pass + + held = getattr(self, "_held_inbound_events", None) + if not held: + return + # Take ownership atomically so a concurrent hold during drain appends + # to a fresh list and is picked up by a later connect. + events = list(held) + held.clear() + logger.warning( + "[Telegram] Redispatching %d held inbound message(s) after reconnect", + len(events), + ) + for idx, event in enumerate(events): + if self._should_drop_delayed_delivery(): + # Disconnect returned mid-drain — re-hold current + remainder. + self._hold_inbound_event(event, where="redispatch-interrupted") + for rest in events[idx + 1 :]: + self._hold_inbound_event(rest, where="redispatch-interrupted") + return + try: + await self.handle_message(event) + except asyncio.CancelledError: + # Task cancelled (disconnect / superseding redispatch) — re-hold. + self._hold_inbound_event(event, where="redispatch-cancelled") + for rest in events[idx + 1 :]: + self._hold_inbound_event(rest, where="redispatch-cancelled") + raise + except Exception: + logger.exception( + "[Telegram] Failed to redispatch held inbound (%d chars)", + len(getattr(event, "text", None) or ""), + ) + def _notification_kwargs( self, metadata: Optional[Dict[str, Any]] ) -> Dict[str, Any]: @@ -4565,12 +4691,27 @@ def collect(task: Optional[asyncio.Task]) -> None: collect(task) collect(getattr(self, "_polling_error_task", None)) collect(getattr(self, "_polling_progress_verifier_task", None)) + # Hold-queue redispatch must be cancellable+awaitable on teardown so it + # cannot dispatch handle_message into a torn-down session (same lifecycle + # rule teknium called out on #72037 for shielded flush dispatch). + collect(getattr(self, "_held_inbound_redispatch_task", None)) for task in pending_tasks: task.cancel() if awaitable_tasks: await asyncio.gather(*awaitable_tasks, return_exceptions=True) + # Salvage buffered inbound events before clearing maps. Cancel alone + # leaves the event in the map with no flush task — clearing without + # hold would silently destroy user messages (PTB already advanced the + # offset). Hold for redispatch on reconnect. + for event in list(self._pending_text_batches.values()): + self._hold_inbound_event(event, where="text-batch-teardown") + for event in list(self._pending_photo_batches.values()): + self._hold_inbound_event(event, where="photo-batch-teardown") + for event in list(self._media_group_events.values()): + self._hold_inbound_event(event, where="media-group-teardown") + self._media_group_tasks.clear() self._media_group_events.clear() self._pending_photo_batch_tasks.clear() @@ -4581,6 +4722,8 @@ def collect(task: Optional[asyncio.Task]) -> None: self._polling_error_task = None if getattr(self, "_polling_progress_verifier_task", None) is not current_task: self._polling_progress_verifier_task = None + if getattr(self, "_held_inbound_redispatch_task", None) is not current_task: + self._held_inbound_redispatch_task = None async def _await_disconnect_step(self, awaitable, timeout: float, step: str) -> bool: """Await one disconnect step; detach on timeout so teardown advances. @@ -9317,7 +9460,7 @@ def _enqueue_text_event(self, event: MessageEvent) -> None: dispatching the combined message. """ if self._should_drop_delayed_delivery(): - logger.debug("[Telegram] Dropping text batch enqueue after disconnect started") + self._hold_inbound_event(event, where="text-enqueue") return key = self._text_batch_key(event) @@ -9351,6 +9494,7 @@ async def _flush_text_batch(self, key: str) -> None: split point, since a continuation chunk is almost certain. """ current_task = asyncio.current_task() + event = None try: # Adaptive delay tiers: # - last chunk ≥ _SPLIT_THRESHOLD: a continuation is almost @@ -9380,13 +9524,20 @@ async def _flush_text_batch(self, key: str) -> None: if not event: return if self._should_drop_delayed_delivery(): - logger.debug("[Telegram] Dropping text batch flush after disconnect started") + self._hold_inbound_event(event, where="text-flush") + event = None return logger.info( "[Telegram] Flushing text batch %s (%d chars)", key, len(event.text or ""), ) await self.handle_message(event) + event = None + except asyncio.CancelledError: + # Cancelled after pop but before durable dispatch — hold, don't lose. + if event is not None: + self._hold_inbound_event(event, where="text-flush-cancelled") + raise finally: if self._pending_text_batch_tasks.get(key) is current_task: self._pending_text_batch_tasks.pop(key, None) @@ -9411,16 +9562,23 @@ def _photo_batch_key(self, event: MessageEvent, msg: Message) -> str: async def _flush_photo_batch(self, batch_key: str) -> None: """Send a buffered photo burst/album as a single MessageEvent.""" current_task = asyncio.current_task() + event = None try: await asyncio.sleep(self._media_batch_delay_seconds) event = self._pending_photo_batches.pop(batch_key, None) if not event: return if self._should_drop_delayed_delivery(): - logger.debug("[Telegram] Dropping photo batch flush after disconnect started") + self._hold_inbound_event(event, where="photo-flush") + event = None return logger.info("[Telegram] Flushing photo batch %s with %d image(s)", batch_key, len(event.media_urls)) await self.handle_message(event) + event = None + except asyncio.CancelledError: + if event is not None: + self._hold_inbound_event(event, where="photo-flush-cancelled") + raise finally: if self._pending_photo_batch_tasks.get(batch_key) is current_task: self._pending_photo_batch_tasks.pop(batch_key, None) @@ -9428,7 +9586,7 @@ async def _flush_photo_batch(self, batch_key: str) -> None: def _enqueue_photo_event(self, batch_key: str, event: MessageEvent) -> None: """Merge photo events into a pending batch and schedule flush.""" if self._should_drop_delayed_delivery(): - logger.debug("[Telegram] Dropping photo batch enqueue after disconnect started") + self._hold_inbound_event(event, where="photo-enqueue") return existing = self._pending_photo_batches.get(batch_key) @@ -9755,7 +9913,7 @@ async def _queue_media_group_event(self, media_group_id: str, event: MessageEven attachments into a single MessageEvent. """ if self._should_drop_delayed_delivery(): - logger.debug("[Telegram] Dropping media group enqueue after disconnect started") + self._hold_inbound_event(event, where="media-group-enqueue") return existing = self._media_group_events.get(media_group_id) @@ -9777,15 +9935,22 @@ async def _queue_media_group_event(self, media_group_id: str, event: MessageEven async def _flush_media_group_event(self, media_group_id: str) -> None: current_task = asyncio.current_task() + event = None try: await asyncio.sleep(self.MEDIA_GROUP_WAIT_SECONDS) event = self._media_group_events.pop(media_group_id, None) - if event is not None: - if self._should_drop_delayed_delivery(): - logger.debug("[Telegram] Dropping media group flush after disconnect started") - return - await self.handle_message(event) + if event is None: + return + if self._should_drop_delayed_delivery(): + self._hold_inbound_event(event, where="media-group-flush") + event = None + return + await self.handle_message(event) + event = None except asyncio.CancelledError: + # Cancelled after pop but before durable dispatch — hold, don't lose. + if event is not None: + self._hold_inbound_event(event, where="media-group-flush-cancelled") return finally: if self._media_group_tasks.get(media_group_id) is current_task: diff --git a/tests/gateway/test_telegram_text_batching.py b/tests/gateway/test_telegram_text_batching.py index 64df829d73338..52c7736ed5d25 100644 --- a/tests/gateway/test_telegram_text_batching.py +++ b/tests/gateway/test_telegram_text_batching.py @@ -47,6 +47,10 @@ def _make_adapter(): adapter._pending_messages = {} adapter._message_handler = AsyncMock() adapter.handle_message = AsyncMock() + # Hold-queue state (preserve inbound across reconnect) + adapter._held_inbound_events = [] + adapter._held_inbound_redispatch_task = None + adapter.HELD_INBOUND_MAX = 64 return adapter @@ -159,3 +163,254 @@ async def test_disconnect_cancels_all_pending_delivery_task_maps(self): assert adapter._media_group_events == {} assert adapter._media_group_tasks == {} assert adapter._polling_error_task is None + + +class TestHoldInboundAcrossReconnect: + """Inbound events must not be destroyed when the disconnect drop-guard fires. + + #55971 introduced ``_drop_delayed_deliveries`` so flushes cannot dispatch + into a torn-down session. That is correct. But the implementation + destroyed the event (debug-level return after pop / before enqueue). + PTB has already advanced the polling offset by then, so Telegram never + redelivers — the user's message is gone with no log and no error. + + Related but distinct from #72037 (cancel-after-pop during follow-up + supersession). This covers the disconnect/reconnect path only. + + Timing: no wall-clock races. Flush paths under test use delay=0 and/or + entered/release ``asyncio.Event`` sync (teknium review rule on #72037). + """ + + @staticmethod + def _zero_batch_delays(adapter) -> None: + """Make flush paths deterministic: no sleep, no timing assumptions.""" + adapter._text_batch_delay_seconds = 0 + adapter._text_batch_split_delay_seconds = 0 + adapter._TEXT_BATCH_FAST_DELAY_S = 0 + adapter._TEXT_BATCH_SHORT_DELAY_S = 0 + adapter._TEXT_BATCH_FAST_LEN = 10**9 + adapter._TEXT_BATCH_SHORT_LEN = 10**9 + adapter._SPLIT_THRESHOLD = 10**9 + adapter._media_batch_delay_seconds = 0 + + @pytest.mark.asyncio + async def test_late_enqueue_held_and_redispatched_on_reconnect(self): + adapter = _make_adapter() + adapter._mark_disconnected() + + adapter._enqueue_text_event(_make_event("should survive disconnect")) + + # Must NOT dispatch into torn-down session + adapter.handle_message.assert_not_called() + assert len(adapter._held_inbound_events) == 1 + assert adapter._held_inbound_events[0].text == "should survive disconnect" + + adapter._mark_connected() + task = adapter._held_inbound_redispatch_task + assert task is not None + await task + + adapter.handle_message.assert_called_once() + assert adapter.handle_message.call_args[0][0].text == "should survive disconnect" + assert adapter._held_inbound_events == [] + + @pytest.mark.asyncio + async def test_flush_during_disconnect_holds_popped_event(self): + """After pop, drop-guard must hold — not destroy — the event. + + Deterministic: delay=0 and drop already True before flush runs, so the + post-pop branch is exercised without wall-clock races. + """ + adapter = _make_adapter() + self._zero_batch_delays(adapter) + event = _make_event("popped then held") + adapter._pending_text_batches["k"] = event + adapter._drop_delayed_deliveries = True + + await adapter._flush_text_batch("k") + + adapter.handle_message.assert_not_called() + assert adapter._pending_text_batches == {} + assert [e.text for e in adapter._held_inbound_events] == ["popped then held"] + + @pytest.mark.asyncio + async def test_flush_cancel_after_pop_holds_event(self): + """Cancel after pop (before handle_message returns) must hold, not lose. + + Uses entered/release Events — no sleep timing (teknium #72037 rule). + """ + adapter = _make_adapter() + self._zero_batch_delays(adapter) + entered = asyncio.Event() + release = asyncio.Event() + + async def _blocking_handle(event): + entered.set() + await release.wait() + + adapter.handle_message = _blocking_handle + adapter._pending_text_batches["k"] = _make_event("in-flight cancel") + task = asyncio.create_task(adapter._flush_text_batch("k")) + adapter._pending_text_batch_tasks["k"] = task + + await entered.wait() # past pop, inside handle_message + task.cancel() + with pytest.raises(asyncio.CancelledError): + await task + release.set() # unblock if anything still waiting + + # handle_message was entered but cancel + CancelledError hold path + # must leave the event recoverable. After cancel during handle, our + # CancelledError handler only holds if event is not None — we set + # event=None only after successful handle. Cancel during handle means + # event is still set → held. + held_texts = [e.text for e in adapter._held_inbound_events] + assert "in-flight cancel" in held_texts + + @pytest.mark.asyncio + async def test_cancel_pending_salvages_batches_into_held_queue(self): + """Teardown must salvage map contents before clear — not discard them.""" + adapter = _make_adapter() + adapter._pending_text_batches["text"] = _make_event("text-salvage") + adapter._pending_photo_batches["photo"] = _make_event("photo-salvage") + adapter._media_group_events["media"] = _make_event("media-salvage") + t1 = asyncio.create_task(asyncio.sleep(60)) + t2 = asyncio.create_task(asyncio.sleep(60)) + t3 = asyncio.create_task(asyncio.sleep(60)) + adapter._pending_text_batch_tasks["text"] = t1 + adapter._pending_photo_batch_tasks["photo"] = t2 + adapter._media_group_tasks["media"] = t3 + + adapter._mark_disconnected() + await adapter._cancel_pending_delivery_tasks() + + held = {e.text for e in adapter._held_inbound_events} + assert held == {"text-salvage", "photo-salvage", "media-salvage"} + assert adapter._pending_text_batches == {} + assert adapter._pending_photo_batches == {} + assert adapter._media_group_events == {} + assert adapter._held_inbound_redispatch_task is None + + @pytest.mark.asyncio + async def test_redispatch_task_cancelled_on_teardown(self): + """In-flight redispatch must be in the cancel map (lifecycle rule).""" + adapter = _make_adapter() + entered = asyncio.Event() + release = asyncio.Event() + + async def _blocking_handle(event): + entered.set() + await release.wait() + + adapter.handle_message = _blocking_handle + adapter._held_inbound_events = [_make_event("during-redispatch")] + adapter._drop_delayed_deliveries = False + task = asyncio.create_task(adapter._redispatch_held_inbound()) + adapter._held_inbound_redispatch_task = task + + await entered.wait() + adapter._mark_disconnected() + await adapter._cancel_pending_delivery_tasks() + + assert task.done() + # Cancel during handle → re-held + assert any(e.text == "during-redispatch" for e in adapter._held_inbound_events) + release.set() + + @pytest.mark.asyncio + async def test_photo_and_media_group_enqueue_held_during_disconnect(self): + adapter = _make_adapter() + adapter._mark_disconnected() + + photo = _make_event("photo caption") + photo.media_urls = ["u1"] + photo.media_types = ["image"] + adapter._enqueue_photo_event("k", photo) + + album = _make_event("album caption") + album.media_urls = ["u2"] + album.media_types = ["image"] + await adapter._queue_media_group_event("mg1", album) + + adapter.handle_message.assert_not_called() + texts = {e.text for e in adapter._held_inbound_events} + assert texts == {"photo caption", "album caption"} + + @pytest.mark.asyncio + async def test_hold_dedupes_same_event_object(self): + adapter = _make_adapter() + event = _make_event("once") + adapter._hold_inbound_event(event, where="a") + adapter._hold_inbound_event(event, where="b") + assert len(adapter._held_inbound_events) == 1 + + @pytest.mark.asyncio + async def test_held_queue_cap_drops_oldest(self): + adapter = _make_adapter() + adapter.HELD_INBOUND_MAX = 2 + adapter._mark_disconnected() + adapter._enqueue_text_event(_make_event("first")) + adapter._enqueue_text_event(_make_event("second")) + adapter._enqueue_text_event(_make_event("third")) + + texts = [e.text for e in adapter._held_inbound_events] + assert texts == ["second", "third"] + + @pytest.mark.asyncio + async def test_redispatch_aborts_cleanly_if_disconnect_returns(self): + """If disconnect re-trips mid-drain, remaining events stay held.""" + adapter = _make_adapter() + adapter._held_inbound_events = [ + _make_event("a"), + _make_event("b"), + _make_event("c"), + ] + + call_count = 0 + + async def _handle(event): + nonlocal call_count + call_count += 1 + if call_count == 1: + adapter._drop_delayed_deliveries = True + + adapter.handle_message = _handle + adapter._drop_delayed_deliveries = False + await adapter._redispatch_held_inbound() + + assert call_count == 1 + held_texts = [e.text for e in adapter._held_inbound_events] + assert held_texts == ["b", "c"] + + @pytest.mark.asyncio + async def test_non_retryable_fatal_discards_held_with_warning(self): + adapter = _make_adapter() + adapter._held_inbound_events = [_make_event("doomed")] + # BasePlatformAdapter._set_fatal_error may need attrs — call ours via + # the override path with a stub super if needed. + from gateway.platforms.base import BasePlatformAdapter + + with patch.object(BasePlatformAdapter, "_set_fatal_error", lambda *a, **k: None): + adapter._set_fatal_error("auth", "revoked", retryable=False) + + assert adapter._held_inbound_events == [] + assert adapter._drop_delayed_deliveries is True + + @pytest.mark.asyncio + async def test_production_text_handler_terminal_step_holds_when_disconnected(self): + """Production path: ``_handle_text_message`` ends in ``_enqueue_text_event``. + + Sweeper rejects helper-only coverage. This pins the call site that + PTB invokes after the update is already acked (offset advanced). + """ + adapter = _make_adapter() + adapter._mark_disconnected() + # Terminal step of _handle_text_message after event construction. + adapter._enqueue_text_event(_make_event("acked-by-ptb-then-held")) + adapter.handle_message.assert_not_called() + assert [e.text for e in adapter._held_inbound_events] == ["acked-by-ptb-then-held"] + + adapter._mark_connected() + await adapter._held_inbound_redispatch_task + adapter.handle_message.assert_called_once() + assert adapter.handle_message.call_args[0][0].text == "acked-by-ptb-then-held" From 42dc17ec975a178e7ca47392da968538fed7062d Mon Sep 17 00:00:00 2001 From: "Daniel V. Baecker" <295684751+dvbaecker@users.noreply.github.com> Date: Tue, 11 Aug 2026 14:51:19 +0100 Subject: [PATCH 005/376] fix(telegram): close hold-lifecycle gaps on permanent fatal and connected drain MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Address review on #83878: - Permanent fatal fences all hold producers and discards pending maps on teardown instead of re-populating a queue that can never drain. - Any hold created while connected schedules a tracked redispatch (cancel- after-pop no longer orphans until a future reconnect). - Redispatch failures re-hold current + remainder without tight-looping. Regression coverage for the three residual paths, plus the interaction with OOF-156's connect-failure classification: the retryable network path (telegram_connect_error) must NOT clear the hold queue — reconnect is precisely what drains it; only non-retryable fatals discard. --- plugins/platforms/telegram/adapter.py | 240 ++++++++++++++----- tests/gateway/test_telegram_text_batching.py | 136 ++++++++++- 2 files changed, 301 insertions(+), 75 deletions(-) diff --git a/plugins/platforms/telegram/adapter.py b/plugins/platforms/telegram/adapter.py index af3578aa7132e..7f42e38093922 100644 --- a/plugins/platforms/telegram/adapter.py +++ b/plugins/platforms/telegram/adapter.py @@ -922,18 +922,7 @@ def _mark_connected(self) -> None: super()._mark_connected() # Drain anything held while we were down. PTB will not redeliver — # these events exist only in our hold queue now. - if not getattr(self, "_held_inbound_events", None): - return - try: - loop = asyncio.get_running_loop() - except RuntimeError: - return - prior = getattr(self, "_held_inbound_redispatch_task", None) - # Single tracked task so disconnect can cancel+await it (teknium - # lifecycle rule from #72037 review: no untracked dispatch). - self._held_inbound_redispatch_task = loop.create_task( - self._redispatch_held_inbound(prior=prior) - ) + self._schedule_held_inbound_redispatch() def _mark_disconnected(self) -> None: self._drop_delayed_deliveries = True @@ -942,16 +931,26 @@ def _mark_disconnected(self) -> None: def _set_fatal_error(self, code: str, message: str, *, retryable: bool) -> None: self._drop_delayed_deliveries = True super()._set_fatal_error(code, message, retryable=retryable) - # Permanent fatal: no reconnect will drain the queue. Surface the loss - # explicitly instead of letting held messages die silently with the process. - if not retryable and getattr(self, "_held_inbound_events", None): - n = len(self._held_inbound_events) - logger.warning( - "[Telegram] Non-retryable fatal (%s); discarding %d held inbound message(s)", - code, - n, - ) - self._held_inbound_events.clear() + # Permanent fatal: no reconnect will drain. Discard the hold queue now + # and refuse further holds (teardown salvage / late enqueue must not + # re-populate a queue that can never drain — review #83878). + if not retryable: + held = getattr(self, "_held_inbound_events", None) + n = len(held) if held else 0 + if held: + held.clear() + if n: + logger.warning( + "[Telegram] Non-retryable fatal (%s); discarding %d held inbound message(s)", + code, + n, + ) + + def _is_permanent_fatal(self) -> bool: + """True after non-retryable fatal — holds must discard, not queue.""" + if not getattr(self, "_fatal_error_code", None): + return False + return not bool(getattr(self, "_fatal_error_retryable", True)) def _should_drop_delayed_delivery(self) -> bool: """True once teardown/fatal-error started — delayed flushes must not dispatch. @@ -962,11 +961,54 @@ def _should_drop_delayed_delivery(self) -> bool: Callers must NOT destroy the event when this returns True: PTB has already advanced the polling offset, so Telegram will never redeliver. - Use ``_hold_inbound_event`` and redispatch on reconnect. + Use ``_hold_inbound_event`` and redispatch on reconnect (unless + permanent fatal, which discards explicitly). """ return bool(getattr(self, "_drop_delayed_deliveries", False)) - def _hold_inbound_event(self, event: "MessageEvent", *, where: str) -> None: + def _schedule_held_inbound_redispatch(self) -> None: + """Ensure a tracked drain runs when held events exist and delivery is live. + + Drain triggers: + - ``_mark_connected`` after reconnect + - any hold created while already connected (e.g. cancel-after-pop) + - end of a drain pass if more events arrived mid-drain + + No-ops while disconnected/tearing down or after permanent fatal. + """ + if self._is_permanent_fatal(): + return + if self._should_drop_delayed_delivery(): + return + held = getattr(self, "_held_inbound_events", None) + if not held: + return + try: + loop = asyncio.get_running_loop() + except RuntimeError: + return + prior = getattr(self, "_held_inbound_redispatch_task", None) + try: + current = asyncio.current_task() + except RuntimeError: + current = None + # Already draining on another task — that pass schedules a follow-up + # if anything remains. Do not stack duplicate tasks. + if prior is not None and not prior.done() and prior is not current: + return + self._held_inbound_redispatch_task = loop.create_task( + self._redispatch_held_inbound( + prior=None if prior is current else prior + ) + ) + + def _hold_inbound_event( + self, + event: "MessageEvent", + *, + where: str, + schedule: bool = True, + ) -> None: """Preserve an inbound event that cannot be dispatched right now. The disconnect drop-guard (#55971) correctly prevents dispatch into a @@ -974,13 +1016,22 @@ def _hold_inbound_event(self, event: "MessageEvent", *, where: str) -> None: enqueue/flush, python-telegram-bot has already acked the update and advanced the offset — silent permanent loss, no log, no error. - Hold the event and redispatch it from ``_mark_connected``. Cap the - queue so a long outage cannot grow without bound. Dedup by object - identity so salvage-after-hold never double-queues the same event. + Hold the event and redispatch from ``_mark_connected`` (or immediately + if already connected). Cap the queue so a long outage cannot grow + without bound. Dedup by object identity so salvage-after-hold never + double-queues the same event. Permanent fatal discards explicitly. + + ``schedule=False`` when the caller is already inside a drain and will + decide follow-up policy (avoids poison-event tight loops). """ - # Dedup: same MessageEvent object already held (flush held, then - # teardown salvage races the same reference — shouldn't happen after - # pop, but identity guard is free and closes the class of bug). + if self._is_permanent_fatal(): + logger.warning( + "[Telegram] Discarding inbound under non-retryable fatal (%s, %d chars)", + where, + len(getattr(event, "text", None) or ""), + ) + return + held = getattr(self, "_held_inbound_events", None) if held is None: self._held_inbound_events = [] @@ -999,17 +1050,23 @@ def _hold_inbound_event(self, event: "MessageEvent", *, where: str) -> None: ) held.append(event) logger.warning( - "[Telegram] Holding inbound during disconnect (%s, %d chars, queue=%d) " - "- will redispatch on reconnect", + "[Telegram] Holding inbound (%s, %d chars, queue=%d)%s", where, len(getattr(event, "text", None) or ""), len(held), + " - will redispatch on reconnect" + if self._should_drop_delayed_delivery() + else (" - scheduling redispatch" if schedule else ""), ) + # Connected cancel-after-pop (and any other live-path hold) must not + # orphan the event waiting for a future reconnect that may never come. + if schedule and not self._should_drop_delayed_delivery(): + self._schedule_held_inbound_redispatch() async def _redispatch_held_inbound( self, prior: Optional[asyncio.Task] = None ) -> None: - """Drain the hold queue after reconnect. + """Drain the hold queue after reconnect or a connected-path hold. ``prior`` is the previous redispatch task, if any — awaited here so ``_mark_connected`` stays synchronous while teardown can still @@ -1023,37 +1080,79 @@ async def _redispatch_held_inbound( except asyncio.CancelledError: pass + if self._is_permanent_fatal(): + held = getattr(self, "_held_inbound_events", None) + if held: + n = len(held) + held.clear() + logger.warning( + "[Telegram] Redispatch aborted; discarded %d held inbound under non-retryable fatal", + n, + ) + return + held = getattr(self, "_held_inbound_events", None) if not held: return # Take ownership atomically so a concurrent hold during drain appends - # to a fresh list and is picked up by a later connect. + # to a fresh list and is picked up by a follow-up schedule. events = list(held) held.clear() logger.warning( - "[Telegram] Redispatching %d held inbound message(s) after reconnect", + "[Telegram] Redispatching %d held inbound message(s)", len(events), ) - for idx, event in enumerate(events): - if self._should_drop_delayed_delivery(): - # Disconnect returned mid-drain — re-hold current + remainder. - self._hold_inbound_event(event, where="redispatch-interrupted") - for rest in events[idx + 1 :]: - self._hold_inbound_event(rest, where="redispatch-interrupted") - return - try: - await self.handle_message(event) - except asyncio.CancelledError: - # Task cancelled (disconnect / superseding redispatch) — re-hold. - self._hold_inbound_event(event, where="redispatch-cancelled") - for rest in events[idx + 1 :]: - self._hold_inbound_event(rest, where="redispatch-cancelled") - raise - except Exception: - logger.exception( - "[Telegram] Failed to redispatch held inbound (%d chars)", - len(getattr(event, "text", None) or ""), - ) + allow_followup_schedule = True + try: + for idx, event in enumerate(events): + if self._is_permanent_fatal() or self._should_drop_delayed_delivery(): + # Disconnect/fatal mid-drain — re-hold current + remainder + # (hold itself discards under permanent fatal). + self._hold_inbound_event( + event, where="redispatch-interrupted", schedule=False + ) + for rest in events[idx + 1 :]: + self._hold_inbound_event( + rest, where="redispatch-interrupted", schedule=False + ) + return + try: + await self.handle_message(event) + except asyncio.CancelledError: + self._hold_inbound_event( + event, where="redispatch-cancelled", schedule=False + ) + for rest in events[idx + 1 :]: + self._hold_inbound_event( + rest, where="redispatch-cancelled", schedule=False + ) + raise + except Exception: + # Retryable failure: keep current + remainder. Do not + # immediately reschedule — a poison event would tight-loop. + # Next mark_connected or a later connected-path hold drains. + logger.exception( + "[Telegram] Failed to redispatch held inbound (%d chars); re-holding", + len(getattr(event, "text", None) or ""), + ) + self._hold_inbound_event( + event, where="redispatch-failed", schedule=False + ) + for rest in events[idx + 1 :]: + self._hold_inbound_event( + rest, where="redispatch-failed", schedule=False + ) + allow_followup_schedule = False + return + finally: + # Events arrived mid-drain while still connected need another pass. + if ( + allow_followup_schedule + and getattr(self, "_held_inbound_events", None) + and not self._should_drop_delayed_delivery() + and not self._is_permanent_fatal() + ): + self._schedule_held_inbound_redispatch() def _notification_kwargs( self, metadata: Optional[Dict[str, Any]] @@ -4701,16 +4800,27 @@ def collect(task: Optional[asyncio.Task]) -> None: if awaitable_tasks: await asyncio.gather(*awaitable_tasks, return_exceptions=True) - # Salvage buffered inbound events before clearing maps. Cancel alone - # leaves the event in the map with no flush task — clearing without - # hold would silently destroy user messages (PTB already advanced the - # offset). Hold for redispatch on reconnect. - for event in list(self._pending_text_batches.values()): - self._hold_inbound_event(event, where="text-batch-teardown") - for event in list(self._pending_photo_batches.values()): - self._hold_inbound_event(event, where="photo-batch-teardown") - for event in list(self._media_group_events.values()): - self._hold_inbound_event(event, where="media-group-teardown") + # Salvage buffered inbound events before clearing maps — unless permanent + # fatal, where no reconnect can drain and hold would re-orphan them + # (#83878). Discard pending sources explicitly in that case. + if self._is_permanent_fatal(): + n_pending = ( + len(self._pending_text_batches) + + len(self._pending_photo_batches) + + len(self._media_group_events) + ) + if n_pending: + logger.warning( + "[Telegram] Non-retryable fatal teardown; discarding %d pending inbound batch(es)", + n_pending, + ) + else: + for event in list(self._pending_text_batches.values()): + self._hold_inbound_event(event, where="text-batch-teardown") + for event in list(self._pending_photo_batches.values()): + self._hold_inbound_event(event, where="photo-batch-teardown") + for event in list(self._media_group_events.values()): + self._hold_inbound_event(event, where="media-group-teardown") self._media_group_tasks.clear() self._media_group_events.clear() diff --git a/tests/gateway/test_telegram_text_batching.py b/tests/gateway/test_telegram_text_batching.py index 52c7736ed5d25..e3e07090a678a 100644 --- a/tests/gateway/test_telegram_text_batching.py +++ b/tests/gateway/test_telegram_text_batching.py @@ -238,13 +238,16 @@ async def test_flush_cancel_after_pop_holds_event(self): """Cancel after pop (before handle_message returns) must hold, not lose. Uses entered/release Events — no sleep timing (teknium #72037 rule). + Connected path then schedules redispatch (#83878). """ adapter = _make_adapter() self._zero_batch_delays(adapter) entered = asyncio.Event() release = asyncio.Event() + seen: list[str] = [] async def _blocking_handle(event): + seen.append(event.text or "") entered.set() await release.wait() @@ -257,15 +260,16 @@ async def _blocking_handle(event): task.cancel() with pytest.raises(asyncio.CancelledError): await task - release.set() # unblock if anything still waiting + release.set() + + drain = adapter._held_inbound_redispatch_task + assert drain is not None + await asyncio.wait_for(drain, timeout=1.0) - # handle_message was entered but cancel + CancelledError hold path - # must leave the event recoverable. After cancel during handle, our - # CancelledError handler only holds if event is not None — we set - # event=None only after successful handle. Cancel during handle means - # event is still set → held. + # Recoverable: held and/or delivered via redispatch (seen may include + # the original in-flight attempt plus the redispatch). held_texts = [e.text for e in adapter._held_inbound_events] - assert "in-flight cancel" in held_texts + assert "in-flight cancel" in seen or "in-flight cancel" in held_texts @pytest.mark.asyncio async def test_cancel_pending_salvages_batches_into_held_queue(self): @@ -386,15 +390,58 @@ async def _handle(event): async def test_non_retryable_fatal_discards_held_with_warning(self): adapter = _make_adapter() adapter._held_inbound_events = [_make_event("doomed")] - # BasePlatformAdapter._set_fatal_error may need attrs — call ours via - # the override path with a stub super if needed. from gateway.platforms.base import BasePlatformAdapter - with patch.object(BasePlatformAdapter, "_set_fatal_error", lambda *a, **k: None): + def _base_fatal(self, code, message, *, retryable): + self._fatal_error_code = code + self._fatal_error_message = message + self._fatal_error_retryable = retryable + self._running = False + + with patch.object(BasePlatformAdapter, "_set_fatal_error", _base_fatal): adapter._set_fatal_error("auth", "revoked", retryable=False) assert adapter._held_inbound_events == [] assert adapter._drop_delayed_deliveries is True + assert adapter._is_permanent_fatal() is True + + @pytest.mark.asyncio + async def test_retryable_fatal_preserves_held_for_reconnect_drain(self): + """Retryable fatals must NOT clear the hold queue. + + OOF-156's connect-failure classification keeps the common network + path ``retryable=True`` (``telegram_connect_error``) — reconnect is + precisely what must drain a hold queue populated during the outage. + Only non-retryable fatals may discard (covered above). + """ + adapter = _make_adapter() + adapter._held_inbound_events = [_make_event("survives-network-fatal")] + adapter._drop_delayed_deliveries = True # fatal/disconnect already set + + from gateway.platforms.base import BasePlatformAdapter + + def _base_fatal(self, code, message, *, retryable): + self._fatal_error_code = code + self._fatal_error_message = message + self._fatal_error_retryable = retryable + + with patch.object(BasePlatformAdapter, "_set_fatal_error", _base_fatal): + adapter._set_fatal_error( + "telegram_connect_error", "connect timed out", retryable=True + ) + + assert [e.text for e in adapter._held_inbound_events] == [ + "survives-network-fatal" + ] + assert adapter._is_permanent_fatal() is False + + # Reconnect drains what the retryable fatal preserved. + adapter._mark_connected() + await adapter._held_inbound_redispatch_task + adapter.handle_message.assert_called_once() + assert ( + adapter.handle_message.call_args[0][0].text == "survives-network-fatal" + ) @pytest.mark.asyncio async def test_production_text_handler_terminal_step_holds_when_disconnected(self): @@ -414,3 +461,72 @@ async def test_production_text_handler_terminal_step_holds_when_disconnected(sel await adapter._held_inbound_redispatch_task adapter.handle_message.assert_called_once() assert adapter.handle_message.call_args[0][0].text == "acked-by-ptb-then-held" + + @pytest.mark.asyncio + async def test_permanent_fatal_teardown_discards_pending_not_rehold(self): + """#83878: permanent fatal must not re-populate hold via teardown salvage.""" + adapter = _make_adapter() + adapter._fatal_error_code = "auth" + adapter._fatal_error_retryable = False + adapter._drop_delayed_deliveries = True + adapter._pending_text_batches["t"] = _make_event("pending-text") + adapter._pending_photo_batches["p"] = _make_event("pending-photo") + adapter._media_group_events["m"] = _make_event("pending-media") + + await adapter._cancel_pending_delivery_tasks() + + assert adapter._held_inbound_events == [] + assert adapter._pending_text_batches == {} + assert adapter._pending_photo_batches == {} + assert adapter._media_group_events == {} + + @pytest.mark.asyncio + async def test_permanent_fatal_late_enqueue_discards(self): + """#83878: late enqueue after permanent fatal must discard, not hold.""" + adapter = _make_adapter() + adapter._fatal_error_code = "auth" + adapter._fatal_error_retryable = False + adapter._drop_delayed_deliveries = True + + adapter._enqueue_text_event(_make_event("too-late")) + adapter.handle_message.assert_not_called() + assert adapter._held_inbound_events == [] + + @pytest.mark.asyncio + async def test_connected_hold_schedules_redispatch(self): + """#83878: hold while connected must drain, not orphan until reconnect.""" + adapter = _make_adapter() + adapter._drop_delayed_deliveries = False + adapter.handle_message = AsyncMock() + + adapter._hold_inbound_event( + _make_event("orphan-without-drain"), where="text-flush-cancelled" + ) + + drain = adapter._held_inbound_redispatch_task + assert drain is not None + await asyncio.wait_for(drain, timeout=1.0) + adapter.handle_message.assert_called_once() + assert adapter.handle_message.call_args[0][0].text == "orphan-without-drain" + assert adapter._held_inbound_events == [] + + @pytest.mark.asyncio + async def test_redispatch_exception_reholds_current_and_remainder(self): + """#83878: handle_message failure must not drop current/remainder.""" + adapter = _make_adapter() + adapter._drop_delayed_deliveries = False + adapter._held_inbound_events = [ + _make_event("boom"), + _make_event("after"), + ] + + async def _handle(event): + if event.text == "boom": + raise RuntimeError("dispatch failed") + return None + + adapter.handle_message = _handle + # Direct drain (no auto follow-up on failure) + await adapter._redispatch_held_inbound() + held_texts = [e.text for e in adapter._held_inbound_events] + assert held_texts == ["boom", "after"] From cb47f59ffa1056732c0a5194d2a1847dc64c2c37 Mon Sep 17 00:00:00 2001 From: kshitij <82637225+kshitijk4poor@users.noreply.github.com> Date: Sat, 15 Aug 2026 03:30:53 +0530 Subject: [PATCH 006/376] fix: standardize media-group CancelledError to raise Pre-existing inconsistency: _flush_media_group_event used return in CancelledError while _flush_text_batch and _flush_photo_batch used raise. Changed to raise for consistency and to properly propagate task cancellation. Made more visible by the hold-queue changes in #83878. --- plugins/platforms/telegram/adapter.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/plugins/platforms/telegram/adapter.py b/plugins/platforms/telegram/adapter.py index 7f42e38093922..bf18ff61d39ad 100644 --- a/plugins/platforms/telegram/adapter.py +++ b/plugins/platforms/telegram/adapter.py @@ -10061,7 +10061,7 @@ async def _flush_media_group_event(self, media_group_id: str) -> None: # Cancelled after pop but before durable dispatch — hold, don't lose. if event is not None: self._hold_inbound_event(event, where="media-group-flush-cancelled") - return + raise finally: if self._media_group_tasks.get(media_group_id) is current_task: self._media_group_tasks.pop(media_group_id, None) From 23954e3e31515a56c947628e674e5c87a7c1dd69 Mon Sep 17 00:00:00 2001 From: Jakub Wolniewicz <4850809+frizikk@users.noreply.github.com> Date: Tue, 11 Aug 2026 19:15:32 +0200 Subject: [PATCH 007/376] fix(desktop): prevent navigation from stealing focus --- apps/desktop/electron/session-windows.test.ts | 16 ++++++++++++++++ apps/desktop/electron/session-windows.ts | 9 ++++++++- 2 files changed, 24 insertions(+), 1 deletion(-) diff --git a/apps/desktop/electron/session-windows.test.ts b/apps/desktop/electron/session-windows.test.ts index a3dbfc04e4a00..1959cc44583a6 100644 --- a/apps/desktop/electron/session-windows.test.ts +++ b/apps/desktop/electron/session-windows.test.ts @@ -203,6 +203,22 @@ test('chatWindowWebPreferences leaves background throttling to the runtime strea assert.equal('backgroundThrottling' in prefs, false) }) +test('chat renderer navigation stays passive while explicit window actions may focus', () => { + const prefs = chatWindowWebPreferences('/tmp/preload.cjs') + + // In-page/SPA navigation can happen while a transcript keeps streaming. It + // must not use Electron's default navigation focus path to activate Hermes. + assert.equal(prefs.focusOnNavigation, false) + + // Re-opening a session is an explicit user action and must still raise the + // existing window; the passive navigation guard does not disable that path. + const registry = createSessionWindowRegistry() + const win = makeFakeWindow() + registry.openOrFocus('s1', () => win) + registry.openOrFocus('s1', () => win) + assert.equal(win.calls.focus, 1) +}) + test('chatWindowWebPreferences passes the preload path through and keeps the hardened defaults', () => { const prefs = chatWindowWebPreferences('/some/preload.cjs') diff --git a/apps/desktop/electron/session-windows.ts b/apps/desktop/electron/session-windows.ts index 81736908fcea0..790f1c03a73d1 100644 --- a/apps/desktop/electron/session-windows.ts +++ b/apps/desktop/electron/session-windows.ts @@ -37,6 +37,12 @@ const SESSION_WINDOW_MIN_HEIGHT = 620 // session is silent" bug. Manual voice-start worked only because the button // click counted as the gesture. This is a native app the user deliberately // launched; there is no drive-by-autoplay concern to protect against. +// +// `focusOnNavigation: false` keeps renderer-driven work passive. Electron's +// default is true, so an in-page/SPA navigation can activate a blurred chat +// window while its transcript is streaming. Explicit user actions still call +// the main-process window focus paths (session re-open, notification/deep-link, +// app activation), preserving intentional raises without background focus theft. function chatWindowWebPreferences(preloadPath: string) { return { preload: preloadPath, @@ -45,7 +51,8 @@ function chatWindowWebPreferences(preloadPath: string) { sandbox: true, nodeIntegration: false, devTools: true, - autoplayPolicy: 'no-user-gesture-required' as const + autoplayPolicy: 'no-user-gesture-required' as const, + focusOnNavigation: false } } From 0611f9f8562be82f93c8a06d9c54e50953be5e71 Mon Sep 17 00:00:00 2001 From: LemonSchneid <122685409+LemonSchneid@users.noreply.github.com> Date: Thu, 13 Aug 2026 18:59:16 -0600 Subject: [PATCH 008/376] fix(desktop): retain HUD composer focus Keep the native HUD window mouse-solid while focus is inside the rich composer. On Windows this prevents click-through from reactivating the app beneath the HUD, stealing the caret, and collapsing the transcript. --- .../desktop/src/app/hud/click-through.test.ts | 18 ++++++++++++++++-- apps/desktop/src/app/hud/click-through.ts | 19 ++++++++++++------- 2 files changed, 28 insertions(+), 9 deletions(-) diff --git a/apps/desktop/src/app/hud/click-through.test.ts b/apps/desktop/src/app/hud/click-through.test.ts index 45e789492de90..2e12553da2e3a 100644 --- a/apps/desktop/src/app/hud/click-through.test.ts +++ b/apps/desktop/src/app/hud/click-through.test.ts @@ -14,6 +14,7 @@ function hud() { shell.setAttribute('data-hud-shell', '') const bar = document.createElement('input') + bar.setAttribute('data-slot', 'composer-rich-input') const overlay = document.createElement('div') overlay.setAttribute('role', 'dialog') @@ -45,10 +46,23 @@ describe('hudIgnoresMouse', () => { expect(hudIgnoresMouse(shell, overlay, null, true)).toBe(false) }) - it('does not pin the window just because the composer holds the caret', () => { + it('keeps the native HUD window solid while the composer holds the caret', () => { const { bar, mount, shell } = hud() - expect(hudIgnoresMouse(shell, mount, bar, true)).toBe(true) + // On Windows, making the native window click-through while its editor owns + // focus can immediately hand the mouse activation back to the app below. + // The caret then flashes, focus leaves, and the transcript collapses before + // the user can read it. + expect(hudIgnoresMouse(shell, mount, bar, true)).toBe(false) + }) + + it('keeps the native HUD window solid while focus is inside the composer editor', () => { + const { bar, mount, shell } = hud() + const editable = document.createElement('span') + editable.contentEditable = 'true' + bar.append(editable) + + expect(hudIgnoresMouse(shell, mount, editable, true)).toBe(false) }) it('pins the window while a portalled overlay holds focus, so an outside click can dismiss it', () => { diff --git a/apps/desktop/src/app/hud/click-through.ts b/apps/desktop/src/app/hud/click-through.ts index 9ab21e5c3eeaa..669f093fd79e5 100644 --- a/apps/desktop/src/app/hud/click-through.ts +++ b/apps/desktop/src/app/hud/click-through.ts @@ -15,10 +15,10 @@ import { type RefObject, useEffect } from 'react' * - Focus BESIDE the shell is a portalled dialog, popover or menu, and it owns * the next click — including the one outside itself that dismisses it, which * the hit test cannot see coming. That pins the window solid. - * - Focus INSIDE the shell does not. The composer holding the caret is the - * HUD's resting state rather than a claim on the whole rectangle, and reading - * it as one is what made an engaged HUD eat every click in its own empty - * space — on a fresh thread, the entire window. + * - Focus INSIDE the shell is normally the HUD's resting state — except for + * the composer caret. On Windows the native window must stay solid while the + * editor owns focus: making it click-through can immediately reactivate the + * application underneath, steal the caret, and collapse the transcript. */ export function hudIgnoresMouse( root: Element, @@ -34,11 +34,16 @@ export function hudIgnoresMouse( } const overSomething = hit !== null && !hit.contains(root) - // `windowFocused` is what stops a stale `active` — the composer keeps focus - // after you click away to another app — pinning the HUD solid forever. + + const composerFocused = + windowFocused && + active !== null && + root.contains(active) && + active.closest('[data-slot="composer-rich-input"]') !== null + const overlayFocused = windowFocused && active !== null && !root.contains(active) && !active.contains(root) - return !overSomething && !overlayFocused + return !composerFocused && !overSomething && !overlayFocused } /** From 957a7c20cdb75d9d608787864c8f7bb7ab794c56 Mon Sep 17 00:00:00 2001 From: nanami7777777 <178757933+nanami7777777@users.noreply.github.com> Date: Thu, 13 Aug 2026 10:47:12 +0800 Subject: [PATCH 009/376] fix(desktop): keep text navigation keys in composer --- apps/desktop/src/app/hooks/use-keybinds.ts | 4 +-- apps/desktop/src/components/find-bar.test.tsx | 9 +++--- apps/desktop/src/lib/keybinds/actions.ts | 2 +- apps/desktop/src/lib/keybinds/combo.test.ts | 28 +++++++++++++------ apps/desktop/src/lib/keybinds/combo.ts | 27 +++++++++++++++--- 5 files changed, 50 insertions(+), 20 deletions(-) diff --git a/apps/desktop/src/app/hooks/use-keybinds.ts b/apps/desktop/src/app/hooks/use-keybinds.ts index bc9a0ead25c17..4f36981d3f4f4 100644 --- a/apps/desktop/src/app/hooks/use-keybinds.ts +++ b/apps/desktop/src/app/hooks/use-keybinds.ts @@ -15,7 +15,7 @@ import { import { onReleaseTypingFocus } from '@/components/ui/keyboard-first' import { findBarClaimsCombo } from '@/lib/find-in-page' import { contributedKeybindHandler, PROFILE_SLOT_COUNT, SESSION_SLOT_COUNT } from '@/lib/keybinds/actions' -import { comboAllowedInInput, comboFromEvent, isEditableTarget } from '@/lib/keybinds/combo' +import { actionAllowedInInput, comboFromEvent, isEditableTarget } from '@/lib/keybinds/combo' import { composerFocusKeysAllowed, isComposerFocusSoftCombo, typeToFocusChar } from '@/lib/keybinds/composer-focus-keys' import { openWorktreeDialog } from '@/store/coding-status' import { toggleCommandPalette } from '@/store/command-palette' @@ -358,7 +358,7 @@ export function useKeybinds(deps: KeybindRuntimeDeps): void { return } - if (isEditableTarget(event.target) && !comboAllowedInInput(combo)) { + if (isEditableTarget(event.target) && !actionAllowedInInput(actionId, combo)) { return } diff --git a/apps/desktop/src/components/find-bar.test.tsx b/apps/desktop/src/components/find-bar.test.tsx index 1b0de2494482b..625aa38d06fb9 100644 --- a/apps/desktop/src/components/find-bar.test.tsx +++ b/apps/desktop/src/components/find-bar.test.tsx @@ -8,7 +8,7 @@ import { en } from '@/i18n/en' import { zh } from '@/i18n/zh' import { findBarClaimsCombo, findBarKeyAction, formatMatchLabel } from '@/lib/find-in-page' import { KEYBIND_ACTIONS } from '@/lib/keybinds/actions' -import { comboAllowedInInput } from '@/lib/keybinds/combo' +import { actionAllowedInInput } from '@/lib/keybinds/combo' import { $findInPage, closeFindBar, @@ -202,10 +202,9 @@ describe('find-in-page keybind registration', () => { }) it('mod+f fires from inside a textarea (browser find behavior)', () => { - // The runtime consults comboAllowedInInput before dispatching a combo - // while an editable element owns focus; if mod combos ever stop - // qualifying, ⌘F from the composer would type 'f' instead of opening find. - expect(comboAllowedInInput('mod+f')).toBe(true) + // The runtime consults actionAllowedInInput before dispatching while an + // editable element owns focus; ⌘F should still open find from the composer. + expect(actionAllowedInInput('view.findInPage', 'mod+f')).toBe(true) }) it('registers the step pair unbound so it cannot conflict with view.toggleReview', () => { diff --git a/apps/desktop/src/lib/keybinds/actions.ts b/apps/desktop/src/lib/keybinds/actions.ts index b042cd4f6e2b6..40b7a7f1a4d39 100644 --- a/apps/desktop/src/lib/keybinds/actions.ts +++ b/apps/desktop/src/lib/keybinds/actions.ts @@ -140,7 +140,7 @@ export const KEYBIND_ACTIONS: readonly KeybindActionMeta[] = [ // is a no-op. ⌘⇧T reopens the last closed tab where it was. { id: 'view.closeTab', category: 'view', defaults: ['mod+w'] }, { id: 'view.reopenTab', category: 'view', defaults: ['mod+shift+t'] }, - // ⌘F — open the find-in-page bar. `comboAllowedInInput` lets the combo + // ⌘F — open the find-in-page bar. `actionAllowedInInput` lets this action // fire from inside a textarea / contenteditable (matches browser behavior // so typing in the composer and pressing ⌘F focuses find, not 'f'). { id: 'view.findInPage', category: 'view', defaults: ['mod+f'] }, diff --git a/apps/desktop/src/lib/keybinds/combo.test.ts b/apps/desktop/src/lib/keybinds/combo.test.ts index 3147538ac3aa1..7b1bd219cb41b 100644 --- a/apps/desktop/src/lib/keybinds/combo.test.ts +++ b/apps/desktop/src/lib/keybinds/combo.test.ts @@ -118,13 +118,25 @@ describe('formatCombo — honest Control labels', () => { }) }) -describe('comboAllowedInInput', () => { - it('lets ctrl combos fire while typing (e.g. ⌃Tab from the composer)', async () => { - const { comboAllowedInInput } = await loadCombo('MacIntel') - - expect(comboAllowedInInput('ctrl+tab')).toBe(true) - expect(comboAllowedInInput('ctrl+shift+tab')).toBe(true) - expect(comboAllowedInInput('mod+k')).toBe(true) - expect(comboAllowedInInput('shift+x')).toBe(false) +describe('actionAllowedInInput', () => { + it('keeps only explicit text-entry-safe global actions active while typing', async () => { + const { actionAllowedInInput } = await loadCombo('MacIntel') + + expect(actionAllowedInInput('session.next', 'ctrl+tab')).toBe(true) + expect(actionAllowedInInput('session.prev', 'ctrl+shift+tab')).toBe(true) + expect(actionAllowedInInput('nav.commandPalette', 'mod+k')).toBe(true) + expect(actionAllowedInInput('view.findInPage', 'mod+f')).toBe(true) + expect(actionAllowedInInput('nav.skills', 'mod+k')).toBe(false) + expect(actionAllowedInInput('view.showTerminal', 'ctrl+`')).toBe(false) + expect(actionAllowedInInput('profile.next', 'mod+shift+]')).toBe(false) + }) + + it('leaves text navigation chords with the focused input even when rebound to an allowed action', async () => { + const { actionAllowedInInput } = await loadCombo('Win32') + + expect(actionAllowedInInput('session.next', 'mod+right')).toBe(false) + expect(actionAllowedInInput('session.prev', 'mod+left')).toBe(false) + expect(actionAllowedInInput('nav.commandPalette', 'mod+pageup')).toBe(false) + expect(actionAllowedInInput('view.findInPage', 'mod+end')).toBe(false) }) }) diff --git a/apps/desktop/src/lib/keybinds/combo.ts b/apps/desktop/src/lib/keybinds/combo.ts index 78f91e96e220c..7f6f757dcee40 100644 --- a/apps/desktop/src/lib/keybinds/combo.ts +++ b/apps/desktop/src/lib/keybinds/combo.ts @@ -223,8 +223,27 @@ export function isEditableTarget(target: EventTarget | null): boolean { ) } -// A primary modifier (Cmd/Ctrl/Control) fires even while typing (e.g. ⌘K or -// ⌃Tab from the composer); bare/Shift-only combos are suppressed in inputs. -export function comboAllowedInInput(combo: string): boolean { - return /^(?:mod|ctrl)(?:\+|$)/.test(combo) +const INPUT_SAFE_ACTIONS = new Set([ + 'composer.modelPicker', + 'composer.voice', + 'keybinds.openPanel', + 'nav.commandPalette', + 'session.next', + 'session.prev', + 'view.findInPage' +]) + +const TEXT_NAVIGATION_KEYS = new Set(['up', 'down', 'left', 'right', 'home', 'end', 'pageup', 'pagedown']) + +// Only explicit text-entry-safe actions fire while typing. Editing/navigation +// chords such as Ctrl+Arrow/PageUp must stay with the input even if a user +// rebinds them to a global navigation action. +export function actionAllowedInInput(actionId: string, combo: string): boolean { + const base = combo.split('+').pop() + + if (base && TEXT_NAVIGATION_KEYS.has(base)) { + return false + } + + return INPUT_SAFE_ACTIONS.has(actionId) } From b55677077fad79c7029ce1e56ccd1f157a149f48 Mon Sep 17 00:00:00 2001 From: Jakub Wolniewicz <4850809+frizikk@users.noreply.github.com> Date: Tue, 11 Aug 2026 19:17:15 +0200 Subject: [PATCH 010/376] fix(desktop): keep completion selection visible --- .../chat/composer/trigger-popover.test.tsx | 155 +++++++++++++++++- .../src/app/chat/composer/trigger-popover.tsx | 66 +++++++- 2 files changed, 214 insertions(+), 7 deletions(-) diff --git a/apps/desktop/src/app/chat/composer/trigger-popover.test.tsx b/apps/desktop/src/app/chat/composer/trigger-popover.test.tsx index 79da0032c018e..52a0df6ee9f35 100644 --- a/apps/desktop/src/app/chat/composer/trigger-popover.test.tsx +++ b/apps/desktop/src/app/chat/composer/trigger-popover.test.tsx @@ -1,4 +1,4 @@ -import { cleanup, render, screen } from '@testing-library/react' +import { cleanup, fireEvent, render, screen } from '@testing-library/react' import { afterEach, describe, expect, it, vi } from 'vitest' import { I18nProvider } from '@/i18n' @@ -25,11 +25,44 @@ function renderPopover(kind: '@' | '/', loading = false) { return { ...rendered, onHover, onPick } } -describe('ComposerTriggerPopover i18n', () => { - afterEach(() => { - cleanup() - }) +function slashItem(command: string) { + return { + id: command, + type: 'slash', + label: command.slice(1), + metadata: { command, display: command, group: 'Skills', meta: '', rawText: command } + } +} + +function rect(top: number, bottom: number): DOMRect { + return { + bottom, + height: bottom - top, + left: 0, + right: 320, + toJSON: () => ({}), + top, + width: 320, + x: 0, + y: top + } +} + +function mockDrawerViewport(drawer: HTMLElement) { + Object.defineProperty(drawer, 'clientHeight', { configurable: true, value: 200 }) + Object.defineProperty(drawer, 'clientTop', { configurable: true, value: 1 }) + vi.spyOn(drawer, 'getBoundingClientRect').mockReturnValue(rect(100, 302)) +} + +function mockRowPosition(row: HTMLElement, top: number, bottom: number) { + return vi.spyOn(row, 'getBoundingClientRect').mockReturnValue(rect(top, bottom)) +} + +afterEach(() => { + cleanup() +}) +describe('ComposerTriggerPopover i18n', () => { it('renders localized empty lookup copy for @ references', () => { const { container } = renderPopover('@') @@ -55,3 +88,115 @@ describe('ComposerTriggerPopover i18n', () => { expect(container.textContent).toContain('/help') }) }) + +describe('ComposerTriggerPopover keyboard scrolling', () => { + const items = [slashItem('/first'), slashItem('/second'), slashItem('/third')] + + function popover(activeIndex: number, onHover = vi.fn(), nextItems = items) { + return ( + + + + ) + } + + it('keeps keyboard navigation visible and restores the group header on wrap', () => { + const { container, rerender } = render(popover(0)) + const drawer = container.querySelector('[data-slot="composer-completion-drawer"]') as HTMLElement + const ancestor = drawer.parentElement as HTMLElement + const secondRow = screen.getAllByRole('button')[1] + + mockDrawerViewport(drawer) + mockRowPosition(secondRow, 290, 330) + ancestor.scrollTop = 48 + drawer.scrollTop = 96 + rerender(popover(1)) + + const activeRow = container.querySelector('[data-highlighted]') as HTMLElement + + expect(activeRow.textContent).toContain('/second') + expect(drawer.scrollTop).toBe(125) + expect(ancestor.scrollTop).toBe(48) + + drawer.scrollTop = 96 + rerender(popover(0)) + + expect(drawer.scrollTop).toBe(0) + }) + + it('uses the nearest drawer edge for upward, visible, and oversized rows', () => { + const { container, rerender } = render(popover(0)) + const drawer = container.querySelector('[data-slot="composer-completion-drawer"]') as HTMLElement + const rows = screen.getAllByRole('button') + + mockDrawerViewport(drawer) + mockRowPosition(rows[1], 80, 120) + const thirdRowRect = mockRowPosition(rows[2], 150, 180) + + drawer.scrollTop = 50 + rerender(popover(1)) + expect(drawer.scrollTop).toBe(29) + + rerender(popover(2)) + expect(drawer.scrollTop).toBe(29) + + thirdRowRect.mockReturnValue(rect(80, 340)) + drawer.scrollTop = 50 + rerender(popover(2, vi.fn(), [...items])) + expect(drawer.scrollTop).toBe(50) + + thirdRowRect.mockReturnValue(rect(150, 400)) + rerender(popover(2, vi.fn(), [...items, slashItem('/fourth')])) + expect(drawer.scrollTop).toBe(99) + + thirdRowRect.mockReturnValue(rect(0, 250)) + drawer.scrollTop = 100 + rerender(popover(2, vi.fn(), [...items, slashItem('/fifth')])) + expect(drawer.scrollTop).toBe(49) + }) + + it('does not scroll for a hover echo and consumes the hover marker', () => { + const onHover = vi.fn() + const { container, rerender } = render(popover(0, onHover)) + const rows = screen.getAllByRole('button') + const drawer = container.querySelector('[data-slot="composer-completion-drawer"]') as HTMLElement + + mockDrawerViewport(drawer) + mockRowPosition(rows[2], 311, 331) + drawer.scrollTop = 40 + fireEvent.mouseEnter(rows[2]) + expect(onHover).toHaveBeenCalledWith(2) + + rerender(popover(2, onHover)) + expect(drawer.scrollTop).toBe(40) + + rerender(popover(0, onHover)) + rerender(popover(2, onHover)) + + expect(drawer.scrollTop).toBe(30) + expect((container.querySelector('[data-highlighted]') as HTMLElement).textContent).toContain('/third') + }) + + it('does not leave a stale hover marker when the active row is hovered', () => { + const onHover = vi.fn() + const { container, rerender } = render(popover(1, onHover)) + const drawer = container.querySelector('[data-slot="composer-completion-drawer"]') as HTMLElement + const activeRow = screen.getAllByRole('button')[1] + + mockDrawerViewport(drawer) + mockRowPosition(activeRow, 311, 331) + fireEvent.mouseEnter(activeRow) + expect(onHover).toHaveBeenCalledWith(1) + + rerender(popover(1, onHover, [...items, slashItem('/fourth')])) + + expect(drawer.scrollTop).toBe(30) + }) +}) diff --git a/apps/desktop/src/app/chat/composer/trigger-popover.tsx b/apps/desktop/src/app/chat/composer/trigger-popover.tsx index 32b4032303408..664d613062b11 100644 --- a/apps/desktop/src/app/chat/composer/trigger-popover.tsx +++ b/apps/desktop/src/app/chat/composer/trigger-popover.tsx @@ -1,5 +1,5 @@ import type { Unstable_TriggerItem } from '@assistant-ui/core' -import { Fragment } from 'react' +import { Fragment, useEffect, useRef } from 'react' import { referenceKind, referenceStyle } from '@/components/assistant-ui/reference-kinds' import { Codicon } from '@/components/ui/codicon' @@ -91,6 +91,62 @@ export function ComposerTriggerPopover({ const copy = t.composer const isSlash = kind === '/' const isEmoji = kind === ':' + const listRef = useRef(null) + const hoverIndexRef = useRef(-1) + + // Only keyboard navigation should move the drawer. A hover echo already points + // at a visible row and scrolling it can shift another row under the pointer. + // eslint-disable-next-line no-restricted-syntax -- legitimate non-atom ref write (see eslint rule comment) + useEffect(() => { + const list = listRef.current + + if (!list) { + return + } + + const isHoverEcho = activeIndex === hoverIndexRef.current + + hoverIndexRef.current = -1 + + if (isHoverEcho) { + return + } + + if (activeIndex === 0) { + // `nearest` keeps the first row visible but can leave its group header + // clipped, so wrapping to the beginning restores the complete top edge. + list.scrollTop = 0 + + return + } + + const highlighted = list.querySelector('[data-highlighted]') + + if (!highlighted) { + return + } + + // Keep scrolling local to the drawer. `scrollIntoView` may also move the + // transcript or window because it operates on every scrollable ancestor. + const listRect = list.getBoundingClientRect() + const highlightedRect = highlighted.getBoundingClientRect() + const visibleTop = listRect.top + list.clientTop + const visibleBottom = visibleTop + list.clientHeight + const topDelta = highlightedRect.top - visibleTop + const bottomDelta = highlightedRect.bottom - visibleBottom + const overflowsTop = topDelta < 0 + const overflowsBottom = bottomDelta > 0 + + // A row that is fully visible needs no movement. An oversized row that + // spans both edges already covers the viewport, so moving it would not + // reveal the whole row and would only add churn. Otherwise align whichever + // edge requires the shorter movement, matching `block: nearest` semantics. + if (overflowsTop === overflowsBottom) { + return + } + + list.scrollTop += Math.abs(topDelta) < Math.abs(bottomDelta) ? topDelta : bottomDelta + }, [activeIndex, items]) let lastGroup: string | undefined @@ -100,6 +156,7 @@ export function ComposerTriggerPopover({ data-slot="composer-completion-drawer" data-state="open" onMouseDown={event => event.preventDefault()} + ref={listRef} role="listbox" > {scope &&
{referenceStyle(scope).label}
} @@ -146,7 +203,12 @@ export function ComposerTriggerPopover({ className={ROW_CLASS} data-highlighted={active ? '' : undefined} onClick={() => onPick(item)} - onMouseEnter={() => onHover(index)} + onMouseEnter={() => { + // React bails out when hovering the already-active row. Do + // not leave a marker behind for a later items refresh. + hoverIndexRef.current = index === activeIndex ? -1 : index + onHover(index) + }} type="button" > {isEmoji ? ( From d7e95315f7211054086268f5c25e01366cda2682 Mon Sep 17 00:00:00 2001 From: LemonSchneid <122685409+LemonSchneid@users.noreply.github.com> Date: Thu, 13 Aug 2026 18:49:51 -0600 Subject: [PATCH 011/376] fix(desktop): expand HUD transcript when resized Use the available HUD window height for non-empty transcript scrollback instead of a fixed glance-band cap. This makes the corner resize affordance reveal additional conversation content while preserving the compact empty HUD state. --- apps/desktop/src/app/hud/hud-shell.tsx | 25 ++++++++++++++----------- apps/desktop/src/app/hud/layout.test.ts | 22 ++++++++++++++++++++++ apps/desktop/src/app/hud/layout.ts | 22 ++++++++++++++++++++++ 3 files changed, 58 insertions(+), 11 deletions(-) create mode 100644 apps/desktop/src/app/hud/layout.test.ts create mode 100644 apps/desktop/src/app/hud/layout.ts diff --git a/apps/desktop/src/app/hud/hud-shell.tsx b/apps/desktop/src/app/hud/hud-shell.tsx index 2ca5e90f2fe97..9b5d1755039ef 100644 --- a/apps/desktop/src/app/hud/hud-shell.tsx +++ b/apps/desktop/src/app/hud/hud-shell.tsx @@ -18,6 +18,7 @@ import { titlebarButtonClass } from '../shell/titlebar' import { useHudClickThrough } from './click-through' import { useHudGlass } from './glass' import { useHudGoto, useReportHudSession } from './handoff' +import { hudTranscriptHeight } from './layout' import { useHudResizeHandle } from './resize-handle' import { useHudThreadFocus } from './thread-focus' @@ -48,15 +49,6 @@ const HUD_COLLAPSE_MS = Math.round(HUD_FADE_MS * 0.66) * so an empty transcript measures a true zero instead of a 12px strip. */ const HUD_SHEET_OVERHANG_PX = 12 -/** Ceiling on the transcript band, which still auto-sizes up from 0. It reads - * over another app, so it is a glance rather than a panel: whichever of these - * is smaller wins, so a tall HUD doesn't turn the band into a second window - * and a short one doesn't get swallowed by it. */ -const HUD_BAND_MAX_PX = 152 -const HUD_BAND_MAX_FRACTION = 0.42 - -const hudBandMaxPx = () => Math.min(window.innerHeight * HUD_BAND_MAX_FRACTION, HUD_BAND_MAX_PX) - /** Composer on top, transcript always hanging below it — Spotlight's shape, * rather than flipping to follow the screen edge the HUD is parked against. */ const HUD_THREAD_ALWAYS_BELOW = true @@ -302,7 +294,14 @@ export function HudShell() { const contentSpan = text < 1 ? 0 : text + HUD_SHEET_OVERHANG_PX - const visible = Math.min(hudBandMaxPx(), Math.max(0, Math.round(contentSpan))) + // Once the HUD has a transcript, a resize must buy readable scrollback. + // The old glance-band ceiling froze this at 152px and turned every extra + // pixel of native window height into empty transparent chrome. + const visible = hudTranscriptHeight({ + barHeight: root.querySelector('[data-slot="composer-dock"]')?.getBoundingClientRect().height ?? 0, + contentHeight: contentSpan, + viewportHeight: window.innerHeight + }) root.style.setProperty('--hud-band-height', `${visible}px`) @@ -323,12 +322,16 @@ export function HudShell() { } // The viewport mounts async (lazy chat surface); poll briefly until it - // exists, then let the ResizeObserver own it. + // exists, then let the ResizeObserver own it. Window resize is separate: + // the transcript's rows may not change size, but the available scrollback + // must, so observing the rows alone cannot update the band. measure() const probe = setInterval(measure, 500) + window.addEventListener('resize', measure) return () => { clearInterval(probe) + window.removeEventListener('resize', measure) ro.disconnect() } }, []) diff --git a/apps/desktop/src/app/hud/layout.test.ts b/apps/desktop/src/app/hud/layout.test.ts new file mode 100644 index 0000000000000..6f7eea4595ff0 --- /dev/null +++ b/apps/desktop/src/app/hud/layout.test.ts @@ -0,0 +1,22 @@ +import { describe, expect, it } from 'vitest' + +import { hudTranscriptHeight } from './layout' + +describe('hudTranscriptHeight', () => { + it('uses the resized window space for a non-empty transcript', () => { + // The transcript is intentionally not constrained to its content height: + // resizing HUD must reveal more scrollback instead of growing an empty + // transparent window below a fixed-height chat band. + expect( + hudTranscriptHeight({ + barHeight: 58, + contentHeight: 72, + viewportHeight: 640 + }) + ).toBe(582) + }) + + it('keeps an empty HUD collapsed', () => { + expect(hudTranscriptHeight({ barHeight: 58, contentHeight: 0, viewportHeight: 640 })).toBe(0) + }) +}) diff --git a/apps/desktop/src/app/hud/layout.ts b/apps/desktop/src/app/hud/layout.ts new file mode 100644 index 0000000000000..081cb5fec3ff2 --- /dev/null +++ b/apps/desktop/src/app/hud/layout.ts @@ -0,0 +1,22 @@ +export interface HudTranscriptHeightInput { + /** Measured message rows, including the HUD sheet overhang. */ + contentHeight: number + /** The composer's measured height. */ + barHeight: number + /** The HUD window's current inner height. */ + viewportHeight: number +} + +/** + * The HUD transcript owns all available space after the composer once there is + * something to show. A resizable HUD must expose more scrollback as it grows; + * sizing the band to its message rows instead leaves a larger empty native + * window around the same fixed-height transcript. + */ +export function hudTranscriptHeight({ barHeight, contentHeight, viewportHeight }: HudTranscriptHeightInput): number { + if (contentHeight < 1) { + return 0 + } + + return Math.max(0, Math.round(viewportHeight - barHeight)) +} From 4712721033202b8109c9ee19d6a2b4aa7ca626ac Mon Sep 17 00:00:00 2001 From: AlexDev_ <56083016+alexdev03@users.noreply.github.com> Date: Sat, 8 Aug 2026 19:41:32 +0200 Subject: [PATCH 012/376] perf(desktop): pause decorative animations when unfocused --- .../src/lib/renderer-loop-pause.test.ts | 34 +++++++++++++++++++ apps/desktop/src/lib/renderer-loop-pause.ts | 22 ++++++++++++ apps/desktop/src/main.tsx | 6 ++++ apps/desktop/src/styles.css | 21 ++++++++++++ 4 files changed, 83 insertions(+) create mode 100644 apps/desktop/src/lib/renderer-loop-pause.test.ts diff --git a/apps/desktop/src/lib/renderer-loop-pause.test.ts b/apps/desktop/src/lib/renderer-loop-pause.test.ts new file mode 100644 index 0000000000000..c6fe0e91de3e0 --- /dev/null +++ b/apps/desktop/src/lib/renderer-loop-pause.test.ts @@ -0,0 +1,34 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' + +import { + installRendererAnimationPauseState, + RENDERER_ANIMATIONS_PAUSED_ATTRIBUTE +} from './renderer-loop-pause' + +describe('installRendererAnimationPauseState', () => { + afterEach(() => { + document.documentElement.removeAttribute(RENDERER_ANIMATIONS_PAUSED_ATTRIBUTE) + vi.restoreAllMocks() + }) + + it('pauses on blur, resumes on focus, and cleans up its root state', () => { + let focused = true + vi.spyOn(document, 'hasFocus').mockImplementation(() => focused) + + const dispose = installRendererAnimationPauseState() + expect(document.documentElement.hasAttribute(RENDERER_ANIMATIONS_PAUSED_ATTRIBUTE)).toBe(false) + + focused = false + window.dispatchEvent(new Event('blur')) + expect(document.documentElement.hasAttribute(RENDERER_ANIMATIONS_PAUSED_ATTRIBUTE)).toBe(true) + + focused = true + window.dispatchEvent(new Event('focus')) + expect(document.documentElement.hasAttribute(RENDERER_ANIMATIONS_PAUSED_ATTRIBUTE)).toBe(false) + + focused = false + window.dispatchEvent(new Event('blur')) + dispose() + expect(document.documentElement.hasAttribute(RENDERER_ANIMATIONS_PAUSED_ATTRIBUTE)).toBe(false) + }) +}) diff --git a/apps/desktop/src/lib/renderer-loop-pause.ts b/apps/desktop/src/lib/renderer-loop-pause.ts index 88b9e3559bc74..55fb19b1c78c1 100644 --- a/apps/desktop/src/lib/renderer-loop-pause.ts +++ b/apps/desktop/src/lib/renderer-loop-pause.ts @@ -3,6 +3,8 @@ interface WindowStatePayload { isVisible?: boolean } +export const RENDERER_ANIMATIONS_PAUSED_ATTRIBUTE = 'data-renderer-animations-paused' + export function createRendererLoopPauseController(onChange: () => void, { pauseWhenUnfocused = true } = {}) { let windowPaused = false let windowFocused = document.hasFocus() @@ -48,3 +50,23 @@ export function createRendererLoopPauseController(onChange: () => void, { pauseW isPaused: () => document.visibilityState === 'hidden' || (pauseWhenUnfocused && !windowFocused) || windowPaused } } + +/** + * Mirrors the main window's observability onto :root so continuous decorative + * CSS animations can sleep with the JS renderer loops. The caller owns the + * returned cleanup; overlay windows intentionally do not install this state. + */ +export function installRendererAnimationPauseState(): () => void { + const root = document.documentElement + let controller: ReturnType + + const sync = () => root.toggleAttribute(RENDERER_ANIMATIONS_PAUSED_ATTRIBUTE, controller.isPaused()) + + controller = createRendererLoopPauseController(sync) + sync() + + return () => { + controller.dispose() + root.removeAttribute(RENDERER_ANIMATIONS_PAUSED_ATTRIBUTE) + } +} diff --git a/apps/desktop/src/main.tsx b/apps/desktop/src/main.tsx index b2ef156a436b5..39e21ce743492 100644 --- a/apps/desktop/src/main.tsx +++ b/apps/desktop/src/main.tsx @@ -25,6 +25,7 @@ import { RootTooltipProvider } from './components/ui/tooltip' import { I18nProvider } from './i18n' import { installClipboardShim } from './lib/clipboard' import { queryClient } from './lib/query-client' +import { installRendererAnimationPauseState } from './lib/renderer-loop-pause' import { ThemeProvider } from './themes/context' installClipboardShim() @@ -50,6 +51,11 @@ if (winParam === 'overlay') { } else if (winParam === 'wake') { void import('./app/wake-indicator/wake-indicator-root').then(({ mountWakeIndicator }) => mountWakeIndicator()) } else { + // CSS animations do not inherit Chromium's JS-loop pause policy. Mirror the + // main window's focus/visibility state to :root so decorative infinite + // animations stop producing frames when nobody can see them. + installRendererAnimationPauseState() + createRoot(document.getElementById('root')!).render( diff --git a/apps/desktop/src/styles.css b/apps/desktop/src/styles.css index 6bd413648fd9a..16a6da217b94d 100644 --- a/apps/desktop/src/styles.css +++ b/apps/desktop/src/styles.css @@ -28,6 +28,27 @@ } } +/* Continuous decorative animations otherwise keep Chromium's renderer awake + behind another app. main.tsx owns this attribute only for the primary + window; wake/pet overlays retain their purpose-built visibility behavior. */ +:root[data-renderer-animations-paused] :is( + .shimmer, + .quest-glow, + .pet-egg, + .pet-egg__glow, + .pet-egg-shadow, + .pet-wobble, + .progress-slide, + .kanban-arc + ), +:root[data-renderer-animations-paused] .arc-border::before, +:root[data-renderer-animations-paused] + [data-slot='aui_assistant-message-content'] + .aui-md + [data-slot='code-card'][data-streaming='true'] { + animation-play-state: paused !important; +} + /* Sidebar sections: tall viewports give each its own scroller; compact ones (this variant) flatten everything into one shared scroll. See ChatSidebar. */ @custom-variant compact (@media (max-height: 768px)); From 62eefff697b33f0cb0c1332d2c9b64fe89164699 Mon Sep 17 00:00:00 2001 From: Aleks Clark Date: Thu, 13 Aug 2026 10:52:28 -0500 Subject: [PATCH 013/376] perf(desktop): bound long-running app resource use MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Persist backend ownership for reliable cleanup, park inactive panes, and evict unreferenced transcripts so Desktop stays responsive over long sessions. 💘 Generated with Crush Assisted-by: Crush:gpt-5.6 --- .../electron/backend-ownership.test.ts | 223 +++++++++++ apps/desktop/electron/backend-ownership.ts | 227 +++++++++++ apps/desktop/electron/main.ts | 376 ++++++++++++++++-- apps/desktop/src/app/contrib/controller.tsx | 8 +- .../hooks/use-session-state-cache.test.tsx | 27 ++ .../session/hooks/use-session-state-cache.ts | 155 +++++--- .../app/session/session-state-cache.test.ts | 133 +++++++ .../src/app/session/session-state-cache.ts | 135 +++++++ .../assistant-ui/thread/list.test.ts | 12 +- .../components/assistant-ui/thread/list.tsx | 20 +- .../pane-shell/pane-lifecycle.test.ts | 72 ++++ .../components/pane-shell/pane-lifecycle.ts | 71 ++++ .../components/pane-shell/pane-visibility.ts | 8 + .../pane-shell/tree/renderer/track-model.ts | 4 + .../pane-shell/tree/renderer/tree-group.tsx | 60 ++- .../src/store/session-states-eviction.test.ts | 14 +- apps/desktop/src/store/session-states.test.ts | 23 ++ apps/desktop/src/store/session-states.ts | 43 +- hermes_cli/web_server.py | 151 ++++++- 19 files changed, 1594 insertions(+), 168 deletions(-) create mode 100644 apps/desktop/electron/backend-ownership.test.ts create mode 100644 apps/desktop/electron/backend-ownership.ts create mode 100644 apps/desktop/src/app/session/session-state-cache.test.ts create mode 100644 apps/desktop/src/app/session/session-state-cache.ts create mode 100644 apps/desktop/src/components/pane-shell/pane-lifecycle.test.ts create mode 100644 apps/desktop/src/components/pane-shell/pane-lifecycle.ts diff --git a/apps/desktop/electron/backend-ownership.test.ts b/apps/desktop/electron/backend-ownership.test.ts new file mode 100644 index 0000000000000..b2afbae7b0b50 --- /dev/null +++ b/apps/desktop/electron/backend-ownership.test.ts @@ -0,0 +1,223 @@ +import assert from 'node:assert/strict' + +import { test, vi } from 'vitest' + +import { + backendCommandMatches, + type BackendIdentity, + createBackendOwnership, + createBackendShutdownCoordinator, + parseBackendOwnership +} from './backend-ownership' + +function memoryStore(initial = '') { + let contents = initial + + return { + read: () => contents, + value: () => contents, + write: (next: string) => { + contents = next + } + } +} + +function identity(overrides: Partial = {}): BackendIdentity { + return { + nonce: 'nonce-42', + pid: 42, + profile: 'default', + startMarker: 'os-start-123', + ...overrides + } +} + +function ownershipEntry(overrides: Partial = {}) { + return { command: 'hermes serve --port 0', ...identity(overrides) } +} + +function stored(entries: object[]): string { + return JSON.stringify({ backends: entries }) +} + +function deferred() { + let resolve!: () => void + + const promise = new Promise(done => { + resolve = done + }) + + return { promise, resolve } +} + +function createOwnership(store = memoryStore(), overrides: Partial[0]> = {}) { + return createBackendOwnership({ + matchesIdentity: async () => true, + stop: () => {}, + store, + ...overrides + }) +} + +test('claim persists the caller-supplied exact identity before resolving', async () => { + const store = memoryStore() + const ownership = createOwnership(store) + const claim = ownershipEntry() + + assert.deepEqual(await ownership.claim(claim), claim) + assert.deepEqual(parseBackendOwnership(store.value()), [claim]) +}) + +test('incomplete claims and persisted records are rejected', async () => { + const store = memoryStore( + stored([ + ownershipEntry(), + { ...ownershipEntry({ pid: 43 }), startMarker: '' }, + { ...ownershipEntry({ pid: 44 }), nonce: undefined }, + { ...ownershipEntry({ pid: 45 }), profile: undefined } + ]) + ) + + const ownership = createOwnership(store) + + await assert.rejects(ownership.claim({ ...ownershipEntry(), startMarker: '' }), /complete process identity/) + assert.deepEqual(parseBackendOwnership(store.value()), [ownershipEntry()]) +}) + +test('failed persistence awaits asynchronous cleanup of the exact identity', async () => { + const cleanup = deferred() + const stop = vi.fn(() => cleanup.promise) + const expected = new Error('disk full') + const claim = ownershipEntry({ pid: 43 }) + + const ownership = createOwnership(memoryStore(), { + stop, + store: { + read: () => null, + write: () => { + throw expected + } + } + }) + + let rejected = false + + const result = ownership.claim(claim).catch(error => { + rejected = true + throw error + }) + + await Promise.resolve() + assert.equal(rejected, false) + assert.deepEqual(stop.mock.calls, [[claim]]) + + cleanup.resolve() + await assert.rejects(result, expected) + assert.equal(rejected, true) +}) + +test('startup reap drops a confirmed PID reuse mismatch without stopping it', async () => { + const entry = ownershipEntry() + const store = memoryStore(stored([entry])) + const matchesIdentity = vi.fn(async () => false) + const stop = vi.fn() + const ownership = createOwnership(store, { matchesIdentity, stop }) + + assert.deepEqual(await ownership.reapOrphans(), []) + assert.deepEqual(matchesIdentity.mock.calls, [[entry]]) + assert.equal(stop.mock.calls.length, 0) + assert.deepEqual(parseBackendOwnership(store.value()), []) +}) + +test('startup reap preserves records when exact identity probing is uncertain or fails', async () => { + const uncertain = ownershipEntry({ pid: 50, nonce: 'uncertain' }) + const failed = ownershipEntry({ pid: 51, nonce: 'failed' }) + const store = memoryStore(stored([uncertain, failed])) + const stop = vi.fn() + + const ownership = createOwnership(store, { + matchesIdentity: async entry => { + if (entry.pid === failed.pid) { + throw new Error('process table unavailable') + } + + return undefined + }, + stop + }) + + assert.deepEqual(await ownership.reapOrphans(), []) + assert.equal(stop.mock.calls.length, 0) + assert.deepEqual(parseBackendOwnership(store.value()), [uncertain, failed]) +}) + +test('startup reap passes the full confirmed identity to stop', async () => { + const entry = ownershipEntry({ pid: 52 }) + const store = memoryStore(stored([entry])) + const stop = vi.fn() + const ownership = createOwnership(store, { stop }) + + assert.deepEqual(await ownership.reapOrphans(), [52]) + assert.deepEqual(stop.mock.calls, [[entry]]) + assert.deepEqual(parseBackendOwnership(store.value()), []) +}) + +test('startup reap preserves failed stops for the next launch', async () => { + const entry = ownershipEntry({ pid: 53 }) + const store = memoryStore(stored([entry])) + + const ownership = createOwnership(store, { + stop: () => { + throw new Error('permission denied') + } + }) + + assert.deepEqual(await ownership.reapOrphans(), []) + assert.deepEqual(parseBackendOwnership(store.value()), [entry]) +}) + +test('release removes only the exact identity rather than every record for its PID', () => { + const oldProcess = ownershipEntry({ nonce: 'old', startMarker: 'start-old' }) + const reusedPid = ownershipEntry({ nonce: 'new', startMarker: 'start-new' }) + const store = memoryStore(stored([oldProcess, reusedPid])) + const ownership = createOwnership(store) + + ownership.release(oldProcess) + + assert.deepEqual(parseBackendOwnership(store.value()), [reusedPid]) +}) + +test('backend identity check matches only serve and dashboard invocation shapes', () => { + assert.equal(backendCommandMatches('/venv/bin/hermes serve --port 0'), true) + assert.equal(backendCommandMatches('python -m hermes_cli.main dashboard --no-open'), true) + assert.equal(backendCommandMatches('/venv/bin/hermes --profile work serve --port 0'), true) + assert.equal(backendCommandMatches('"C:\\Hermes Runtime\\hermes.exe" dashboard --no-open'), true) + assert.equal(backendCommandMatches('hermes chat --query serve'), false) + assert.equal(backendCommandMatches('unrelated dashboard'), false) +}) + +test('shutdown coordinator returns one promise and awaits teardown exactly once', async () => { + const completion = deferred() + const teardown = vi.fn(() => completion.promise) + const coordinator = createBackendShutdownCoordinator(teardown) + + const first = coordinator.run() + const second = coordinator.run() + + assert.equal(first, second) + assert.equal(coordinator.hasStarted(), true) + await Promise.resolve() + assert.equal(teardown.mock.calls.length, 1) + + let finished = false + first.then(() => { + finished = true + }) + await Promise.resolve() + assert.equal(finished, false) + + completion.resolve() + await second + assert.equal(finished, true) + assert.equal(coordinator.run(), first) +}) diff --git a/apps/desktop/electron/backend-ownership.ts b/apps/desktop/electron/backend-ownership.ts new file mode 100644 index 0000000000000..feb50fb8e6b48 --- /dev/null +++ b/apps/desktop/electron/backend-ownership.ts @@ -0,0 +1,227 @@ +export interface BackendIdentity { + nonce: string + pid: number + profile: string + startMarker: string +} + +export interface BackendOwnershipEntry extends BackendIdentity { + command?: string +} + +export interface BackendOwnershipStore { + read: () => string | null + write: (contents: string) => void +} + +export interface BackendOwnershipDeps { + matchesIdentity: (identity: BackendIdentity) => Promise + stop: (identity: BackendIdentity) => Promise | void + store: BackendOwnershipStore +} + +export interface BackendClaim extends BackendIdentity { + command?: string +} + +function isNonEmptyString(value: unknown): value is string { + return typeof value === 'string' && value.length > 0 +} + +function isCompleteIdentity(value: unknown): value is BackendIdentity { + if (!value || typeof value !== 'object') { + return false + } + + const candidate = value as Partial + + return ( + Number.isInteger(candidate.pid) && + Number(candidate.pid) > 0 && + isNonEmptyString(candidate.startMarker) && + isNonEmptyString(candidate.nonce) && + isNonEmptyString(candidate.profile) + ) +} + +function identitiesMatch(left: BackendIdentity, right: BackendIdentity): boolean { + return ( + left.pid === right.pid && + left.startMarker === right.startMarker && + left.nonce === right.nonce && + left.profile === right.profile + ) +} + +export function parseBackendOwnership(contents: unknown): BackendOwnershipEntry[] { + let parsed: unknown + + try { + parsed = JSON.parse(String(contents ?? '')) + } catch { + return [] + } + + const values = Array.isArray(parsed) + ? parsed + : parsed && typeof parsed === 'object' && Array.isArray((parsed as { backends?: unknown }).backends) + ? (parsed as { backends: unknown[] }).backends + : [] + + const entries: BackendOwnershipEntry[] = [] + + for (const value of values) { + if (!isCompleteIdentity(value)) { + continue + } + + const candidate = value as BackendOwnershipEntry + + const entry: BackendOwnershipEntry = { + nonce: candidate.nonce, + pid: candidate.pid, + profile: candidate.profile, + startMarker: candidate.startMarker + } + + if (typeof candidate.command === 'string') { + entry.command = candidate.command + } + + if (!entries.some(existing => identitiesMatch(existing, entry))) { + entries.push(entry) + } + } + + return entries +} + +export function serializeBackendOwnership(entries: BackendOwnershipEntry[]): string { + return `${JSON.stringify({ backends: entries }, null, 2)}\n` +} + +/** + * Persistent ownership for local backend roots. + * + * Claiming is asynchronous so a failed persistence transaction can await child + * cleanup before reporting failure to the caller. + */ +export function createBackendOwnership(deps: BackendOwnershipDeps) { + const read = () => parseBackendOwnership(deps.store.read()) + const write = (entries: BackendOwnershipEntry[]) => deps.store.write(serializeBackendOwnership(entries)) + + return { + async claim(claim: BackendClaim): Promise { + if (!isCompleteIdentity(claim)) { + throw new Error('Cannot own a backend without a complete process identity.') + } + + const entry: BackendOwnershipEntry = { + nonce: claim.nonce, + pid: claim.pid, + profile: claim.profile, + startMarker: claim.startMarker + } + + if (typeof claim.command === 'string') { + entry.command = claim.command + } + + try { + const entries = read().filter(candidate => candidate.pid !== entry.pid) + write([...entries, entry]) + } catch (error) { + try { + await deps.stop(entry) + } catch { + // Persistence remains the claim failure even if cleanup also fails. + } + + throw error + } + + return entry + }, + + release(identity: BackendIdentity): void { + if (!isCompleteIdentity(identity)) { + throw new Error('Cannot release a backend without a complete process identity.') + } + + const entries = read() + const next = entries.filter(entry => !identitiesMatch(entry, identity)) + + if (next.length !== entries.length) { + write(next) + } + }, + + async reapOrphans(): Promise { + const entries = read() + const survivors: BackendOwnershipEntry[] = [] + const reaped: number[] = [] + + for (const entry of entries) { + let matches: boolean | undefined + + try { + matches = await deps.matchesIdentity(entry) + } catch { + survivors.push(entry) + + continue + } + + if (matches === false) { + continue + } + + if (matches !== true) { + survivors.push(entry) + + continue + } + + try { + await deps.stop(entry) + reaped.push(entry.pid) + } catch { + // Preserve failed ownership so a later startup can retry it. + survivors.push(entry) + } + } + + write(survivors) + + return reaped + }, + + clear(): void { + write([]) + } + } +} + +export function backendCommandMatches(command: unknown): boolean { + return /(?:^|[\s/\\"])(?:hermes(?:\.exe)?|hermes_cli\.main|hermes_cli[/\\]main\.py)"?(?:\s+(?:--profile|-p)\s+\S+)?\s+(?:serve|dashboard)(?:\s|$)/i.test( + String(command ?? '') + ) +} + +/** Coordinates all quit paths so asynchronous backend teardown runs once. */ +export function createBackendShutdownCoordinator(teardown: () => Promise | void) { + let completion: Promise | undefined + + return { + run(): Promise { + if (!completion) { + completion = Promise.resolve().then(teardown) + } + + return completion + }, + hasStarted(): boolean { + return completion !== undefined + } + } +} diff --git a/apps/desktop/electron/main.ts b/apps/desktop/electron/main.ts index b6addda017c1e..6fd8a0dba830f 100644 --- a/apps/desktop/electron/main.ts +++ b/apps/desktop/electron/main.ts @@ -37,6 +37,11 @@ import { dashboardFallbackArgs, sourceDeclaresServe } from './backend-command' import { createBackendConnectionState } from './backend-connection-state' import { buildDesktopBackendEnv, hermesManagedNodePathEntries, normalizeHermesHomeRoot } from './backend-env' import { isReauthRequiredError, waitForHermesReady } from './backend-health' +import { + backendCommandMatches, + createBackendOwnership, + createBackendShutdownCoordinator +} from './backend-ownership' import { canImportHermesCli, execProbeSync, @@ -467,7 +472,7 @@ if (IS_WINDOWS) { try { app.relaunch({ args: buildNoSandboxRelaunchArgs(process.argv.slice(1)) }) - app.exit(0) + void exitAfterBackendShutdown(0) } catch (error) { console.error(`[hermes] --no-sandbox relaunch failed: ${error?.message || error}`) } @@ -655,6 +660,7 @@ const DESKTOP_CONNECTION_CONFIG_PATH = path.join(app.getPath('userData'), 'conne const DESKTOP_INSTALLATION_PATH = path.join(app.getPath('userData'), 'desktop-installation.json') const DESKTOP_UPDATE_CONFIG_PATH = path.join(app.getPath('userData'), 'updates.json') const DESKTOP_WINDOW_STATE_PATH = path.join(app.getPath('userData'), 'window-state.json') +const DESKTOP_BACKEND_OWNERSHIP_PATH = path.join(app.getPath('userData'), 'backend-ownership.json') // active-profile.json records which Hermes profile the desktop launches its // local backend as. When set, startHermes() passes `hermes --profile // dashboard …`, which deterministically pins HERMES_HOME (see @@ -1122,6 +1128,7 @@ const POOL_IDLE_MS = Math.max(60_000, Number(process.env.HERMES_DESKTOP_POOL_IDL // killing one to honor the soft cap would abort a running agent. const POOL_KEEPALIVE_FRESH_MS = 90_000 let poolIdleReaper = null +let backendOrphanReapPromise = null // Auto-reload budget for renderer crashes, shared by EVERY window (primary, // secondary session, instance) so a crash loop anywhere is suppressed after // the same budget instead of reloading per-window forever. A deterministic @@ -2885,6 +2892,226 @@ function forceKillProcessTree(pid) { } } +function writeBackendOwnership(contents) { + fs.mkdirSync(path.dirname(DESKTOP_BACKEND_OWNERSHIP_PATH), { recursive: true }) + const tempPath = `${DESKTOP_BACKEND_OWNERSHIP_PATH}.${process.pid}.tmp` + + try { + fs.writeFileSync(tempPath, contents, { encoding: 'utf8', mode: 0o600 }) + fs.renameSync(tempPath, DESKTOP_BACKEND_OWNERSHIP_PATH) + } finally { + try { + fs.rmSync(tempPath, { force: true }) + } catch { + void 0 + } + } +} + +function execText(command, args) { + return new Promise((resolve, reject) => { + execFile(command, args, hiddenWindowsChildOptions({ encoding: 'utf8', timeout: 3000 }), (error, stdout) => { + if (error) { + reject(error) + } else { + resolve(String(stdout || '').trim()) + } + }) + }) +} + +async function processStartMarker(pid) { + if (process.platform === 'linux') { + const stat = await fs.promises.readFile(`/proc/${pid}/stat`, 'utf8') + const fields = stat.slice(stat.lastIndexOf(')') + 1).trim().split(/\s+/) + + if (!/^\d+$/.test(fields[19] || '')) { + throw new Error(`Invalid /proc start marker for PID ${pid}`) + } + + return `linux:${fields[19]}` + } + + if (IS_WINDOWS) { + const ticks = await execText('powershell.exe', [ + '-NoProfile', + '-NonInteractive', + '-Command', + `$p = Get-Process -Id ${pid} -ErrorAction Stop; $p.StartTime.ToUniversalTime().Ticks` + ]) + + if (!/^\d+$/.test(ticks)) { + throw new Error(`Invalid Windows start marker for PID ${pid}`) + } + + return `win:${ticks}` + } + + const started = await execText('ps', ['-p', String(pid), '-o', 'lstart=']) + + if (!started) { + throw new Error(`Missing process start marker for PID ${pid}`) + } + + return `ps:${started}` +} + +async function backendCommandForPid(pid) { + try { + const command = IS_WINDOWS ? 'powershell.exe' : 'ps' + + const args = IS_WINDOWS + ? ['-NoProfile', '-NonInteractive', '-Command', `(Get-CimInstance Win32_Process -Filter 'ProcessId = ${pid}').CommandLine`] + : ['-p', String(pid), '-o', 'command='] + + return (await execText(command, args)) || null + } catch { + return null + } +} + +async function processIdentityMatches(identity) { + try { + return (await processStartMarker(identity.pid)) === identity.startMarker + } catch (error) { + return error?.code === 'ENOENT' || error?.code === 'ESRCH' ? false : undefined + } +} + +async function backendIdentityMatches(identity) { + const processMatches = await processIdentityMatches(identity) + + if (processMatches !== true) { + return processMatches + } + + const command = await backendCommandForPid(identity.pid) + + return command === null ? undefined : backendCommandMatches(command) +} + +async function stopOwnedBackend(identity) { + if ((await processIdentityMatches(identity)) !== true) { + return + } + + if (IS_WINDOWS) { + forceKillProcessTree(identity.pid) + } else { + try { + process.kill(-identity.pid, 'SIGTERM') + } catch { + try { + process.kill(identity.pid, 'SIGTERM') + } catch { + return + } + } + + const deadline = Date.now() + 1500 + + while (Date.now() < deadline) { + if ((await processIdentityMatches(identity)) !== true) { + return + } + + await new Promise(resolve => setTimeout(resolve, 50)) + } + + // Revalidate immediately before escalation so PID reuse cannot target a + // replacement process. + if ((await processIdentityMatches(identity)) === true) { + try { + process.kill(-identity.pid, 'SIGKILL') + } catch { + process.kill(identity.pid, 'SIGKILL') + } + } + } + + await new Promise(resolve => setTimeout(resolve, 50)) + const remaining = await processIdentityMatches(identity) + + if (remaining !== false) { + throw new Error(`Backend PID ${identity.pid} did not stop cleanly.`) + } +} + +const backendOwnership = createBackendOwnership({ + matchesIdentity: backendIdentityMatches, + stop: stopOwnedBackend, + store: { + read: () => { + try { + return fs.readFileSync(DESKTOP_BACKEND_OWNERSHIP_PATH, 'utf8') + } catch { + return null + } + }, + write: writeBackendOwnership + } +}) + +let desktopParentStartMarkerPromise = null + +function desktopParentStartMarker() { + desktopParentStartMarkerPromise ??= processStartMarker(process.pid) + + return desktopParentStartMarkerPromise +} + +async function claimBackendChild(child, command, profile, nonce) { + try { + const identity = await backendOwnership.claim({ + command, + nonce, + pid: child.pid, + profile, + startMarker: await processStartMarker(child.pid) + }) + + child.hermesBackendIdentity = identity + + return identity + } catch (error) { + stopBackendChild(child) + await waitForBackendExit(child) + throw new Error(`Could not persist ownership for the Hermes backend: ${error.message}`) + } +} + +function releaseBackendChild(child) { + const identity = child?.hermesBackendIdentity + + if (!identity) { + return + } + + try { + backendOwnership.release(identity) + } catch (error) { + rememberLog(`Could not release backend ownership for PID ${identity.pid}: ${error.message}`) + } +} + +function reapOrphanedBackendsOnce() { + if (!backendOrphanReapPromise) { + backendOrphanReapPromise = backendOwnership + .reapOrphans() + .then(pids => { + if (pids.length) { + rememberLog(`Reaped orphaned desktop backend PID(s): ${pids.join(', ')}`) + } + }) + .catch(error => { + backendOrphanReapPromise = null + throw error + }) + } + + return backendOrphanReapPromise +} + // Before handing off the update on Windows, the desktop MUST stop every backend // it spawned and WAIT for the venv shim to actually unlock. The old code did // `hermesProcess.kill('SIGTERM')` + `app.quit()` fire-and-forget: SIGTERM on @@ -7350,6 +7577,7 @@ const desktopInstallationId = loadOrCreateInstallationId(DESKTOP_INSTALLATION_PA const sshBootstrapCoordinator = createBootstrapCoordinator() let sshQuitTeardownDone = false +let backendQuitTeardownDone = false function sshScopeKey(profile) { return connectionScopeKey(profile) || '' @@ -8050,42 +8278,52 @@ function sendConnectionApplied() { } async function waitForBackendExit(child, timeoutMs = 5000) { - if (!child) { + if (!child || child.exitCode !== null || child.signalCode !== null) { return } - if (child.exitCode !== null || child.signalCode !== null) { + const exited = () => child.exitCode !== null || child.signalCode !== null + + const wait = delay => + new Promise(resolve => { + if (exited()) { + resolve() + + return + } + + const timer = setTimeout(resolve, delay) + child.once('exit', () => { + clearTimeout(timer) + resolve() + }) + }) + + await wait(timeoutMs) + + if (exited()) { return } - await new Promise(resolve => { - const timer = setTimeout(() => { + try { + if (IS_WINDOWS && Number.isInteger(child.pid)) { + forceKillProcessTree(child.pid) + } else if (Number.isInteger(child.pid)) { try { - if (IS_WINDOWS && Number.isInteger(child.pid)) { - forceKillProcessTree(child.pid) - } else if (Number.isInteger(child.pid)) { - // POSIX: SIGKILL the whole group (pgid==pid, start_new_session) so - // MCP grandchildren die with the backend. Fall back to the child. - try { - process.kill(-child.pid, 'SIGKILL') - } catch { - child.kill('SIGKILL') - } - } else { - child.kill('SIGKILL') - } + process.kill(-child.pid, 'SIGKILL') } catch { - // Already gone. + child.kill('SIGKILL') } + } else { + child.kill('SIGKILL') + } + } catch { + return + } - resolve() - }, timeoutMs) - - child.once('exit', () => { - clearTimeout(timer) - resolve() - }) - }) + // Await the escalation as well; do not let shutdown or failed adoption race + // a still-running backend. + await wait(1000) } // The profile the primary (window) backend runs as. readActiveDesktopProfile() @@ -8142,8 +8380,13 @@ async function ensureBackend(profile) { remoteBaseUrl: null } - entry.connectionPromise = spawnPoolBackend(key, entry).catch(error => { - backendPool.delete(key) + entry.connectionPromise = spawnPoolBackend(key, entry).catch(async error => { + if (backendPool.get(key) === entry) { + backendPool.delete(key) + } + + stopBackendChild(entry.process) + await waitForBackendExit(entry.process) throw error }) backendPool.set(key, entry) @@ -8227,6 +8470,7 @@ function startPoolIdleReaper() { // local-spawn portion of startHermes() but without the boot-progress UI, // bootstrap, or remote handling (those belong to the primary backend only). async function spawnPoolBackend(profile, entry) { + await reapOrphanedBackendsOnce() // A profile may point at its OWN remote backend (connection.json // `profiles[name]`), or inherit the app-wide remote (env / global settings). // In either case there is no local child to spawn — we just verify the @@ -8285,6 +8529,9 @@ async function spawnPoolBackend(profile, entry) { rememberLog(`Starting Hermes backend for profile "${profile}" via ${backend.label}`) + const parentStartMarker = await desktopParentStartMarker() + const backendNonce = crypto.randomBytes(16).toString('hex') + const child = spawn( backend.command, backend.args, @@ -8302,11 +8549,11 @@ async function spawnPoolBackend(profile, entry) { // Marks this dashboard backend as desktop-spawned so it runs the cron // scheduler tick loop (the gateway isn't running under the app). HERMES_DESKTOP: '1', - // Our PID so the backend's parent-death watchdog self-exits if we die - // uncleanly (crash / SIGKILL / update handoff) instead of leaking a - // serving backend + its MCP child subtree. See web_server.py - // _start_parent_death_watchdog. + // Exact parent identity lets the backend self-exit after an unclean + // Desktop death without mistaking a reused PID for its owner. HERMES_PARENT_PID: String(process.pid), + HERMES_PARENT_START_MARKER: parentStartMarker, + HERMES_PARENT_NONCE: backendNonce, HERMES_WEB_DIST: webDist, ...(readyFile ? { HERMES_DESKTOP_READY_FILE: readyFile } : {}) }, @@ -8317,6 +8564,7 @@ async function spawnPoolBackend(profile, entry) { entry.process = child entry.token = token + await claimBackendChild(child, `${backend.command} ${backend.args.join(' ')}`, profile, backendNonce) child.stdout.on('data', rememberLog) child.stderr.on('data', rememberLog) @@ -8330,11 +8578,13 @@ async function spawnPoolBackend(profile, entry) { child.once('error', error => { rememberLog(`Hermes backend for profile "${profile}" failed to start: ${error.message}`) + releaseBackendChild(child) backendPool.delete(profile) rejectStart?.(error) }) child.once('exit', (code, signal) => { rememberLog(`Hermes backend for profile "${profile}" exited (${signal || code})`) + releaseBackendChild(child) backendPool.delete(profile) if (!ready) { @@ -8420,6 +8670,26 @@ function stopAllPoolBackends() { } } +const backendShutdown = createBackendShutdownCoordinator(async () => { + const primary = backendConnectionState.invalidate() + const pooled = [...backendPool.values()].map(entry => entry.process).filter(Boolean) + + stopBackendChild(primary) + stopAllPoolBackends() + + if (poolIdleReaper) { + clearInterval(poolIdleReaper) + poolIdleReaper = null + } + + await Promise.all([waitForBackendExit(primary), ...pooled.map(child => waitForBackendExit(child))]) +}) + +async function exitAfterBackendShutdown(code) { + await backendShutdown.run() + app.exit(code) +} + // Returns the profile name whose backend was torn down, or null when the // request is not a profile-delete. The caller uses this to skip ensureBackend // for the just-torn-down profile — otherwise ensureBackend respawns a pool @@ -8455,6 +8725,8 @@ async function prepareProfileDeleteRequest(request) { } async function startHermes() { + await reapOrphanedBackendsOnce() + // Latched-failure short-circuit: once bootstrap has failed in this // process, every subsequent startHermes() call re-throws the same error // without re-running install.ps1. This prevents the renderer's @@ -8578,6 +8850,10 @@ async function startHermes() { await advanceBootProgress('backend.spawn', `Starting Hermes backend via ${backend.label}`, 84) rememberLog(`Starting Hermes backend via ${backend.label}`) + const profile = primaryProfileKey() + const parentStartMarker = await desktopParentStartMarker() + const backendNonce = crypto.randomBytes(16).toString('hex') + const hermesProcess = spawn( backend.command, backend.args, @@ -8600,11 +8876,11 @@ async function startHermes() { // Marks this dashboard backend as desktop-spawned so it runs the cron // scheduler tick loop (the gateway isn't running under the app). HERMES_DESKTOP: '1', - // Our PID so the backend's parent-death watchdog self-exits if we die - // uncleanly (crash / SIGKILL / update handoff) instead of leaking a - // serving backend + its MCP child subtree. See web_server.py - // _start_parent_death_watchdog. + // Exact parent identity lets the backend self-exit after an unclean + // Desktop death without mistaking a reused PID for its owner. HERMES_PARENT_PID: String(process.pid), + HERMES_PARENT_START_MARKER: parentStartMarker, + HERMES_PARENT_NONCE: backendNonce, HERMES_WEB_DIST: webDist, ...(readyFile ? { HERMES_DESKTOP_READY_FILE: readyFile } : {}) }, @@ -8613,10 +8889,13 @@ async function startHermes() { }) ) + await claimBackendChild(hermesProcess, `${backend.command} ${backend.args.join(' ')}`, profile, backendNonce) const processOwner = backendConnectionState.attachProcess(connectionAttempt, hermesProcess) if (!processOwner) { stopBackendChild(hermesProcess) + await waitForBackendExit(hermesProcess) + releaseBackendChild(hermesProcess) throw new Error('Hermes backend start was superseded by a newer connection attempt.') } @@ -8630,6 +8909,8 @@ async function startHermes() { }) hermesProcess.once('error', error => { + releaseBackendChild(hermesProcess) + if (!backendConnectionState.clearForCurrentProcess(processOwner)) { rememberLog(`Ignoring stale Hermes backend error: ${error.message}`) rejectBackendStart?.(new Error('Hermes backend start was superseded by a newer connection attempt.')) @@ -8651,6 +8932,8 @@ async function startHermes() { rejectBackendStart?.(error) }) hermesProcess.once('exit', (code, signal) => { + releaseBackendChild(hermesProcess) + if (!backendConnectionState.clearForCurrentProcess(processOwner)) { rememberLog(`Ignoring stale Hermes backend exit (${signal || code})`) @@ -8741,11 +9024,15 @@ async function startHermes() { logs: hermesLog.slice(-80), ...getWindowState() } - })().catch(error => { + })().catch(async error => { if (!backendConnectionState.clearPromiseForAttempt(connectionAttempt)) { throw error } + const failedProcess = backendConnectionState.invalidate() + stopBackendChild(failedProcess) + await waitForBackendExit(failedProcess) + if (error instanceof FirstRunSetupResetError) { throw error } @@ -9992,7 +10279,7 @@ function createWindow() { try { app.relaunch({ args: buildNoSandboxRelaunchArgs(process.argv.slice(1)) }) - app.exit(0) + void exitAfterBackendShutdown(0) } catch (err) { rememberLog(`[renderer] --no-sandbox relaunch failed: ${err?.message || err}`) } @@ -12750,6 +13037,14 @@ app.on('before-quit', event => { return } + if (!backendQuitTeardownDone) { + event.preventDefault() + void backendShutdown.run().finally(() => { + backendQuitTeardownDone = true + app.quit() + }) + } + if ((sshConnections.size > 0 || sshBootstrapCoordinator.promises().length > 0) && !sshQuitTeardownDone) { event.preventDefault() sshBootstrapCoordinator.cancelAll() @@ -12824,8 +13119,7 @@ app.on('before-quit', event => { disposeTerminalSession(id) } - stopBackendChild(backendConnectionState.getProcess()) - stopAllPoolBackends() + void backendShutdown.run() }) app.on('window-all-closed', () => { diff --git a/apps/desktop/src/app/contrib/controller.tsx b/apps/desktop/src/app/contrib/controller.tsx index a70efb6f6549b..80e16fa272068 100644 --- a/apps/desktop/src/app/contrib/controller.tsx +++ b/apps/desktop/src/app/contrib/controller.tsx @@ -184,7 +184,13 @@ registry.registerMany([ // NO minHeight: a tool panel drags all the way down to its collapsed // header (the sash floors it at COLLAPSED_ZONE_PX and folds the zone to // its rail there). A real floor left a sliver of unusable terminal. - data: { placement: 'bottom', height: '20vh', maxHeight: '80vh', revealOnPreset: true }, + data: { + placement: 'bottom', + height: '20vh', + maxHeight: '80vh', + revealOnPreset: true, + lifecycleKeepAlive: true + }, render: () => }, { diff --git a/apps/desktop/src/app/session/hooks/use-session-state-cache.test.tsx b/apps/desktop/src/app/session/hooks/use-session-state-cache.test.tsx index f96307f361cf2..bb6f4e74a81e2 100644 --- a/apps/desktop/src/app/session/hooks/use-session-state-cache.test.tsx +++ b/apps/desktop/src/app/session/hooks/use-session-state-cache.test.tsx @@ -21,6 +21,7 @@ import { setCurrentServiceTier, setTurnStartedAt } from '@/store/session' +import { $sessionStates } from '@/store/session-states' import { useSessionStateCache } from './use-session-state-cache' @@ -377,6 +378,10 @@ function assistantError(id: string, error: string): ChatMessage { return { id, role: 'assistant', parts: [], error, pending: false } } +function transcriptForCache(id: string): ChatMessage[] { + return [userMessage(`${id}-user`, id), assistantText(`${id}-assistant`, `reply ${id}`)] +} + interface ViewHarnessProps { activeSessionId: string | null onReady: (cache: Cache) => void @@ -405,6 +410,7 @@ describe('useSessionStateCache — cross-thread error isolation', () => { afterEach(() => { cleanup() $messages.set([]) + $sessionStates.set({}) }) it('does not leak a failed turn into another thread on switch', () => { @@ -475,6 +481,27 @@ describe('useSessionStateCache — cross-thread error isolation', () => { expect($messages.get().some(message => message.error === 'OpenRouter 403')).toBe(true) }) + it('evicts the oldest warm transcript with its reverse ownership while retaining lightweight state', () => { + let cache!: Cache + render( (cache = value)} selectedStoredSessionId={null} />) + + act(() => { + for (let index = 0; index < 25; index += 1) { + cache.updateSessionState( + `runtime-${index}`, + state => ({ ...state, messages: transcriptForCache(`message-${index}`) }), + `stored-${index}` + ) + } + }) + + expect(cache.sessionStateByRuntimeIdRef.current.has('runtime-0')).toBe(false) + expect(cache.runtimeIdByStoredSessionIdRef.current.has('stored-0')).toBe(false) + expect($sessionStates.get()['runtime-0']).toMatchObject({ storedSessionId: 'stored-0', busy: false }) + expect($sessionStates.get()['runtime-0']?.messages).toEqual([]) + expect(cache.getRuntimeIdForStoredSession('stored-24')).toBe('runtime-24') + }) + it('only returns a runtime whose cached state owns the requested stored session', () => { let cache!: Cache render( (cache = value)} selectedStoredSessionId={null} />) diff --git a/apps/desktop/src/app/session/hooks/use-session-state-cache.ts b/apps/desktop/src/app/session/hooks/use-session-state-cache.ts index 04fe08513e572..43dd15ea91f79 100644 --- a/apps/desktop/src/app/session/hooks/use-session-state-cache.ts +++ b/apps/desktop/src/app/session/hooks/use-session-state-cache.ts @@ -20,9 +20,10 @@ import { setTurnStartedAt, setYoloActive } from '@/store/session' -import { publishSessionState } from '@/store/session-states' +import { $sessionTiles, publishSessionState, releaseSessionTranscript } from '@/store/session-states' import type { ClientSessionState } from '../../types' +import { SessionStateCache } from '../session-state-cache' import { chatMessageArraysEquivalent } from './use-session-actions/utils' @@ -54,6 +55,7 @@ export function useSessionStateCache({ setMessages }: SessionStateCacheOptions) { const busy = useStore($busy) + const sessionTiles = useStore($sessionTiles) const activeSessionIdRef = useRef(activeSessionId) const selectedStoredSessionIdRef = useRef(selectedStoredSessionId) @@ -81,8 +83,35 @@ export function useSessionStateCache({ selectedStoredSessionIdRef.current = selectedStoredSessionId } - const sessionStateByRuntimeIdRef = useRef(new Map()) const runtimeIdByStoredSessionIdRef = useRef(new Map()) + const sessionStateByRuntimeIdRef = useRef(null!) + + if (sessionStateByRuntimeIdRef.current === null) { + sessionStateByRuntimeIdRef.current = new SessionStateCache({ + isReferenced: (runtimeId, state) => + runtimeId === activeSessionIdRef.current || + state.storedSessionId === selectedStoredSessionIdRef.current || + $sessionTiles + .get() + .some( + tile => + tile.runtimeId === runtimeId || + (state.storedSessionId !== null && tile.storedSessionId === state.storedSessionId) + ), + onEvict: (runtimeId, state) => { + // Ownership is removed with the transcript, but only if both sides still + // describe this exact binding. A recycled runtime must not erase its + // new owner's reverse entry. + if (state.storedSessionId && runtimeIdByStoredSessionIdRef.current.get(state.storedSessionId) === runtimeId) { + runtimeIdByStoredSessionIdRef.current.delete(state.storedSessionId) + } + + releaseSessionTranscript(runtimeId) + } + }) + } + + const sessionStateCache = sessionStateByRuntimeIdRef.current const pendingViewStateRef = useRef<{ sessionId: string; state: ClientSessionState } | null>(null) const viewSyncRafRef = useRef(null) // Runtime id whose transcript currently occupies `$messages` — lets the @@ -94,58 +123,62 @@ export function useSessionStateCache({ setMutableRef(busyRef, busy) }, [busy, busyRef]) - const ensureSessionState = useCallback((sessionId: string, storedSessionId?: string | null) => { - const existing = sessionStateByRuntimeIdRef.current.get(sessionId) - - if (existing) { - if (storedSessionId !== undefined && storedSessionId !== existing.storedSessionId) { - // Stored id changed (e.g. auto-compression rotated it). Create a NEW - // state object rather than mutating in place — updateSessionState needs - // the PREVIOUS state to detect transitions (busy→idle, id rotation). - const updated = { ...existing, storedSessionId } - - sessionStateByRuntimeIdRef.current.set(sessionId, updated) - - // Drop the obsolete stored→runtime reverse mapping as soon as the id - // rotates (e.g. auto-compression forks a continuation). Leaving the - // stale key lets getRuntimeIdForStoredSession resolve the old stored id - // to this runtime, which the compression route-follow logic relies on - // being absent. The rotation signal was previously emitted centrally - // from handleTransition (session-states.ts), but updateSessionState - // now skips publishSessionState (and thus handleTransition) when the - // updater is a no-op — fire it here so the route-follow effect still - // tracks compression without needing a dummy state write. - if (existing.storedSessionId && existing.storedSessionId !== storedSessionId) { - runtimeIdByStoredSessionIdRef.current.delete(existing.storedSessionId) - - // A rotation event needs a real next id — a null/cleared stored id - // is a detach, not a rotation the route-follow effect should chase. - if (storedSessionId && sessionId === $activeSessionId.get()) { - setActiveSessionStoredIdRotation({ - nextStoredSessionId: storedSessionId, - previousStoredSessionId: existing.storedSessionId, - runtimeSessionId: sessionId - }) + const ensureSessionState = useCallback( + (sessionId: string, storedSessionId?: string | null) => { + const existing = sessionStateCache.get(sessionId) + + if (existing) { + if (storedSessionId !== undefined && storedSessionId !== existing.storedSessionId) { + // Stored id changed (e.g. auto-compression rotated it). Create a NEW + // state object rather than mutating in place — updateSessionState needs + // the PREVIOUS state to detect transitions (busy→idle, id rotation). + const updated = { ...existing, storedSessionId } + + // Drop the obsolete stored→runtime reverse mapping as soon as the id + // rotates (e.g. auto-compression forks a continuation). Leaving the + // stale key lets getRuntimeIdForStoredSession resolve the old stored id + // to this runtime, which the compression route-follow logic relies on + // being absent. The rotation signal was previously emitted centrally + // from handleTransition (session-states.ts), but updateSessionState + // now skips publishSessionState (and thus handleTransition) when the + // updater is a no-op — fire it here so the route-follow effect still + // tracks compression without needing a dummy state write. + if (existing.storedSessionId && existing.storedSessionId !== storedSessionId) { + runtimeIdByStoredSessionIdRef.current.delete(existing.storedSessionId) + + // A rotation event needs a real next id — a null/cleared stored id + // is a detach, not a rotation the route-follow effect should chase. + if (storedSessionId && sessionId === $activeSessionId.get()) { + setActiveSessionStoredIdRotation({ + nextStoredSessionId: storedSessionId, + previousStoredSessionId: existing.storedSessionId, + runtimeSessionId: sessionId + }) + } } - } - if (storedSessionId) { - runtimeIdByStoredSessionIdRef.current.set(storedSessionId, sessionId) + if (storedSessionId) { + runtimeIdByStoredSessionIdRef.current.set(storedSessionId, sessionId) + } + + sessionStateCache.set(sessionId, updated) } + + return sessionStateCache.get(sessionId)! } - return sessionStateByRuntimeIdRef.current.get(sessionId)! - } + const created = createClientSessionState(storedSessionId ?? null) - const created = createClientSessionState(storedSessionId ?? null) - sessionStateByRuntimeIdRef.current.set(sessionId, created) + if (storedSessionId) { + runtimeIdByStoredSessionIdRef.current.set(storedSessionId, sessionId) + } - if (storedSessionId) { - runtimeIdByStoredSessionIdRef.current.set(storedSessionId, sessionId) - } + sessionStateCache.set(sessionId, created) - return created - }, []) + return created + }, + [sessionStateCache] + ) const resetViewSync = useCallback(() => { // Drop any RAF-pending transcript stage so a backgrounded turn cannot @@ -299,7 +332,7 @@ export function useSessionStateCache({ return previous } - sessionStateByRuntimeIdRef.current.set(sessionId, next) + sessionStateCache.set(sessionId, next) // Crash-survivable turn progress: journal the running turn's visible // tail (throttled localStorage write; cleared the moment the turn // settles) so a renderer/app death mid-turn can be recovered on resume. @@ -308,24 +341,32 @@ export function useSessionStateCache({ // (watchdog, settle grace, unread marker, compression id rotation) inside // publishSessionState — no manual transition call needed. publishSessionState(sessionId, next) + sessionStateCache.prune() syncSessionStateToView(sessionId, next) return next }, - [ensureSessionState, syncSessionStateToView] + [ensureSessionState, sessionStateCache, syncSessionStateToView] ) - const getRuntimeIdForStoredSession = useCallback((storedSessionId: string): string | null => { - const runtimeId = runtimeIdByStoredSessionIdRef.current.get(storedSessionId) + useEffect(() => { + sessionStateCache.prune() + }, [activeSessionId, selectedStoredSessionId, sessionStateCache, sessionTiles]) - if (!runtimeId) { - return null - } + const getRuntimeIdForStoredSession = useCallback( + (storedSessionId: string): string | null => { + const runtimeId = runtimeIdByStoredSessionIdRef.current.get(storedSessionId) - const runtimeState = sessionStateByRuntimeIdRef.current.get(runtimeId) + if (!runtimeId) { + return null + } - return runtimeState?.storedSessionId === storedSessionId ? runtimeId : null - }, []) + const runtimeState = sessionStateCache.get(runtimeId) + + return runtimeState?.storedSessionId === storedSessionId ? runtimeId : null + }, + [sessionStateCache] + ) return { activeSessionIdRef, @@ -334,7 +375,7 @@ export function useSessionStateCache({ resetViewSync, runtimeIdByStoredSessionIdRef, selectedStoredSessionIdRef, - sessionStateByRuntimeIdRef, + sessionStateByRuntimeIdRef: sessionStateByRuntimeIdRef as MutableRefObject>, syncSessionStateToView, updateSessionState } diff --git a/apps/desktop/src/app/session/session-state-cache.test.ts b/apps/desktop/src/app/session/session-state-cache.test.ts new file mode 100644 index 0000000000000..a52ce3238eff4 --- /dev/null +++ b/apps/desktop/src/app/session/session-state-cache.test.ts @@ -0,0 +1,133 @@ +import { beforeEach, describe, expect, it } from 'vitest' + +import type { ClientSessionState } from '@/app/types' +import type { ChatMessage } from '@/lib/chat-messages' +import { createClientSessionState } from '@/lib/chat-runtime' +import { $sessionStates, $sessionTiles, releaseSessionTranscript } from '@/store/session-states' + +import { SessionStateCache } from './session-state-cache' + +function transcript(id: string, text = id): ChatMessage[] { + return [ + { id: `${id}-user`, role: 'user', parts: [{ type: 'text', text }] }, + { id: `${id}-assistant`, role: 'assistant', parts: [{ type: 'text', text: `reply ${text}` }] } + ] +} + +function settled(storedSessionId: string, text = storedSessionId): ClientSessionState { + return { ...createClientSessionState(storedSessionId), messages: transcript(storedSessionId, text) } +} + +describe('SessionStateCache', () => { + beforeEach(() => { + $sessionStates.set({}) + $sessionTiles.set([]) + }) + + it('bounds warm settled transcripts by LRU count and cleans ownership atomically', () => { + const owners = new Map() + const evicted: string[] = [] + + const cache = new SessionStateCache( + { + isReferenced: () => false, + onEvict: (runtimeId, state) => { + if (state.storedSessionId && owners.get(state.storedSessionId) === runtimeId) { + owners.delete(state.storedSessionId) + } + + evicted.push(runtimeId) + } + }, + { maxBytes: Number.POSITIVE_INFINITY, maxCount: 2 } + ) + + for (const id of ['a', 'b', 'c']) { + owners.set(`stored-${id}`, `runtime-${id}`) + cache.set(`runtime-${id}`, settled(`stored-${id}`)) + } + + // A read makes A warmer than B, so B is the oldest when pruning. + cache.get('runtime-a') + cache.prune() + + expect([...cache.keys()].sort()).toEqual(['runtime-a', 'runtime-c']) + expect(evicted).toEqual(['runtime-b']) + expect(owners.has('stored-b')).toBe(false) + + // A recycled reverse mapping is not owned by the evicted runtime and must + // survive cleanup. + owners.set('stored-a', 'runtime-new-owner') + cache.set('runtime-d', settled('stored-d')) + owners.set('stored-d', 'runtime-d') + cache.prune() + expect(owners.get('stored-a')).toBe('runtime-new-owner') + }) + + it('uses transcript bytes as well as count', () => { + const evicted: string[] = [] + + const cache = new SessionStateCache( + { isReferenced: () => false, onEvict: runtimeId => evicted.push(runtimeId) }, + { maxBytes: 600, maxCount: 10 } + ) + + cache.set('small', settled('small', 'x')) + cache.set('large', settled('large', 'x'.repeat(500))) + cache.prune() + + expect(evicted).toEqual(['small', 'large']) + expect(cache.size).toBe(0) + }) + + it.each([ + ['active', (state: ClientSessionState) => state, true], + ['tiled', (state: ClientSessionState) => state, true], + ['busy', (state: ClientSessionState) => ({ ...state, busy: true }), false], + ['awaiting', (state: ClientSessionState) => ({ ...state, awaitingResponse: true }), false], + ['needs input', (state: ClientSessionState) => ({ ...state, needsInput: true }), false] + ])('never evicts %s transcripts', (_label, decorate, referenced) => { + const protectedState = decorate(settled('protected')) + + const cache = new SessionStateCache( + { + isReferenced: runtimeId => referenced && runtimeId === 'protected', + onEvict: () => undefined + }, + { maxBytes: 0, maxCount: 0 } + ) + + cache.set('protected', protectedState) + cache.prune() + + expect(cache.get('protected')).toBe(protectedState) + }) + + it('keeps unsaved drafts and pending messages out of the eviction pool', () => { + const draft = { ...createClientSessionState(null), messages: transcript('draft') } + const pending = settled('pending') + pending.messages = [{ id: 'pending-assistant', role: 'assistant', parts: [], pending: true }] + + const cache = new SessionStateCache( + { isReferenced: () => false, onEvict: () => undefined }, + { maxBytes: 0, maxCount: 0 } + ) + + cache.set('draft', draft) + cache.set('pending', pending) + cache.prune() + + expect(cache.has('draft')).toBe(true) + expect(cache.has('pending')).toBe(true) + }) + + it('retains lightweight status while releasing an evicted transcript', () => { + const state = { ...settled('stored'), needsInput: false } + $sessionStates.set({ runtime: state }) + + releaseSessionTranscript('runtime') + + expect($sessionStates.get().runtime).toMatchObject({ storedSessionId: 'stored', busy: false, needsInput: false }) + expect($sessionStates.get().runtime.messages).toEqual([]) + }) +}) diff --git a/apps/desktop/src/app/session/session-state-cache.ts b/apps/desktop/src/app/session/session-state-cache.ts new file mode 100644 index 0000000000000..9c4197e6d28c9 --- /dev/null +++ b/apps/desktop/src/app/session/session-state-cache.ts @@ -0,0 +1,135 @@ +import type { ClientSessionState } from '../types' + +export const DEFAULT_WARM_SESSION_TRANSCRIPT_COUNT = 24 +export const DEFAULT_WARM_SESSION_TRANSCRIPT_BYTES = 32 * 1024 * 1024 + +interface SessionStateCacheLimits { + maxBytes?: number + maxCount?: number +} + +interface SessionStateCacheCallbacks { + isReferenced: (runtimeId: string, state: ClientSessionState) => boolean + onEvict: (runtimeId: string, state: ClientSessionState) => void +} + +function transcriptBytes(state: ClientSessionState): number { + if (state.messages.length === 0) { + return 0 + } + + // JS strings occupy two bytes per UTF-16 code unit. JSON also accounts for + // ids, part tags, tool payloads, attachment metadata, and error text without + // retaining a second serialized copy in the cache. + return JSON.stringify(state.messages).length * 2 +} + +function hasDraftOrInFlightMessage(state: ClientSessionState): boolean { + return state.messages.some(message => message.pending === true) +} + +/** + * Runtime state map whose settled, unreferenced transcripts form a weighted + * LRU. Live/visible states and unsaved drafts are outside both limits. + */ +export class SessionStateCache extends Map { + readonly #callbacks: SessionStateCacheCallbacks + readonly #maxBytes: number + readonly #maxCount: number + readonly #recency = new Map() + #clock = 0 + + constructor(callbacks: SessionStateCacheCallbacks, limits: SessionStateCacheLimits = {}) { + super() + this.#callbacks = callbacks + this.#maxBytes = limits.maxBytes ?? DEFAULT_WARM_SESSION_TRANSCRIPT_BYTES + this.#maxCount = limits.maxCount ?? DEFAULT_WARM_SESSION_TRANSCRIPT_COUNT + } + + override get(runtimeId: string): ClientSessionState | undefined { + const state = super.get(runtimeId) + + if (state) { + this.#touch(runtimeId) + } + + return state + } + + override set(runtimeId: string, state: ClientSessionState): this { + super.set(runtimeId, state) + this.#touch(runtimeId) + + return this + } + + override delete(runtimeId: string): boolean { + this.#recency.delete(runtimeId) + + return super.delete(runtimeId) + } + + override clear(): void { + this.#recency.clear() + super.clear() + } + + prune(): void { + const candidates: Array<{ bytes: number; runtimeId: string; state: ClientSessionState; touched: number }> = [] + let bytes = 0 + + for (const [runtimeId, state] of this.entries()) { + if (!this.#isWarmSettled(runtimeId, state)) { + continue + } + + const weight = transcriptBytes(state) + candidates.push({ bytes: weight, runtimeId, state, touched: this.#recency.get(runtimeId) ?? 0 }) + bytes += weight + } + + let count = candidates.length + + if (count <= this.#maxCount && bytes <= this.#maxBytes) { + return + } + + candidates.sort((a, b) => a.touched - b.touched) + + for (const candidate of candidates) { + if (count <= this.#maxCount && bytes <= this.#maxBytes) { + break + } + + // References and activity can change between insertion and pruning. + const current = super.get(candidate.runtimeId) + + if (current !== candidate.state || !this.#isWarmSettled(candidate.runtimeId, current)) { + continue + } + + super.delete(candidate.runtimeId) + this.#recency.delete(candidate.runtimeId) + count -= 1 + bytes -= candidate.bytes + this.#callbacks.onEvict(candidate.runtimeId, candidate.state) + } + } + + #isWarmSettled(runtimeId: string, state: ClientSessionState): boolean { + return ( + Boolean(state.storedSessionId) && + state.messages.length > 0 && + !state.busy && + !state.awaitingResponse && + !state.needsInput && + !hasDraftOrInFlightMessage(state) && + !this.#callbacks.isReferenced(runtimeId, state) + ) + } + + #touch(runtimeId: string): void { + this.#clock += 1 + this.#recency.set(runtimeId, this.#clock) + } +} diff --git a/apps/desktop/src/components/assistant-ui/thread/list.test.ts b/apps/desktop/src/components/assistant-ui/thread/list.test.ts index f2a3d66b4b5ee..f6d6a77e02731 100644 --- a/apps/desktop/src/components/assistant-ui/thread/list.test.ts +++ b/apps/desktop/src/components/assistant-ui/thread/list.test.ts @@ -3,11 +3,13 @@ import { describe, expect, it } from 'vitest' import { buildGroups, firstVisibleGroupIndex, + HIDDEN_TRANSCRIPT_RENDER_BUDGET, LIVE_TAIL_MIN_GROUPS, LIVE_TAIL_PARTS, liveTailStart, type MessageGroup, - resolveThreadScrollTarget + resolveThreadScrollTarget, + transcriptPaneBudget } from './list' // Signature rows are `${index}:${id}:${role}:${weight}` (see the useAuiState @@ -15,6 +17,14 @@ import { const signature = (rows: [string, string, number][]) => rows.map(([id, role, weight], index) => `${index}:${id}:${role}:${weight}`).join('\n') +describe('transcriptPaneBudget', () => { + it('uses a fixed live-tail budget while hidden instead of charging every mounted transcript', () => { + expect(transcriptPaneBudget(1, true)).toBe(HIDDEN_TRANSCRIPT_RENDER_BUDGET) + expect(transcriptPaneBudget(4, true)).toBe(HIDDEN_TRANSCRIPT_RENDER_BUDGET) + expect(transcriptPaneBudget(1, false)).toBeGreaterThan(HIDDEN_TRANSCRIPT_RENDER_BUDGET) + }) +}) + describe('buildGroups', () => { it('returns no groups for an empty signature', () => { expect(buildGroups('')).toEqual([]) diff --git a/apps/desktop/src/components/assistant-ui/thread/list.tsx b/apps/desktop/src/components/assistant-ui/thread/list.tsx index afede603857cf..b0d2c1cf3a1e6 100644 --- a/apps/desktop/src/components/assistant-ui/thread/list.tsx +++ b/apps/desktop/src/components/assistant-ui/thread/list.tsx @@ -17,6 +17,7 @@ import { } from 'react' import { type GetTargetScrollTop, useStickToBottom } from 'use-stick-to-bottom' +import { usePaneLifecycle } from '@/components/pane-shell/pane-visibility' import { useI18n } from '@/i18n' import { messagePaintWeight } from '@/lib/render-weight' import { cn } from '@/lib/utils' @@ -96,6 +97,15 @@ const MIN_VISIBLE_GROUPS = 8 // interruptibly, so the only thing a smaller budget changes is how much work // blocks the click-to-paint path. const FIRST_PAINT_BUDGET = 20 +// A hot-hidden transcript is retained for instant tab return, but keeping its +// full scrollback mounted defeats the bounded pane cache. Preserve only the +// live tail while hidden; revealing it resumes stepped backfill. +export const HIDDEN_TRANSCRIPT_RENDER_BUDGET = 40 + +export const transcriptPaneBudget = (mountedPanes: number, hidden: boolean): number => + hidden + ? HIDDEN_TRANSCRIPT_RENDER_BUDGET + : Math.max(Math.ceil(RENDER_BUDGET / Math.max(1, mountedPanes)), RENDER_BUDGET / 4) // Units the backfill adds per committed step (see the backfill effect). ~8-15 // ordinary turns or 1-2 tool-heavy ones per frame — big enough to fill a page // in ~10 frames, small enough that no single commit approaches a frame budget. @@ -354,8 +364,10 @@ const ThreadMessageListInner: FC = ({ }, []) const mountedPanes = useStore($mountedTranscriptPanes) - // This pane's share of the render budget — see $mountedTranscriptPanes. - const paneBudget = Math.max(Math.ceil(RENDER_BUDGET / Math.max(1, mountedPanes)), RENDER_BUDGET / 4) + const paneLifecycle = usePaneLifecycle() + // Hidden panes retain only a live-tail budget. Visible panes share the normal + // screen budget; a reveal backfills older rows in bounded transition steps. + const paneBudget = transcriptPaneBudget(mountedPanes, paneLifecycle === 'hot-hidden') const [renderBudget, setRenderBudget] = useState(FIRST_PAINT_BUDGET) @@ -379,6 +391,10 @@ const ThreadMessageListInner: FC = ({ setBudgetSessionKey(sessionKey) setHadGroups(hasGroups) setRenderBudget(FIRST_PAINT_BUDGET) + } else if (renderBudget > paneBudget) { + // Apply the hidden budget during render so React never first commits the + // stale full transcript after this pane moves to the background. + setRenderBudget(paneBudget) } else if (hadGroups !== hasGroups) { setHadGroups(hasGroups) diff --git a/apps/desktop/src/components/pane-shell/pane-lifecycle.test.ts b/apps/desktop/src/components/pane-shell/pane-lifecycle.test.ts new file mode 100644 index 0000000000000..f8e2f5838c661 --- /dev/null +++ b/apps/desktop/src/components/pane-shell/pane-lifecycle.test.ts @@ -0,0 +1,72 @@ +import { describe, expect, it } from 'vitest' + +import { emptyPaneLifecycleState, reconcilePaneLifecycle } from './pane-lifecycle' + +const visit = (state: ReturnType, activeId: string, paneIds: string[]) => + reconcilePaneLifecycle(state, { activeId, paneIds }) + +describe('per-zone pane lifecycle', () => { + it('keeps a small recent hidden set and parks older panes', () => { + let state = emptyPaneLifecycleState() + + state = visit(state, 'a', ['a', 'b', 'c', 'd']) + state = visit(state, 'b', ['a', 'b', 'c', 'd']) + state = visit(state, 'c', ['a', 'b', 'c', 'd']) + state = visit(state, 'd', ['a', 'b', 'c', 'd']) + + expect(state.entries).toMatchObject({ + a: { lifecycle: 'parked' }, + b: { lifecycle: 'hot-hidden' }, + c: { lifecycle: 'hot-hidden' }, + d: { lifecycle: 'visible' } + }) + }) + + it('tracks recency independently for each zone state', () => { + const zoneA = visit(visit(emptyPaneLifecycleState(), 'a', ['a', 'b']), 'b', ['a', 'b']) + const zoneB = visit(emptyPaneLifecycleState(), 'x', ['x', 'y']) + + expect(zoneA.entries.a.lifecycle).toBe('hot-hidden') + expect(zoneA.entries.b.lifecycle).toBe('visible') + expect(zoneB.entries.x.lifecycle).toBe('visible') + expect(zoneB.entries.y).toBeUndefined() + }) + + it('keeps a hidden terminal alive outside the normal cap', () => { + let state = emptyPaneLifecycleState() + const paneIds = ['terminal', 'a', 'b', 'c'] + + const reconcile = (activeId: string) => { + state = reconcilePaneLifecycle(state, { + activeId, + hotHiddenCap: 1, + keepAlive: id => id === 'terminal', + paneIds + }) + } + + reconcile('terminal') + reconcile('a') + reconcile('b') + reconcile('c') + + expect(state.entries.terminal.lifecycle).toBe('hot-hidden') + expect(state.entries.b.lifecycle).toBe('hot-hidden') + expect(state.entries.a.lifecycle).toBe('parked') + }) + + it('forgets panes that leave a zone and remounts a parked pane when selected', () => { + let state = visit(emptyPaneLifecycleState(), 'a', ['a', 'b', 'c', 'd']) + + for (const active of ['b', 'c', 'd']) { + state = visit(state, active, ['a', 'b', 'c', 'd']) + } + + expect(state.entries.a.lifecycle).toBe('parked') + + state = visit(state, 'a', ['a', 'b', 'c']) + + expect(state.entries.a.lifecycle).toBe('visible') + expect(state.entries.d).toBeUndefined() + }) +}) diff --git a/apps/desktop/src/components/pane-shell/pane-lifecycle.ts b/apps/desktop/src/components/pane-shell/pane-lifecycle.ts new file mode 100644 index 0000000000000..edac97d3dc5f5 --- /dev/null +++ b/apps/desktop/src/components/pane-shell/pane-lifecycle.ts @@ -0,0 +1,71 @@ +export type PaneLifecycle = 'visible' | 'hot-hidden' | 'parked' + +export const DEFAULT_HOT_HIDDEN_PANE_CAP = 2 + +interface PaneLifecycleEntry { + lifecycle: PaneLifecycle + lastVisible: number +} + +export interface PaneLifecycleState { + clock: number + entries: Record +} + +export const emptyPaneLifecycleState = (): PaneLifecycleState => ({ clock: 0, entries: {} }) + +interface ReconcilePaneLifecycleOptions { + activeId: string + hotHiddenCap?: number + keepAlive?: (id: string) => boolean + paneIds: readonly string[] +} + +/** + * Reconcile one zone's mounted pane cache. + * + * The foreground pane is visible, the most recently visible inactive panes stay + * hot up to a small cap, and the rest park (unmount). Explicit keep-alive panes + * such as the terminal remain hot outside that cap so hiding UI never kills the + * stateful resource they host. + */ +export function reconcilePaneLifecycle( + previous: PaneLifecycleState, + { activeId, hotHiddenCap = DEFAULT_HOT_HIDDEN_PANE_CAP, keepAlive = () => false, paneIds }: ReconcilePaneLifecycleOptions +): PaneLifecycleState { + const present = new Set(paneIds) + const entries: Record = {} + let clock = previous.clock + + for (const id of paneIds) { + const prior = previous.entries[id] + + if (prior) { + entries[id] = { ...prior, lifecycle: 'parked' } + } + } + + if (present.has(activeId)) { + const prior = previous.entries[activeId] + + if (!prior || prior.lifecycle !== 'visible') { + clock += 1 + } + + entries[activeId] = { lifecycle: 'visible', lastVisible: clock } + } + + const inactive = paneIds + .filter(id => id !== activeId && entries[id]) + .sort((a, b) => entries[b].lastVisible - entries[a].lastVisible) + + for (const id of inactive.filter(keepAlive)) { + entries[id] = { ...entries[id], lifecycle: 'hot-hidden' } + } + + for (const id of inactive.filter(id => !keepAlive(id)).slice(0, Math.max(0, hotHiddenCap))) { + entries[id] = { ...entries[id], lifecycle: 'hot-hidden' } + } + + return { clock, entries } +} diff --git a/apps/desktop/src/components/pane-shell/pane-visibility.ts b/apps/desktop/src/components/pane-shell/pane-visibility.ts index 0a7169e68bece..52276270b2c18 100644 --- a/apps/desktop/src/components/pane-shell/pane-visibility.ts +++ b/apps/desktop/src/components/pane-shell/pane-visibility.ts @@ -12,6 +12,8 @@ import { createContext, useContext } from 'react' +import type { PaneLifecycle } from './pane-lifecycle' + /** Marks a mounted-but-hidden pane layer (an inactive tab in a stack). */ export const PANE_HIDDEN_ATTR = 'data-pane-hidden' @@ -28,6 +30,12 @@ export const PaneVisibleContext = createContext(true) export const usePaneVisible = (): boolean => useContext(PaneVisibleContext) +/** Lifecycle face for expensive descendants. Outside a pane tree the surface is + * visible; hot-hidden panes stay mounted but can lower their render budget. */ +export const PaneLifecycleContext = createContext('visible') + +export const usePaneLifecycle = (): PaneLifecycle => useContext(PaneLifecycleContext) + /** Fallback group key for a surface rendered outside the layout tree (secondary * windows, plain routes) — one bucket, since there are no sibling zones there * to tell apart. */ diff --git a/apps/desktop/src/components/pane-shell/tree/renderer/track-model.ts b/apps/desktop/src/components/pane-shell/tree/renderer/track-model.ts index f70b8222ec755..f1d27741e25a6 100644 --- a/apps/desktop/src/components/pane-shell/tree/renderer/track-model.ts +++ b/apps/desktop/src/components/pane-shell/tree/renderer/track-model.ts @@ -59,6 +59,10 @@ interface PaneChrome extends PaneSizing { /** Spawn corner for `placement: 'floating'` (default `'top-right'`). The * pane also TRACKS that corner's edges when the window resizes. */ anchor?: FloatingAnchor + /** Keep this pane mounted when hidden even after the zone's bounded hot + * cache fills. Reserved for stateful resources whose lifetime must not track + * tab visibility (for example terminal PTYs). */ + lifecycleKeepAlive?: boolean /** No Close in the tab menu — the one surface the app can't lose (the * main workspace). Session tiles share `placement: 'main'` but close. */ uncloseable?: boolean diff --git a/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.tsx b/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.tsx index 3f237d1290b28..6e7abf05ccf3e 100644 --- a/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.tsx +++ b/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.tsx @@ -10,7 +10,7 @@ */ import { useStore } from '@nanostores/react' -import { type CSSProperties, Fragment, type ReactNode, type RefObject, useEffect, useRef, useState } from 'react' +import { type CSSProperties, Fragment, type ReactNode, type RefObject, useRef, useState } from 'react' import { ActionsContextMenu, type MenuKit, renderActionItem } from '@/components/ui/actions-menu' import { Codicon } from '@/components/ui/codicon' @@ -32,7 +32,8 @@ import { cn } from '@/lib/utils' import { $layoutEditMode } from '../../edit-mode' import { useWindowControlsOverlap } from '../../geometry' -import { hiddenPaneProps, PaneGroupContext, PaneVisibleContext } from '../../pane-visibility' +import { emptyPaneLifecycleState, reconcilePaneLifecycle } from '../../pane-lifecycle' +import { hiddenPaneProps, PaneGroupContext, PaneLifecycleContext, PaneVisibleContext } from '../../pane-visibility' import type { DropPosition, GroupNode } from '../model' import { $dropHint, @@ -212,29 +213,24 @@ export function TreeGroup({ const active = paneFor(activeId) const isEmpty = node.panes.length === 0 - // KEEP-ALIVE: every pane that has been ACTIVE in this zone stays mounted — - // an inactive tab merely hides (visibility), it does not unmount. Remounting - // on every tab switch re-measured and re-scrolled the content from scratch - // (the thread visibly layout-shifted each time a session tab was revisited). - // Lazy on purpose: a pane first mounts when first activated, so a - // boot-restored tab stack doesn't resume every session up front. - const everActivePanesRef = useRef>(new Set()) - - useEffect(() => { - if (!node.minimized && !isEmpty) { - everActivePanesRef.current.add(activeId) - } - - // Prune panes that left the zone (closed / moved to another group), so a - // long-lived zone doesn't pin stale ids forever. - for (const id of everActivePanesRef.current) { - if (!node.panes.includes(id)) { - everActivePanesRef.current.delete(id) - } - } - }) + // BOUNDED KEEP-ALIVE: the active pane is visible, a small per-zone LRU stays + // hot-hidden, and older panes park (unmount). This preserves fast tab + // round-trips without letting a long-lived zone pin every transcript it has + // ever visited. Stateful resources can opt out of parking (the terminal keeps + // its PTY alive while hidden). Lazy remains deliberate: restored background + // tabs have no lifecycle entry and do not mount until first activation. + const lifecycleRef = useRef(emptyPaneLifecycleState()) + + if (!node.minimized && !isEmpty) { + lifecycleRef.current = reconcilePaneLifecycle(lifecycleRef.current, { + activeId, + keepAlive: id => Boolean(paneChrome(paneFor(id)).lifecycleKeepAlive), + paneIds: shown + }) + } - const keptPanes = shown.filter(id => id === activeId || everActivePanesRef.current.has(id)) + const paneLifecycle = lifecycleRef.current.entries + const keptPanes = shown.filter(id => paneLifecycle[id] && paneLifecycle[id].lifecycle !== 'parked') // ONE header style: the app's compact pane-header. DEFAULT is contextual — // a single pane isn't a "tab", so its header auto-hides; a stack shows its @@ -593,8 +589,8 @@ export function TreeGroup({ )} - {/* Body: the zone's pane content — every kept (ever-active) pane stays - mounted in an absolute layer; only the active one is visible. + {/* Body: the zone's pane content — the active pane and bounded hot-hidden + cache stay mounted in absolute layers; parked panes are unmounted. `visibility` (not display) keeps the hidden pane's layout box, so scroll positions and measurements survive the round-trip — which also makes a hidden layer's rect identical to the visible one's, hence the @@ -627,11 +623,13 @@ export function TreeGroup({ // Reload remounts the contribution (effects re-run, state // resets) while the layer — and every other tab — stays. - - - - - + + + + + + + ) : ( isActive && ( diff --git a/apps/desktop/src/store/session-states-eviction.test.ts b/apps/desktop/src/store/session-states-eviction.test.ts index 8958df05ee655..b82aa79756028 100644 --- a/apps/desktop/src/store/session-states-eviction.test.ts +++ b/apps/desktop/src/store/session-states-eviction.test.ts @@ -8,8 +8,8 @@ import { $sessionStates, $sessionTiles, closeSessionTile, publishSessionState } * The closed-tile leak: gateway events keep publishing for sessions whose * surface is gone, and every parked transcript taxes every later publish (map * spread + the status projections run per entry per message delta). A settled - * state nothing references must leave the map; everything a surface still - * needs must stay. + * state nothing references must release its transcript; lightweight status + * stays so sidebar projections remain available. */ const state = (storedId: string, patch: Partial> = {}) => ({ @@ -28,13 +28,14 @@ beforeEach(() => { }) describe('publish-time eviction', () => { - it('evicts a settling session no surface references, keeping its unread dot', () => { + it('releases an unreferenced settled transcript while keeping status and its unread dot', () => { publishSessionState('rt-1', state('stored-1', { busy: true })) expect($sessionStates.get()['rt-1']).toBeDefined() publishSessionState('rt-1', state('stored-1', { busy: false })) - expect($sessionStates.get()['rt-1']).toBeUndefined() + expect($sessionStates.get()['rt-1']?.messages).toEqual([]) + expect($sessionStates.get()['rt-1']).toMatchObject({ storedSessionId: 'stored-1', busy: false }) // The settle transition still fired: the sidebar's unread marker landed. expect($unreadFinishedSessionIds.get()).toContain('stored-1') }) @@ -100,8 +101,9 @@ describe('closeSessionTile eviction', () => { expect($sessionStates.get()['rt-1']).toBeDefined() - // ... and its settle publish is what evicts it. + // ... and its settle publish releases only the heavy transcript. publishSessionState('rt-1', state('stored-1', { busy: false })) - expect($sessionStates.get()['rt-1']).toBeUndefined() + expect($sessionStates.get()['rt-1']?.messages).toEqual([]) + expect($sessionStates.get()['rt-1']).toMatchObject({ storedSessionId: 'stored-1', busy: false }) }) }) diff --git a/apps/desktop/src/store/session-states.test.ts b/apps/desktop/src/store/session-states.test.ts index 3bff4ba520806..0abc89bfd5f3c 100644 --- a/apps/desktop/src/store/session-states.test.ts +++ b/apps/desktop/src/store/session-states.test.ts @@ -6,16 +6,39 @@ import { $layoutTree } from '@/components/pane-shell/tree/store' import { $selectedStoredSessionId } from '@/store/session' import type { SessionTile } from '@/store/session-states' import { + $sessionStates, blankDraftTile, focusedSessionNeedsRoute, markSelectionRestore, orderTilesByTree, + releaseSessionTranscript, selectionHomesToWorkspace } from '@/store/session-states' const tile = (storedSessionId: string): SessionTile => ({ storedSessionId }) const tilePane = (id: string) => `session-tile:${id}` +describe('releaseSessionTranscript', () => { + afterEach(() => { + $sessionStates.set({}) + }) + + it('normalizes legacy state whose messages field is undefined', () => { + const legacy = { busy: false, storedSessionId: 'stored' } as ClientSessionState + $sessionStates.set({ runtime: legacy }) + + expect(() => releaseSessionTranscript('runtime')).not.toThrow() + expect($sessionStates.get().runtime).toEqual({ ...legacy, messages: [] }) + }) + + it('ignores a legacy undefined state without throwing', () => { + $sessionStates.set({ runtime: undefined } as unknown as Record) + + expect(() => releaseSessionTranscript('runtime')).not.toThrow() + expect($sessionStates.get()).toHaveProperty('runtime', undefined) + }) +}) + describe('orderTilesByTree', () => { it('no-ops (null) without a tree or below two tiles', () => { expect(orderTilesByTree(null, [tile('a'), tile('b')])).toBeNull() diff --git a/apps/desktop/src/store/session-states.ts b/apps/desktop/src/store/session-states.ts index 1064b90d7ad3d..79653baaadc31 100644 --- a/apps/desktop/src/store/session-states.ts +++ b/apps/desktop/src/store/session-states.ts @@ -228,15 +228,13 @@ function evictable(runtimeId: string, state: ClientSessionState): boolean { * is updated independently by the caller, so the visual path stays live * without the store churn. * - * A settled state nothing references is EVICTED instead of republished: - * gateway events keep flowing for sessions whose tile was closed mid-turn, - * and parking each one's full transcript here forever is the leak that made - * the app crawl after a day of tile use — every entry taxes every later - * publish (map spread + the status-set projections). Transition side effects - * still fire, so the closed session's settle keeps its unread dot. Only an - * entry already in the map is evicted — a FIRST publish always lands, because - * a resume can publish its idle state a beat before `$activeSessionId` / - * the tile's runtime binding points at it. */ + * A settled state nothing references releases its transcript instead of + * republishing it. Gateway events keep flowing for sessions whose tile was + * closed mid-turn, and parking each one's full transcript here forever is the + * leak that made the app crawl after a day of tile use. Transition side + * effects still fire, so lightweight status and the unread dot survive. A + * FIRST publish always lands in full because a resume can publish its idle + * state a beat before `$activeSessionId` / the tile binding points at it. */ export function publishSessionState(runtimeId: string, state: ClientSessionState) { const current = $sessionStates.get() const prev = current[runtimeId] ?? null @@ -247,8 +245,7 @@ export function publishSessionState(runtimeId: string, state: ClientSessionState if (prev && evictable(runtimeId, state)) { handleTransition(prev, state, runtimeId) - const { [runtimeId]: _dropped, ...rest } = current - $sessionStates.set(rest) + releaseSessionTranscript(runtimeId, state) return } @@ -257,6 +254,30 @@ export function publishSessionState(runtimeId: string, state: ClientSessionState handleTransition(prev, state, runtimeId) } +/** Keep the cheap status projection for a cold session while releasing its + * transcript. Unread completion is stored separately, so it survives too. */ +export function releaseSessionTranscript(runtimeId: string, state?: ClientSessionState) { + const current = $sessionStates.get() + + if (!(runtimeId in current)) { + return + } + + const retained = state ?? current[runtimeId] + + // Older persisted snapshots can contain an undefined state or omit the + // messages field. Treat either shape as already cold instead of throwing + // while memory pressure is being relieved. + if (!retained) { + return + } + + const lightweight = + Array.isArray(retained.messages) && retained.messages.length === 0 ? retained : { ...retained, messages: [] } + + $sessionStates.set({ ...current, [runtimeId]: lightweight }) +} + export function dropSessionState(runtimeId: string) { // Disarm the watchdog — a dropped runtime must not fire a stale clear later. // Settle-grace entries are keyed by stored id and self-expire; leave them so diff --git a/hermes_cli/web_server.py b/hermes_cli/web_server.py index 7a42a601b40b7..e64e06d30d5ad 100644 --- a/hermes_cli/web_server.py +++ b/hermes_cli/web_server.py @@ -137,6 +137,94 @@ WEB_DIST = Path(os.environ["HERMES_WEB_DIST"]) if "HERMES_WEB_DIST" in os.environ else Path(__file__).parent / "web_dist" _log = logging.getLogger(__name__) + +def _process_start_marker(pid: int) -> str: + """Return a cross-runtime marker for the current incarnation of ``pid``. + + ``ProcessLookupError`` means the process is absent. Other failures are left + distinct so callers can fail safe rather than killing a healthy backend. + """ + if sys.platform == "linux": + try: + stat_line = Path(f"/proc/{pid}/stat").read_text(encoding="utf-8") + except FileNotFoundError as exc: + raise ProcessLookupError(pid) from exc + + # The command in field 2 may contain spaces or parentheses. Splitting + # after its final ')' leaves field 3 at index zero and field 22 at 19. + fields = stat_line.rsplit(")", 1)[1].strip().split() + if len(fields) < 20 or not fields[19].isdigit(): + raise OSError(f"invalid /proc stat data for PID {pid}") + return f"linux:{fields[19]}" + + if os.name == "nt": + import ctypes + from ctypes import wintypes + + process_query_limited_information = 0x1000 + kernel32 = ctypes.WinDLL("kernel32", use_last_error=True) + kernel32.OpenProcess.argtypes = [wintypes.DWORD, wintypes.BOOL, wintypes.DWORD] + kernel32.OpenProcess.restype = wintypes.HANDLE + kernel32.GetProcessTimes.argtypes = [ + wintypes.HANDLE, + ctypes.POINTER(wintypes.FILETIME), + ctypes.POINTER(wintypes.FILETIME), + ctypes.POINTER(wintypes.FILETIME), + ctypes.POINTER(wintypes.FILETIME), + ] + kernel32.GetProcessTimes.restype = wintypes.BOOL + kernel32.CloseHandle.argtypes = [wintypes.HANDLE] + kernel32.CloseHandle.restype = wintypes.BOOL + handle = kernel32.OpenProcess(process_query_limited_information, False, pid) + if not handle: + error = ctypes.get_last_error() + if error in (87, 1168): # invalid parameter / not found + raise ProcessLookupError(pid) + raise OSError(error, f"OpenProcess failed for PID {pid}") + + creation = wintypes.FILETIME() + exit_time = wintypes.FILETIME() + kernel = wintypes.FILETIME() + user = wintypes.FILETIME() + try: + if not kernel32.GetProcessTimes( + handle, + ctypes.byref(creation), + ctypes.byref(exit_time), + ctypes.byref(kernel), + ctypes.byref(user), + ): + error = ctypes.get_last_error() + raise OSError(error, f"GetProcessTimes failed for PID {pid}") + finally: + kernel32.CloseHandle(handle) + + filetime = (creation.dwHighDateTime << 32) | creation.dwLowDateTime + return f"win:{filetime + 504911232000000000}" + + result = subprocess.run( + ["ps", "-p", str(pid), "-o", "lstart="], + capture_output=True, + text=True, + check=False, + ) + marker = result.stdout.strip() + if result.returncode == 0 and marker: + return f"ps:{marker}" + if result.returncode == 1 and not marker: + raise ProcessLookupError(pid) + raise OSError(f"ps could not inspect PID {pid}: {result.stderr.strip()}") + + +def _valid_parent_start_marker(marker: str) -> bool: + prefix, separator, value = marker.partition(":") + if not separator or not value or value != value.strip(): + return False + if prefix in ("linux", "win"): + return value.isdigit() + return prefix == "ps" + + # --------------------------------------------------------------------------- # Per-channel subscriber registry used by /api/pub (PTY-side gateway → dashboard) # and /api/events (dashboard → browser sidebar). Keyed by an opaque channel id @@ -17901,53 +17989,80 @@ def _open(): threading.Thread(target=_open, daemon=True).start() -def _is_serve_orphaned(desktop_pid: int, pid_exists=None) -> bool: - """True when the Desktop process that owns this serve backend is gone. +def _is_serve_orphaned( + desktop_pid: int, + expected_start_marker: Optional[str] = None, + *, + pid_exists=None, + process_start_marker=None, +) -> bool: + """True when the exact Desktop process that owns this backend is gone. ``HERMES_PARENT_PID`` is the Electron Desktop PID, not necessarily this Python process's immediate PPID. On Windows the venv ``hermes.exe`` launcher introduces one or more shim processes, so comparing ``os.getppid()`` to the Electron PID incorrectly treats a healthy backend as orphaned and exits 0. - Probe the recorded Desktop PID directly instead. - Any liveness-probe failure is fail-safe: keep serving rather than killing a - backend whose owner could not be conclusively shown to be dead. + New Desktop versions also provide the owner's process-start marker. This + prevents a recycled PID from keeping an orphan alive. Older versions remain + compatible through the PID-only probe. Any inconclusive probe failure is + fail-safe: keep serving rather than killing a backend whose owner could not + be conclusively shown to be dead. """ try: + if expected_start_marker is not None: + probe = process_start_marker or _process_start_marker + return probe(int(desktop_pid)) != expected_start_marker + if pid_exists is None: from gateway.status import _pid_exists pid_exists = _pid_exists return not bool(pid_exists(int(desktop_pid))) + except ProcessLookupError: + return True except Exception: return False def _start_parent_death_watchdog() -> None: - """Exit when the desktop parent that spawned this backend dies. + """Exit when the exact desktop parent that spawned this backend dies. - The desktop passes its own PID via HERMES_PARENT_PID. When that process - vanishes (crash, SIGKILL, update handoff exiting before it reaps us) this - orphaned backend would otherwise keep serving forever and leak its MCP - child subtree. os._exit propagates to the MCP watchdogs parented here. - - No-op for standalone `hermes serve` (env unset). Poll interval tunable via - HERMES_SERVE_WATCHDOG_POLL_S. + The desktop passes its PID and, in newer versions, its process-start marker + plus a per-spawn nonce. The marker distinguishes a live owner from PID reuse; + the nonce makes partial/mixed-version identity plumbing fail safe. Legacy + Desktop versions that provide only ``HERMES_PARENT_PID`` retain PID-only + tracking. """ - raw = os.environ.get("HERMES_PARENT_PID") - if not raw: - return + raw_pid = os.environ.get("HERMES_PARENT_PID") + start_marker = os.environ.get("HERMES_PARENT_START_MARKER") + nonce = os.environ.get("HERMES_PARENT_NONCE") + try: - desktop_pid = int(raw) + desktop_pid = int(raw_pid or "") except (TypeError, ValueError): return + if desktop_pid <= 0: + return + + has_marker = start_marker is not None + has_nonce = nonce is not None + if has_marker != has_nonce: + return + if has_marker and ( + not _valid_parent_start_marker(start_marker or "") + or not nonce + or nonce != nonce.strip() + ): + return + try: poll = max(0.5, float(os.environ.get("HERMES_SERVE_WATCHDOG_POLL_S", "2.0"))) except (TypeError, ValueError): poll = 2.0 def _loop() -> None: - while not _is_serve_orphaned(desktop_pid): + while not _is_serve_orphaned(desktop_pid, start_marker): time.sleep(poll) os._exit(0) From 56121e528d88be1d7414e450c27aff71d7f9c000 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 18:46:07 -0700 Subject: [PATCH 014/376] chore: map contributor email (audit_pr_attribution) --- contributors/emails/aleks.clark@gmail.com | 1 + 1 file changed, 1 insertion(+) create mode 100644 contributors/emails/aleks.clark@gmail.com diff --git a/contributors/emails/aleks.clark@gmail.com b/contributors/emails/aleks.clark@gmail.com new file mode 100644 index 0000000000000..b75d577a857c7 --- /dev/null +++ b/contributors/emails/aleks.clark@gmail.com @@ -0,0 +1 @@ +aleksclark From cf63b794678576bbf23ce79376af7571051b8701 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E6=9D=8E=E7=81=B5=E8=88=AA?= <602028@ky-tech.com.cn> Date: Wed, 12 Aug 2026 12:19:28 +0800 Subject: [PATCH 015/376] fix(desktop): drop pending stream rows whose reply the transcript already carries MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A still-pending assistant stream row (id `assistant-stream-*`) whose reply the authoritative transcript already committed used to fall through to `preserved.push` when ordinal pairing missed it — the commit shifted the row's ordinal under compaction/history rewrites, so `nextByRoleOrdinal` returned nothing and the local copy was appended to the tail, rendering the same answer twice (reported as A B C D E C D tail duplication). The #70209 guard only covers SETTLED local rows (`pending !== true`); pending rows were unprotected. Match pending rows against SETTLED authoritative rows before appending: - identical answer text -> authoritative already carries it - authoritative extends local text -> authoritative is the settled final version of the still-streaming local copy - local extends authoritative text -> replace the committed row with the richer local body instead of appending Live projection shells (still-pending candidates) never match, so the traces-only local row keeps replacing the empty shell. --- .../hooks/use-session-actions/utils.test.ts | 58 +++++++++++++++++++ .../hooks/use-session-actions/utils.ts | 58 +++++++++++++++++++ 2 files changed, 116 insertions(+) diff --git a/apps/desktop/src/app/session/hooks/use-session-actions/utils.test.ts b/apps/desktop/src/app/session/hooks/use-session-actions/utils.test.ts index e7a21b17cd5ba..4a18397200e63 100644 --- a/apps/desktop/src/app/session/hooks/use-session-actions/utils.test.ts +++ b/apps/desktop/src/app/session/hooks/use-session-actions/utils.test.ts @@ -1086,6 +1086,64 @@ describe('preserveLocalPendingTurnMessages', () => { expect(chatMessageText(preserved[1])).toBe('first answer') expect(preserved.filter(message => message.role === 'assistant')).toHaveLength(2) }) + + // A still-PENDING stream row whose committed twin the authoritative history + // already carries (ordinal shifted under compaction) used to fall through to + // `preserved.push` and render the same answer twice — the reported tail + // duplication (A B C D E C D). The #70209 guard only covers settled local + // rows (`pending !== true`); these cover the pending ones. + it('does not re-append a pending stream row the authoritative history already carries', () => { + const previous = [ + msg('1-user', 'user', '查金价'), + msg('2-a', 'assistant', 'X'), + streamingMsg('assistant-stream-live', '面板内容') + ] + + const next = [msg('1-user', 'user', '查金价'), msg('9-assistant', 'assistant', '面板内容')] + + expect(preserveLocalPendingTurnMessages(next, previous)).toBe(next) + }) + + it('drops a pending stream row whose text the committed authoritative reply extends', () => { + const previous = [ + msg('1-user', 'user', '查金价'), + msg('2-a', 'assistant', 'X'), + streamingMsg('assistant-stream-live', '面板') + ] + + const next = [msg('1-user', 'user', '查金价'), msg('9-assistant', 'assistant', '面板内容完整版')] + + expect(preserveLocalPendingTurnMessages(next, previous)).toBe(next) + }) + + it('replaces the committed row with a further-along pending copy instead of appending', () => { + const previous = [ + msg('1-user', 'user', '查金价'), + msg('2-a', 'assistant', 'X'), + streamingMsg('assistant-stream-live', '面板内容完整版') + ] + + const next = [msg('1-user', 'user', '查金价'), msg('9-assistant', 'assistant', '面板')] + + const preserved = preserveLocalPendingTurnMessages(next, previous) + + expect(preserved.map(message => message.id)).toEqual(['1-user', '9-assistant']) + expect(chatMessageText(preserved[1])).toBe('面板内容完整版') + }) + + // The authoritative history genuinely does not have this reply yet — the + // pending row is the only copy and must survive (same contract as the + // settled-row variant above). + it('still keeps a pending stream row when the authoritative history has no reply', () => { + const previous = [msg('1-user', 'user', '查金价'), streamingMsg('assistant-stream-live', '面板内容')] + + const next = [msg('1-user', 'user', '查金价')] + + expect(preserveLocalPendingTurnMessages(next, previous).map(message => message.id)).toEqual([ + '1-user', + 'assistant-stream-live' + ]) + }) }) describe('appendLiveSessionProjection', () => { diff --git a/apps/desktop/src/app/session/hooks/use-session-actions/utils.ts b/apps/desktop/src/app/session/hooks/use-session-actions/utils.ts index 29e8d105dd92c..6c031768acc76 100644 --- a/apps/desktop/src/app/session/hooks/use-session-actions/utils.ts +++ b/apps/desktop/src/app/session/hooks/use-session-actions/utils.ts @@ -603,6 +603,64 @@ export function preserveLocalPendingTurnMessages( } } + // Ordinal pairing missed (the committed row shifted ordinal when history + // was compacted / the authoritative list is shorter), yet the + // authoritative transcript already carries this same reply under its + // committed id. The #70209 guard above only covers SETTLED local rows + // (`pending !== true`); a still-pending stream row that slips past + // pairing falls through to `preserved.push` and renders the answer + // twice — the reported A B C D E C D tail duplication. + // + // Three-way same-turn check against SETTLED authoritative rows only + // (a live projection shell must not swallow the richer local row, see + // the traces-only replacement test): + // 1. identical answer text -> authoritative already has it + // 2. authoritative extends local text -> authoritative is the settled + // final version of the still-streaming local copy + // 3. local extends authoritative text -> local is further along; replace + // the committed row with the richer body instead of appending + if (isPendingAssistant) { + const nextText = textWithoutReferenceLines(chatMessageText(message)) + + const committedMatch = nextMessages.find( + candidate => + candidate.role === 'assistant' && + !isLiveTailRow(candidate) && + (textWithoutReferenceLines(chatMessageText(candidate)) === nextText || + isStrictAnswerTextExtension( + textWithoutReferenceLines(chatMessageText(candidate)), + nextText + )) + ) + + if (committedMatch) { + continue + } + + const committedPrefix = nextMessages.find( + candidate => + candidate.role === 'assistant' && + !isLiveTailRow(candidate) && + isStrictAnswerTextExtension( + nextText, + textWithoutReferenceLines(chatMessageText(candidate)) + ) + ) + + if (committedPrefix) { + // Keep the COMMITTED id (not the local stream id): the turn is + // already in the authoritative transcript, so the merged row must + // stay addressable as that durable row — a stream id would read as a + // live row again next reconcile and re-enter this same path. + replacements.set(committedPrefix.id, { + ...withAuthoritativeTurnState(message, committedPrefix), + id: committedPrefix.id + }) + + continue + } + } + preserved.push(message) } From 71a98f69eefffff43c201a2a631cba1e53eee398 Mon Sep 17 00:00:00 2001 From: spfcraze Date: Sun, 2 Aug 2026 00:29:49 -0400 Subject: [PATCH 016/376] fix(desktop): settle final reply onto interim even after message.start reset the boundary flag completeAssistantMessage merged a turn's final text onto its sealed interim bubble only when the session's volatile interimBoundaryPending flag was still true. A subsequent message.start (chained turn, follow-up, or mid-turn compaction) resets that flag to false; when it landed between the same turn's message.interim and message.complete, the flag-based gate fell through and the UI appended a duplicate bubble. Key the merge on the message's OWN durable interim state instead of the session flag. finalContinuesInterim already requires existing.interim plus prefix continuity, so distinct replies (which don't continue the interim) are still appended as their own bubble. Adds a regression test: interim + message.start + completing final that continues the interim must yield one bubble, not two. --- .../session/hooks/use-message-stream/index.ts | 11 ++++++++++- .../use-message-stream/interim-sealing.test.tsx | 17 +++++++++++++++++ 2 files changed, 27 insertions(+), 1 deletion(-) diff --git a/apps/desktop/src/app/session/hooks/use-message-stream/index.ts b/apps/desktop/src/app/session/hooks/use-message-stream/index.ts index 979bbc1e2f5e2..90a6fcad9a880 100644 --- a/apps/desktop/src/app/session/hooks/use-message-stream/index.ts +++ b/apps/desktop/src/app/session/hooks/use-message-stream/index.ts @@ -632,7 +632,7 @@ export function useMessageStream({ nextMessages = prev.map((message, messageIndex) => messageIndex === index ? completeMessage(message) : message ) - } else if (interimBoundaryPending && (responsePreviewed || finalContinuesInterim)) { + } else if (existing.interim && (responsePreviewed || finalContinuesInterim)) { // Settle the interim in place instead of creating a duplicate — // the DB has one row, so the live UI must agree. Previously this // was gated on `responsePreviewed` alone, so a NON-previewed @@ -642,6 +642,15 @@ export function useMessageStream({ // for ordinary tool-call turns while `responsePreviewed` still // covers the verify-on-stop continuation-budget case even when the // final text was rewritten and no longer shares a prefix. + // + // We key on the message's OWN durable `interim` state, not the + // session's volatile `interimBoundaryPending` flag: the flag is + // reset to false by a subsequent `message.start` (a chained turn, + // follow-up, or mid-turn compaction). When that reset lands + // between this turn's `message.interim` and `message.complete`, the + // flag-based gate wrongly fell through to appending a duplicate + // bubble (#74560). The message row still says `interim`, so the + // continuation merges correctly regardless of the flag. nextMessages = prev.map((message, messageIndex) => messageIndex === index ? completeMessage(message) : message ) diff --git a/apps/desktop/src/app/session/hooks/use-message-stream/interim-sealing.test.tsx b/apps/desktop/src/app/session/hooks/use-message-stream/interim-sealing.test.tsx index 7825904d3dce4..90ac198a98ccd 100644 --- a/apps/desktop/src/app/session/hooks/use-message-stream/interim-sealing.test.tsx +++ b/apps/desktop/src/app/session/hooks/use-message-stream/interim-sealing.test.tsx @@ -217,6 +217,23 @@ describe('useMessageStream interim text sealing', () => { expect(texts[0]).toBe('partial answer continued') }) + it('settles final onto interim even after message.start reset the boundary flag (#74560)', async () => { + await mountStream() + await start() + await delta('partial') + await interim('partial') + // A chained turn / follow-up re-emits message.start, which resets + // interimBoundaryPending to false BEFORE the same turn's message.complete. + // The continuation must still settle onto the interim — not append a + // duplicate bubble. Regression for #74560. + await start() + await complete('partial answer continued') + + const texts = assistantMessages() + expect(texts.filter(t => t.includes('partial'))).toHaveLength(1) + expect(texts[0]).toBe('partial answer continued') + }) + it('appends a genuinely different final as its own bubble (two real assistant segments)', async () => { await mountStream() await start() From 9cc428cff8141c88c1cd02195d69646f585aa66f Mon Sep 17 00:00:00 2001 From: spfcraze Date: Sun, 2 Aug 2026 01:47:43 -0400 Subject: [PATCH 017/376] fix(desktop): keep responsePreviewed settle gated on the boundary flag MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Sweeper review on #76583: dropping interimBoundaryPending from the previewed settle path let a previewed final arriving after a message.start reset OVERWRITE a distinct interim instead of appending (interim('old') → message.start → complete({response_previewed: true, text: 'new'}) destroyed 'old'). responsePreviewed may rewrite the final with no prefix guarantee, so it must stay flag-gated; only finalContinuesInterim (prefix-either-way continuity, which can only hold for the same message) settles flag-free. New test: distinct previewed final after a reset appends its own bubble. Production ordering cited in the test: compaction-resume events exclude message.start (gateway-event.ts), and the TUI gateway emits message.complete before goal-followup starts (tui_gateway/server.py). 74/74 use-message-stream tests pass. --- .../session/hooks/use-message-stream/index.ts | 38 ++++++++++--------- .../interim-sealing.test.tsx | 20 ++++++++++ 2 files changed, 41 insertions(+), 17 deletions(-) diff --git a/apps/desktop/src/app/session/hooks/use-message-stream/index.ts b/apps/desktop/src/app/session/hooks/use-message-stream/index.ts index 90a6fcad9a880..01cf8df499a6f 100644 --- a/apps/desktop/src/app/session/hooks/use-message-stream/index.ts +++ b/apps/desktop/src/app/session/hooks/use-message-stream/index.ts @@ -632,25 +632,29 @@ export function useMessageStream({ nextMessages = prev.map((message, messageIndex) => messageIndex === index ? completeMessage(message) : message ) - } else if (existing.interim && (responsePreviewed || finalContinuesInterim)) { + } else if ((interimBoundaryPending && responsePreviewed) || finalContinuesInterim) { // Settle the interim in place instead of creating a duplicate — - // the DB has one row, so the live UI must agree. Previously this - // was gated on `responsePreviewed` alone, so a NON-previewed - // tool-call turn whose final matched its sealed interim appended a - // second bubble (the "renders twice: partial first copy + clean - // final" bug, #63679). `finalContinuesInterim` closes that gap - // for ordinary tool-call turns while `responsePreviewed` still - // covers the verify-on-stop continuation-budget case even when the - // final text was rewritten and no longer shares a prefix. + // the DB has one row, so the live UI must agree. Two distinct + // settle paths with different boundary requirements: // - // We key on the message's OWN durable `interim` state, not the - // session's volatile `interimBoundaryPending` flag: the flag is - // reset to false by a subsequent `message.start` (a chained turn, - // follow-up, or mid-turn compaction). When that reset lands - // between this turn's `message.interim` and `message.complete`, the - // flag-based gate wrongly fell through to appending a duplicate - // bubble (#74560). The message row still says `interim`, so the - // continuation merges correctly regardless of the flag. + // • responsePreviewed covers the verify-on-stop continuation- + // budget case, where the final may be a rewrite sharing no + // prefix with the interim. Because there is no continuity + // guarantee, it must stay gated on the session's + // `interimBoundaryPending` flag: after a new `message.start` + // resets the flag, a previewed final is a DISTINCT reply and + // must append its own bubble, never overwrite the interim + // (otherwise interim('old') → message.start → + // complete({response_previewed: true, text: 'new'}) would + // silently destroy 'old'). + // + // • finalContinuesInterim (prefix-either-way continuity, same + // text or one a prefix of the other) is safe to settle + // flag-free: continuity can only hold for the SAME message, + // so a `message.start` reset landing between this turn's + // `message.interim` and `message.complete` must not force an + // append of a duplicate bubble (#74560). This also closes the + // non-previewed tool-call gap from #63679. nextMessages = prev.map((message, messageIndex) => messageIndex === index ? completeMessage(message) : message ) diff --git a/apps/desktop/src/app/session/hooks/use-message-stream/interim-sealing.test.tsx b/apps/desktop/src/app/session/hooks/use-message-stream/interim-sealing.test.tsx index 90ac198a98ccd..0aa549ec50f3d 100644 --- a/apps/desktop/src/app/session/hooks/use-message-stream/interim-sealing.test.tsx +++ b/apps/desktop/src/app/session/hooks/use-message-stream/interim-sealing.test.tsx @@ -234,6 +234,26 @@ describe('useMessageStream interim text sealing', () => { expect(texts[0]).toBe('partial answer continued') }) + it('appends a distinct previewed final after a message.start reset instead of overwriting the interim', async () => { + await mountStream() + await start() + await interim('old interim text') + // A genuinely new turn begins — message.start resets interimBoundaryPending. + // Production ordering: mid-turn compaction-resume events do NOT include + // message.start (COMPACTION_RESUME_EVENT_TYPES in gateway-event.ts), and + // the TUI gateway emits message.complete BEFORE goal-followup starts + // (tui_gateway/server.py), so a previewed final arriving after the reset + // is a DISTINCT reply, not a rewrite of the interim. It must append its + // own bubble — never overwrite the old one (sweeper review on #76583). + await start() + await completePreviewed('totally new answer') + + const texts = assistantMessages() + expect(texts).toContain('old interim text') + expect(texts).toContain('totally new answer') + expect(texts).toHaveLength(2) + }) + it('appends a genuinely different final as its own bubble (two real assistant segments)', async () => { await mountStream() await start() From f0748b451c2ed2cc8e57f7154257c14cc1891253 Mon Sep 17 00:00:00 2001 From: chelsealong Date: Mon, 10 Aug 2026 00:48:30 +0000 Subject: [PATCH 018/376] fix(desktop): stop empty REST transcript refresh from wiping a warm resume session.activate's persisted-transcript refresh reconciled unconditionally against getLatestSessionMessages, so a transient empty REST page (e.g. a backend respawn racing its own state.db read after a wake/reconnect) wiped a transcript the activate response had just restored. Guard it the same way the activate payload itself already is guarded a few lines above: an empty authoritative page never overrides a non-empty cached transcript. --- .../hooks/use-session-actions.test.tsx | 71 +++++++++++++++++++ .../hooks/use-session-actions/index.ts | 8 ++- 2 files changed, 78 insertions(+), 1 deletion(-) diff --git a/apps/desktop/src/app/session/hooks/use-session-actions.test.tsx b/apps/desktop/src/app/session/hooks/use-session-actions.test.tsx index 0decc4c5bd556..e0d2d76a9122e 100644 --- a/apps/desktop/src/app/session/hooks/use-session-actions.test.tsx +++ b/apps/desktop/src/app/session/hooks/use-session-actions.test.tsx @@ -1610,6 +1610,77 @@ describe('resumeSession warm-cache mapping integrity', () => { expect(renderedMessages).not.toContain('stale runtime answer') }) + it('keeps the activated transcript when a persisted transcript refresh returns empty rows', async () => { + // Regression: after a wake/reconnect, session.activate can legitimately + // rebind a session with a non-empty transcript while the concurrent REST + // refresh (getLatestSessionMessages) races a just-respawned backend and + // resolves with zero rows. That empty page must not be trusted over the + // transcript activate already restored. + const runtimeIdByStoredSessionIdRef: MutableRefObject> = { + current: new Map([['stored-A', 'rt-A']]) + } + + const state = clientState('stored-A') + state.messages = [ + { + id: 'cached-user', + role: 'user', + parts: [{ type: 'text', text: 'still here after wake' }] + }, + { + id: 'cached-assistant', + role: 'assistant', + parts: [{ type: 'text', text: 'still here after wake too' }] + } + ] + + const sessionStateByRuntimeIdRef: MutableRefObject> = { + current: new Map([['rt-A', state]]) + } + + const activatedMessages = [ + { content: 'still here after wake', role: 'user', timestamp: 1 }, + { content: 'still here after wake too', role: 'assistant', timestamp: 2 } + ] + + vi.mocked(getLatestSessionMessages).mockResolvedValue({ messages: [], session_id: 'stored-A' } as never) + + const requestGateway = vi.fn(async (method: string) => { + if (method === 'session.activate') { + return { + session_id: 'rt-A', + session_key: 'stored-A', + resumed: 'stored-A', + message_count: activatedMessages.length, + messages: activatedMessages, + running: false, + info: {} + } as never + } + + return {} as never + }) + + let resumedState: ClientSessionState | undefined + let resume: ((storedSessionId: string, replaceRoute?: boolean) => Promise) | null = null + + render( + (resume = ready)} + onStateUpdate={(_sessionId, next) => (resumedState = next)} + requestGateway={requestGateway} + runtimeIdByStoredSessionIdRef={runtimeIdByStoredSessionIdRef} + sessionStateByRuntimeIdRef={sessionStateByRuntimeIdRef} + /> + ) + await waitFor(() => expect(resume).not.toBeNull()) + await resume!('stored-A', true) + + const renderedMessages = JSON.stringify(resumedState?.messages) + expect(renderedMessages).toContain('still here after wake') + expect(renderedMessages).toContain('still here after wake too') + }) + it('keeps a warm runtime and optimistic turn on a transient activation timeout', async () => { const runtimeIdByStoredSessionIdRef: MutableRefObject> = { current: new Map([['stored-A', 'rt-A']]) diff --git a/apps/desktop/src/app/session/hooks/use-session-actions/index.ts b/apps/desktop/src/app/session/hooks/use-session-actions/index.ts index 23781c4942a63..3db4e8008e366 100644 --- a/apps/desktop/src/app/session/hooks/use-session-actions/index.ts +++ b/apps/desktop/src/app/session/hooks/use-session-actions/index.ts @@ -804,7 +804,13 @@ export function useSessionActions({ !activatedStoredSessionId || persisted.session_id === activatedStoredSessionId - if (persisted && persistedMatchesActivatedSession) { + // An empty REST page is not proof the transcript is empty — it's + // also what a backend respawn returns while its state.db read + // races the activate response. Reconciling against it anyway + // wipes the just-restored activate/cache transcript (the same + // wipe the `activated.messages.length || ...` guard above + // already prevents for the activate payload itself). + if (persisted && persistedMatchesActivatedSession && (persisted.messages.length || !activatedMessages.length)) { activatedMessages = reconcileAuthoritativeMessages(persisted.messages, activatedMessages) } } From 1a2b0ca8cbb7d6e70e875927e2b20987808877fb Mon Sep 17 00:00:00 2001 From: Guilherme Aguiar Date: Fri, 14 Aug 2026 14:58:55 -0300 Subject: [PATCH 019/376] fix(desktop): refresh active transcript on session changes --- .../contrib/hooks/use-background-sync.test.ts | 297 +++++++++++++++++- .../app/contrib/hooks/use-background-sync.ts | 199 +++++++++++- apps/desktop/src/app/contrib/wiring.tsx | 70 ++--- 3 files changed, 499 insertions(+), 67 deletions(-) diff --git a/apps/desktop/src/app/contrib/hooks/use-background-sync.test.ts b/apps/desktop/src/app/contrib/hooks/use-background-sync.test.ts index 22a46927af0e4..b4d5b6e6c257a 100644 --- a/apps/desktop/src/app/contrib/hooks/use-background-sync.test.ts +++ b/apps/desktop/src/app/contrib/hooks/use-background-sync.test.ts @@ -1,5 +1,9 @@ +import { act, cleanup, renderHook, waitFor } from '@testing-library/react' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { createClientSessionState } from '@/lib/chat-runtime' +import { $changeEventsAvailable, notifySessionsChanged, resetLiveSync } from '@/store/live-sync' +import { $activeSessionId, $selectedStoredSessionId, setBusy, setMessagingSessions, setSessions } from '@/store/session' import { $attentionSessionIds, $stalledSessionIds, @@ -8,19 +12,300 @@ import { SESSION_WATCHDOG_TIMEOUT_MS } from '@/store/session-states' -import { rehydrateLiveSessionStatuses } from './use-background-sync' +import { + type ActiveTranscriptRefreshDeps, + reconcileActiveTranscript, + rehydrateLiveSessionStatuses, + resolveActiveTranscriptSession, + useBackgroundSync +} from './use-background-sync' -describe('rehydrateLiveSessionStatuses', () => { +vi.mock('@/hermes', async importOriginal => ({ + ...(await importOriginal()), + getLatestSessionMessages: vi.fn() +})) + +const { getLatestSessionMessages } = await import('@/hermes') + +const ACTIVE_RUNTIME_ID = 'runtime-active' +const ACTIVE_STORED_ID = 'stored-active' + +function transcript(answer: string) { + return { + messages: [ + { content: 'question', role: 'user', timestamp: 1 }, + { content: answer, role: 'assistant', timestamp: 2 } + ], + session_id: ACTIVE_STORED_ID + } +} + +function makeRefresh(resolveSession: ActiveTranscriptRefreshDeps['resolveSession'] = () => ({ profile: 'default' })) { + const activeSessionIdRef = { current: ACTIVE_RUNTIME_ID as string | null } + const selectedStoredSessionIdRef = { current: ACTIVE_STORED_ID as string | null } + const busyRef = { current: false } + const requestSequenceRef = { current: 0 } + const signatureRef = { current: new Map() } + const state = createClientSessionState(ACTIVE_STORED_ID) + const states = new Map([[ACTIVE_RUNTIME_ID, state]]) + + const updateSessionState = vi.fn((sessionId: string, updater: (value: typeof state) => typeof state) => { + const next = updater(states.get(sessionId) ?? createClientSessionState(ACTIVE_STORED_ID)) + states.set(sessionId, next) + + return next + }) + + const refresh = () => + reconcileActiveTranscript({ + activeSessionIdRef, + busyRef, + requestSequenceRef, + resolveSession, + selectedStoredSessionIdRef, + signatureRef, + updateSessionState + }) + + return { activeSessionIdRef, busyRef, refresh, selectedStoredSessionIdRef, state, states, updateSessionState } +} + +function useSyncHarness({ + activeIsMessaging = false, + activeSessionId, + activeStoredSessionId, + refreshActiveTranscript +}: { + activeIsMessaging?: boolean + activeSessionId: string | null + activeStoredSessionId: string | null + refreshActiveTranscript: () => Promise +}) { + useBackgroundSync({ + activeGatewayProfile: 'default', + activeIsMessaging, + activeSessionId, + activeStoredSessionId, + freshDraftReady: false, + gatewayState: 'open', + refreshActiveTranscript, + refreshCronJobs: vi.fn(), + refreshCurrentModel: vi.fn(), + refreshHermesConfig: vi.fn(), + refreshMessagingSessions: vi.fn(), + refreshSessions: vi.fn(), + requestGateway: vi.fn(async () => ({ sessions: [] })) as never + }) +} + +function renderSync( + refreshActiveTranscript: () => Promise, + options: { activeIsMessaging?: boolean; activeSessionId?: null | string; activeStoredSessionId?: null | string } = {} +) { + return renderHook(() => + useSyncHarness({ + activeSessionId: ACTIVE_RUNTIME_ID, + activeStoredSessionId: ACTIVE_STORED_ID, + refreshActiveTranscript, + ...options + }) + ) +} + +afterEach(() => { + cleanup() + vi.clearAllTimers() + vi.useRealTimers() + resetLiveSync() + $activeSessionId.set(null) + $selectedStoredSessionId.set(null) + setSessions([]) + setMessagingSessions([]) + setBusy(false) + vi.clearAllMocks() + clearAllSessionStates() +}) + +describe('active transcript refresh', () => { beforeEach(() => { + vi.mocked(getLatestSessionMessages).mockResolvedValue(transcript('answer') as never) + }) + + it('refreshes a local/Desktop session when sessions.changed ticks', async () => { + $changeEventsAvailable.set(true) + $activeSessionId.set(ACTIVE_RUNTIME_ID) + $selectedStoredSessionId.set(ACTIVE_STORED_ID) + setSessions([{ id: ACTIVE_STORED_ID, profile: 'desktop-profile', source: 'desktop' } as never]) + const fixture = makeRefresh(resolveActiveTranscriptSession) + vi.mocked(getLatestSessionMessages).mockResolvedValue(transcript('external answer') as never) + + renderSync(fixture.refresh) + + act(() => notifySessionsChanged()) + + await waitFor(() => + expect(fixture.states.get(ACTIVE_RUNTIME_ID)?.messages.at(-1)?.parts[0]).toMatchObject({ + text: 'external answer' + }) + ) + }) + + it('does not add a periodic transcript poll to local/Desktop sessions', async () => { vi.useFakeTimers() + $changeEventsAvailable.set(true) + const refresh = vi.fn(async () => undefined) + + renderSync(refresh) + expect(refresh).not.toHaveBeenCalled() + + await act(async () => { + vi.advanceTimersByTime(60_000) + await Promise.resolve() + }) + + expect(refresh).not.toHaveBeenCalled() }) - afterEach(() => { - vi.clearAllTimers() - vi.useRealTimers() - clearAllSessionStates() + it('retains the existing periodic backstop for messaging sessions', async () => { + vi.useFakeTimers() + $changeEventsAvailable.set(true) + const refresh = vi.fn(async () => undefined) + + renderSync(refresh, { activeIsMessaging: true }) + expect(refresh).toHaveBeenCalledTimes(1) + await act(async () => Promise.resolve()) + refresh.mockClear() + + await act(async () => { + vi.advanceTimersByTime(30_000) + await Promise.resolve() + }) + + expect(refresh).toHaveBeenCalledTimes(1) }) + it('only defers an external tick while busy, then refreshes once after idle', async () => { + $changeEventsAvailable.set(true) + setBusy(true) + const refresh = vi.fn(async () => undefined) + + renderSync(refresh) + + act(() => setBusy(false)) + expect(refresh).not.toHaveBeenCalled() + act(() => setBusy(true)) + + act(() => { + notifySessionsChanged() + notifySessionsChanged() + }) + expect(refresh).not.toHaveBeenCalled() + + act(() => setBusy(false)) + await waitFor(() => expect(refresh).toHaveBeenCalledTimes(1)) + }) + + it('coalesces a burst of global ticks, including writes for another session', async () => { + vi.useFakeTimers() + $changeEventsAvailable.set(true) + const refresh = vi.fn(async () => undefined) + + renderSync(refresh) + + act(() => { + for (let index = 0; index < 20; index += 1) { + notifySessionsChanged() + } + }) + expect(refresh).toHaveBeenCalledTimes(1) + + await act(async () => { + vi.advanceTimersByTime(10_000) + await Promise.resolve() + }) + + expect(refresh).toHaveBeenCalledTimes(1) + }) +}) + +describe('reconcileActiveTranscript', () => { + it('resolves and hydrates a messaging session from the messaging sessions store', async () => { + setMessagingSessions([{ id: ACTIVE_STORED_ID, profile: 'messaging-profile', source: 'telegram' } as never]) + const fixture = makeRefresh(resolveActiveTranscriptSession) + vi.mocked(getLatestSessionMessages).mockResolvedValue(transcript('telegram answer') as never) + + await fixture.refresh() + + expect(getLatestSessionMessages).toHaveBeenCalledWith(ACTIVE_STORED_ID, 'messaging-profile') + expect(fixture.states.get(ACTIVE_RUNTIME_ID)?.messages.at(-1)?.parts[0]).toMatchObject({ + text: 'telegram answer' + }) + }) + + it('publishes changed authoritative messages once without duplicates', async () => { + const fixture = makeRefresh() + vi.mocked(getLatestSessionMessages).mockResolvedValue(transcript('new answer') as never) + + await fixture.refresh() + + expect(fixture.updateSessionState).toHaveBeenCalledTimes(1) + const messages = fixture.states.get(ACTIVE_RUNTIME_ID)?.messages ?? [] + expect(messages.map(message => message.role)).toEqual(['user', 'assistant']) + expect(new Set(messages.map(message => message.id)).size).toBe(messages.length) + + await fixture.refresh() + + expect(fixture.updateSessionState).toHaveBeenCalledTimes(1) + }) + + it('preserves a local assistant error while hydrating authoritative messages', async () => { + const fixture = makeRefresh() + fixture.state.messages = [ + { id: '1-0-user', parts: [{ text: 'question', type: 'text' }], role: 'user' }, + { error: 'local failure', id: 'assistant-error', parts: [], role: 'assistant' } + ] + vi.mocked(getLatestSessionMessages).mockResolvedValue({ + messages: [{ content: 'question', role: 'user', timestamp: 1 }], + session_id: ACTIVE_STORED_ID + } as never) + + await fixture.refresh() + + const messages = fixture.states.get(ACTIVE_RUNTIME_ID)?.messages ?? [] + expect(messages.map(message => message.id)).toEqual(['1-0-user', 'assistant-error']) + expect(messages.at(-1)?.error).toBe('local failure') + }) + + it('does not clobber a busy stream', async () => { + const fixture = makeRefresh() + fixture.busyRef.current = true + + await fixture.refresh() + + expect(getLatestSessionMessages).not.toHaveBeenCalled() + expect(fixture.updateSessionState).not.toHaveBeenCalled() + }) + + it('discards a response when the active session changes in flight', async () => { + const fixture = makeRefresh() + let resolve: ((value: unknown) => void) | undefined + vi.mocked(getLatestSessionMessages).mockReturnValueOnce( + new Promise(currentResolve => { + resolve = currentResolve + }) as never + ) + + const request = fixture.refresh() + fixture.selectedStoredSessionIdRef.current = 'stored-other' + fixture.activeSessionIdRef.current = 'runtime-other' + resolve?.(transcript('stale answer')) + await request + + expect(fixture.updateSessionState).not.toHaveBeenCalled() + }) +}) + +describe('rehydrateLiveSessionStatuses', () => { it('restores running sessions after reconnect without opening them', () => { const now = 1_800_000_000_000 diff --git a/apps/desktop/src/app/contrib/hooks/use-background-sync.ts b/apps/desktop/src/app/contrib/hooks/use-background-sync.ts index 32abacdacd241..793d5da2b8f17 100644 --- a/apps/desktop/src/app/contrib/hooks/use-background-sync.ts +++ b/apps/desktop/src/app/contrib/hooks/use-background-sync.ts @@ -1,11 +1,23 @@ import { useStore } from '@nanostores/react' -import { useEffect } from 'react' +import { type MutableRefObject, useCallback, useEffect, useRef } from 'react' +import { getLatestSessionMessages } from '@/hermes' +import { preserveLocalAssistantErrors, toChatMessages } from '@/lib/chat-messages' import { createClientSessionState } from '@/lib/chat-runtime' +import { sessionMessagesSignature } from '@/lib/session-signatures' import { $changeEventsAvailable, $cronChangeTick, $sessionsChangeTick } from '@/store/live-sync' import { $onBattery, batteryPollInterval } from '@/store/power' import { refreshActiveProfile } from '@/store/profile' -import { $activeSessionId, $currentCwd, setCurrentCwd } from '@/store/session' +import { + $activeSessionId, + $busy, + $currentCwd, + $messagingSessions, + $selectedStoredSessionId, + $sessions, + sessionMatchesStoredId, + setCurrentCwd +} from '@/store/session' import { $sessionStates, publishSessionState, @@ -13,8 +25,93 @@ import { setSessionStalled } from '@/store/session-states' +import type { ClientSessionState } from '../../types' import type { GatewayRequester } from '../types' +interface ActiveTranscriptSession { + profile?: string | null +} + +/** Resolve an active transcript from either local recents or messaging slices. */ +export function resolveActiveTranscriptSession(storedSessionId: string): ActiveTranscriptSession | undefined { + return ( + $sessions.get().find(session => sessionMatchesStoredId(session, storedSessionId)) ?? + $messagingSessions.get().find(session => sessionMatchesStoredId(session, storedSessionId)) + ) +} + +export interface ActiveTranscriptRefreshDeps { + activeSessionIdRef: MutableRefObject + busyRef: MutableRefObject + requestSequenceRef: MutableRefObject + selectedStoredSessionIdRef: MutableRefObject + resolveSession: (storedSessionId: string) => ActiveTranscriptSession | null | undefined + signatureRef: MutableRefObject> + updateSessionState: ( + sessionId: string, + updater: (state: ClientSessionState) => ClientSessionState, + storedSessionId?: string | null + ) => ClientSessionState +} + +/** Reconcile one persisted transcript snapshot into the currently viewed session. */ +export async function reconcileActiveTranscript({ + activeSessionIdRef, + busyRef, + requestSequenceRef, + resolveSession, + selectedStoredSessionIdRef, + signatureRef, + updateSessionState +}: ActiveTranscriptRefreshDeps): Promise { + const storedSessionId = selectedStoredSessionIdRef.current + const runtimeSessionId = activeSessionIdRef.current + + if (!storedSessionId || !runtimeSessionId || busyRef.current) { + return + } + + const stored = resolveSession(storedSessionId) + + if (!stored) { + return + } + + const requestId = requestSequenceRef.current + 1 + requestSequenceRef.current = requestId + + try { + const latest = await getLatestSessionMessages(storedSessionId, stored.profile) + + if ( + requestId !== requestSequenceRef.current || + busyRef.current || + selectedStoredSessionIdRef.current !== storedSessionId || + activeSessionIdRef.current !== runtimeSessionId + ) { + return + } + + const signatureKey = `${stored.profile ?? 'default'}:${storedSessionId}` + const signature = sessionMessagesSignature(latest.messages) + + if (signatureRef.current.get(signatureKey) === signature) { + return + } + + signatureRef.current.set(signatureKey, signature) + const messages = toChatMessages(latest.messages) + + updateSessionState( + runtimeSessionId, + state => ({ ...state, messages: preserveLocalAssistantErrors(messages, state.messages) }), + storedSessionId + ) + } catch { + // Non-fatal: the next change event or manual resume can hydrate the view. + } +} + // Cron sessions are written by a background scheduler tick, messaging turns by // the background gateway (Telegram, WeChat, Discord, …) — neither signals the // desktop websocket directly. Backends with the change watcher broadcast @@ -177,9 +274,10 @@ interface BackgroundSyncParams { activeGatewayProfile: string activeIsMessaging: boolean activeSessionId: null | string + activeStoredSessionId: null | string freshDraftReady: boolean gatewayState: string - refreshActiveMessagingTranscript: () => Promise | unknown + refreshActiveTranscript: () => Promise | unknown refreshCronJobs: () => Promise | unknown refreshCurrentModel: (force?: boolean) => Promise | unknown refreshHermesConfig: () => Promise | unknown @@ -226,9 +324,10 @@ export function useBackgroundSync({ activeGatewayProfile, activeIsMessaging, activeSessionId, + activeStoredSessionId, freshDraftReady, gatewayState, - refreshActiveMessagingTranscript, + refreshActiveTranscript, refreshCronJobs, refreshCurrentModel, refreshHermesConfig, @@ -239,6 +338,48 @@ export function useBackgroundSync({ const changeEventsAvailable = useStore($changeEventsAvailable) const cronChangeTick = useStore($cronChangeTick) const sessionsChangeTick = useStore($sessionsChangeTick) + const activeTranscriptBusy = useStore($busy) + const activeTranscriptRefreshPendingRef = useRef(null) + + const requestActiveTranscriptRefresh = useCallback( + (preservePending: boolean) => { + if (!activeStoredSessionId || !activeSessionId) { + return + } + + const storedSessionId = activeStoredSessionId + const runtimeSessionId = activeSessionId + const sessionKey = `${storedSessionId}:${runtimeSessionId}` + + if (preservePending) { + activeTranscriptRefreshPendingRef.current = sessionKey + } + + if ($busy.get()) { + return + } + + if (preservePending && activeTranscriptRefreshPendingRef.current === sessionKey) { + activeTranscriptRefreshPendingRef.current = null + } + + void Promise.resolve(refreshActiveTranscript()).finally(() => { + // If streaming began while the read was in flight, reconciliation was + // discarded and the external event still needs one idle retry. + if ( + preservePending && + $busy.get() && + $activeSessionId.get() === runtimeSessionId && + $selectedStoredSessionId.get() === storedSessionId + ) { + activeTranscriptRefreshPendingRef.current = sessionKey + + return + } + }) + }, + [activeSessionId, activeStoredSessionId, refreshActiveTranscript] + ) useEffect(() => { if (gatewayState !== 'open') { @@ -332,6 +473,7 @@ export function useBackgroundSync({ lastRunAt = Date.now() void refreshSessions() void refreshMessagingSessions() + requestActiveTranscriptRefresh(true) } const unsubscribe = $sessionsChangeTick.listen(() => { @@ -354,7 +496,7 @@ export function useBackgroundSync({ window.clearTimeout(timer) } } - }, [changeEventsAvailable, gatewayState, refreshMessagingSessions, refreshSessions]) + }, [changeEventsAvailable, gatewayState, refreshMessagingSessions, refreshSessions, requestActiveTranscriptRefresh]) // Keep the cron-jobs section live without a user action (scheduler ticks in // the background). cron.changed (jobs.json moved: CRUD or a scheduler tick's @@ -374,24 +516,47 @@ export function useBackgroundSync({ ) }, [changeEventsAvailable, cronChangeTick, gatewayState, refreshCronJobs]) - // Only the open messaging transcript needs its own cadence — local chats are - // live over the websocket already. sessions.changed re-pulls it via the tick - // dep; the visible poll is the backstop. + // A busy transition only consumes a pending sessions.changed refresh. It + // never creates one, so an ordinary local turn going busy -> idle does not + // add a REST read. The event itself is coalesced by the list throttle above. useEffect(() => { - if (gatewayState !== 'open' || !activeIsMessaging) { + if ( + gatewayState !== 'open' || + activeTranscriptBusy || + !activeSessionId || + !activeStoredSessionId || + activeTranscriptRefreshPendingRef.current !== `${activeStoredSessionId}:${activeSessionId}` + ) { return } - const dispose = visiblePoll( - changeEventsAvailable ? ACTIVE_MESSAGING_SESSION_BACKSTOP_INTERVAL_MS : ACTIVE_MESSAGING_SESSION_POLL_INTERVAL_MS, - () => void refreshActiveMessagingTranscript() - ) + requestActiveTranscriptRefresh(true) + }, [activeSessionId, activeStoredSessionId, activeTranscriptBusy, gatewayState, requestActiveTranscriptRefresh]) - void refreshActiveMessagingTranscript() + // Preserve the pre-existing messaging behavior: refresh once when a + // messaging transcript opens, then keep its visibility backstop. Desktop + // sessions never enter this effect and therefore gain no periodic timer. + useEffect(() => { + if (gatewayState !== 'open' || !activeIsMessaging || !activeSessionId || !activeStoredSessionId) { + return + } + + const runScheduledRefresh = () => requestActiveTranscriptRefresh(false) + + runScheduledRefresh() - return dispose - // sessionsChangeTick: an inbound turn re-pulls the open transcript. - }, [activeIsMessaging, changeEventsAvailable, gatewayState, refreshActiveMessagingTranscript, sessionsChangeTick]) + return visiblePoll( + changeEventsAvailable ? ACTIVE_MESSAGING_SESSION_BACKSTOP_INTERVAL_MS : ACTIVE_MESSAGING_SESSION_POLL_INTERVAL_MS, + runScheduledRefresh + ) + }, [ + activeIsMessaging, + activeSessionId, + activeStoredSessionId, + changeEventsAvailable, + gatewayState, + requestActiveTranscriptRefresh + ]) // Messaging session lists against an older backend: no sessions.changed, so // keep the legacy visible poll. (Event-capable backends fold this into the diff --git a/apps/desktop/src/app/contrib/wiring.tsx b/apps/desktop/src/app/contrib/wiring.tsx index 12160a7823f20..7eeaeaa47978b 100644 --- a/apps/desktop/src/app/contrib/wiring.tsx +++ b/apps/desktop/src/app/contrib/wiring.tsx @@ -26,7 +26,6 @@ import { RemoteDisplayBanner } from '@/components/remote-display-banner' import { emitGatewayEvent } from '@/contrib/events' import { getLatestSessionMessages, triggerCronJob } from '@/hermes' import { type ChatMessage, chatMessageText, preserveLocalAssistantErrors, toChatMessages } from '@/lib/chat-messages' -import { sessionMessagesSignature } from '@/lib/session-signatures' import { isMessagingSource } from '@/lib/session-source' import { latestSessionTodos } from '@/lib/todos' import { activateWakeIndicator } from '@/lib/wake-indicator' @@ -122,7 +121,11 @@ import { TitlebarControls } from '../shell/titlebar-controls' import { UpdatesOverlay } from '../updates-overlay' import { ContribWiringContext } from './context' -import { useBackgroundSync } from './hooks/use-background-sync' +import { + reconcileActiveTranscript, + resolveActiveTranscriptSession, + useBackgroundSync +} from './hooks/use-background-sync' import { useDesktopIntegrations } from './hooks/use-desktop-integrations' import { usePetBridge } from './hooks/use-pet-bridge' import { useQuickEntryBridge } from './hooks/use-quick-entry-bridge' @@ -159,7 +162,8 @@ export function ContribWiring({ children }: { children: ReactNode }) { // intent counter here; the ref skips the initial mount value. const billingSettingsSeenRef = useRef(0) const cronReviewSeenRef = useRef(0) - const messagingTranscriptSignatureRef = useRef(new Map()) + const activeTranscriptSignatureRef = useRef(new Map()) + const activeTranscriptRequestSequenceRef = useRef(0) // Stable identity for the whole callback surface (see WiringActions). Mutated // in place each render so memoized surfaces never re-render on churn. const actionsRef = useRef(null) @@ -374,44 +378,21 @@ export function ContribWiring({ children }: { children: ReactNode }) { [activeSessionIdRef, selectedStoredSessionIdRef, updateSessionState] ) - // Refresh the open messaging transcript (inbound platform turns arrive via - // the background gateway, not the desktop websocket). Signature-gated so a - // no-change poll doesn't churn the thread. - const refreshActiveMessagingTranscript = useCallback(async () => { - const storedSessionId = selectedStoredSessionIdRef.current - const runtimeSessionId = activeSessionIdRef.current - - if (!storedSessionId || !runtimeSessionId || busyRef.current) { - return - } - - const stored = $messagingSessions.get().find(s => sessionMatchesStoredId(s, storedSessionId)) - - if (!stored || !isMessagingSource(stored.source)) { - return - } - - try { - const latest = await getLatestSessionMessages(storedSessionId, stored.profile) - const signatureKey = `${stored.profile ?? 'default'}:${storedSessionId}` - const sig = sessionMessagesSignature(latest.messages) - - if (messagingTranscriptSignatureRef.current.get(signatureKey) === sig) { - return - } - - messagingTranscriptSignatureRef.current.set(signatureKey, sig) - const messages = toChatMessages(latest.messages) - - updateSessionState( - runtimeSessionId, - state => ({ ...state, messages: preserveLocalAssistantErrors(messages, state.messages) }), - storedSessionId - ) - } catch { - // Non-fatal: next poll or manual refresh can hydrate. - } - }, [activeSessionIdRef, busyRef, selectedStoredSessionIdRef, updateSessionState]) + // Refresh any active transcript changed by another process. Signature-gated + // so a no-change event does not churn the thread. + const refreshActiveTranscript = useCallback( + () => + reconcileActiveTranscript({ + activeSessionIdRef, + busyRef, + requestSequenceRef: activeTranscriptRequestSequenceRef, + resolveSession: resolveActiveTranscriptSession, + selectedStoredSessionIdRef, + signatureRef: activeTranscriptSignatureRef, + updateSessionState + }), + [activeSessionIdRef, busyRef, selectedStoredSessionIdRef, updateSessionState] + ) const { handleGatewayEvent } = useMessageStream({ activeGatewayProfile, @@ -769,21 +750,22 @@ export function ContribWiring({ children }: { children: ReactNode }) { } }, [gatewayState, requestGateway]) - // Only the open messaging transcript needs its own poll — local chats are - // live over the websocket already. const activeIsMessaging = !!selectedStoredSessionId && isMessagingSource(messagingSessions.find(s => sessionMatchesStoredId(s, selectedStoredSessionId))?.source) + // sessions.changed refreshes every open transcript; only messaging retains + // the periodic safety-net it already had before this fix. // Keep app data live while the gateway is open (on-connect reseed + the // cron / messaging / transcript visibility polls + fresh-draft reseed). useBackgroundSync({ activeGatewayProfile, activeIsMessaging, activeSessionId, + activeStoredSessionId: selectedStoredSessionId, freshDraftReady, gatewayState, - refreshActiveMessagingTranscript, + refreshActiveTranscript, refreshCronJobs, refreshCurrentModel, refreshHermesConfig, From 01e542b764442d5271add29a12470cba66375812 Mon Sep 17 00:00:00 2001 From: Guilherme Aguiar Date: Fri, 14 Aug 2026 17:47:06 -0300 Subject: [PATCH 020/376] fix(desktop): address transcript refresh review feedback --- .../app/contrib/hooks/use-background-sync.test.ts | 4 ++-- .../src/app/contrib/hooks/use-background-sync.ts | 12 +++++++++--- 2 files changed, 11 insertions(+), 5 deletions(-) diff --git a/apps/desktop/src/app/contrib/hooks/use-background-sync.test.ts b/apps/desktop/src/app/contrib/hooks/use-background-sync.test.ts index b4d5b6e6c257a..4cb9f0bbc7a5a 100644 --- a/apps/desktop/src/app/contrib/hooks/use-background-sync.test.ts +++ b/apps/desktop/src/app/contrib/hooks/use-background-sync.test.ts @@ -205,7 +205,7 @@ describe('active transcript refresh', () => { await waitFor(() => expect(refresh).toHaveBeenCalledTimes(1)) }) - it('coalesces a burst of global ticks, including writes for another session', async () => { + it('coalesces a burst of global session-change ticks', async () => { vi.useFakeTimers() $changeEventsAvailable.set(true) const refresh = vi.fn(async () => undefined) @@ -220,7 +220,7 @@ describe('active transcript refresh', () => { expect(refresh).toHaveBeenCalledTimes(1) await act(async () => { - vi.advanceTimersByTime(10_000) + vi.advanceTimersByTime(9_999) await Promise.resolve() }) diff --git a/apps/desktop/src/app/contrib/hooks/use-background-sync.ts b/apps/desktop/src/app/contrib/hooks/use-background-sync.ts index 793d5da2b8f17..2105bb050c9aa 100644 --- a/apps/desktop/src/app/contrib/hooks/use-background-sync.ts +++ b/apps/desktop/src/app/contrib/hooks/use-background-sync.ts @@ -363,18 +363,24 @@ export function useBackgroundSync({ activeTranscriptRefreshPendingRef.current = null } + let sawBusyDuringRead = false + + const unsubscribeBusy = $busy.listen(busy => { + sawBusyDuringRead ||= busy + }) + void Promise.resolve(refreshActiveTranscript()).finally(() => { + unsubscribeBusy() + // If streaming began while the read was in flight, reconciliation was // discarded and the external event still needs one idle retry. if ( preservePending && - $busy.get() && + (sawBusyDuringRead || $busy.get()) && $activeSessionId.get() === runtimeSessionId && $selectedStoredSessionId.get() === storedSessionId ) { activeTranscriptRefreshPendingRef.current = sessionKey - - return } }) }, From 5a23ab513c41ba6549c1c2fd63c8dec849208cb0 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 18:46:01 -0700 Subject: [PATCH 021/376] chore: map salvage contributor emails to GitHub usernames --- contributors/emails/602028@ky-tech.com.cn | 1 + contributors/emails/guilherme@guilhermeaguiar.com | 1 + 2 files changed, 2 insertions(+) create mode 100644 contributors/emails/602028@ky-tech.com.cn create mode 100644 contributors/emails/guilherme@guilhermeaguiar.com diff --git a/contributors/emails/602028@ky-tech.com.cn b/contributors/emails/602028@ky-tech.com.cn new file mode 100644 index 0000000000000..4c4104b138784 --- /dev/null +++ b/contributors/emails/602028@ky-tech.com.cn @@ -0,0 +1 @@ +baihemax diff --git a/contributors/emails/guilherme@guilhermeaguiar.com b/contributors/emails/guilherme@guilhermeaguiar.com new file mode 100644 index 0000000000000..8cbf238a0bb77 --- /dev/null +++ b/contributors/emails/guilherme@guilhermeaguiar.com @@ -0,0 +1 @@ +guilhermeraiuga From 21b57c61f4be6f08833897b9044c2e4be9b70544 Mon Sep 17 00:00:00 2001 From: Jakub Wolniewicz <4850809+frizikk@users.noreply.github.com> Date: Thu, 13 Aug 2026 13:06:10 +0200 Subject: [PATCH 022/376] fix(desktop): page remote profile session reads --- apps/desktop/electron/main.ts | 17 +- .../electron/profile-session-routing.test.ts | 149 ++++++++++++++- .../electron/profile-session-routing.ts | 176 +++++++++++++++++- 3 files changed, 334 insertions(+), 8 deletions(-) diff --git a/apps/desktop/electron/main.ts b/apps/desktop/electron/main.ts index 6fd8a0dba830f..b66faf6b82165 100644 --- a/apps/desktop/electron/main.ts +++ b/apps/desktop/electron/main.ts @@ -186,7 +186,11 @@ import { createKeepAwake } from './power-save' import { FirstRunSetupResetError, runPrimaryBackendStartup } from './primary-backend-startup' import { rehomePrimaryConnection } from './primary-connection-rehome' import { decideProfileDeleteAction, profileNameFromDeleteRequest, resolveRouteProfile } from './profile-delete-routing' -import { fetchPrimaryProfileSessions } from './profile-session-routing' +import { + fetchPrimaryProfileSessions, + fetchRemoteProfileSessions, + mergeProfileSessionWindow +} from './profile-session-routing' import { createQuickEntryShortcut, quickEntryWindowBounds, sanitizeQuickEntrySettings } from './quick-entry' import { type ActiveWork, mergeActiveWork, normalizeActiveWork, quitPromptFor } from './quit-guard' import * as remoteLifecycle from './remote-lifecycle' @@ -11212,9 +11216,7 @@ const rowsOf = data => (Array.isArray(data?.sessions) ? data.sessions : []) // A remote profile's session list, read from its remote host and tagged with the // desktop-facing profile name (the remote's /api/sessions doesn't know it). async function remoteSessionList(profile, searchParams) { - const qs = new URLSearchParams(searchParams) - qs.delete('profile') // remote serves its own db; no cross-profile read there - const data = await fetchJsonForProfile(profile, `/api/sessions?${qs}`) + const data = await fetchRemoteProfileSessions(profile, searchParams, fetchJsonForProfile) for (const s of rowsOf(data)) { s.profile = profile @@ -11286,7 +11288,12 @@ async function mergeRemoteProfileSessions(searchParams, remoteProfiles) { const recency = s => s?.[order] ?? s?.started_at ?? 0 merged.sort((a, b) => recency(b) - recency(a)) - return { ...(base as any), sessions: merged.slice(offset, offset + limit), total, profile_totals: profileTotals } + return { + ...(base as any), + sessions: mergeProfileSessionWindow(merged, offset, limit), + total, + profile_totals: profileTotals + } } ipcMain.handle('hermes:api', async (_event, request) => { diff --git a/apps/desktop/electron/profile-session-routing.test.ts b/apps/desktop/electron/profile-session-routing.test.ts index 199740519bd6b..a96c84d062f25 100644 --- a/apps/desktop/electron/profile-session-routing.test.ts +++ b/apps/desktop/electron/profile-session-routing.test.ts @@ -2,7 +2,11 @@ import assert from 'node:assert/strict' import { test } from 'vitest' -import { fetchPrimaryProfileSessions } from './profile-session-routing' +import { + fetchPrimaryProfileSessions, + fetchRemoteProfileSessions, + mergeProfileSessionWindow +} from './profile-session-routing' test('primary session reads use the profile-aware request path', async () => { const calls: Array<{ profile: string | null; path: string }> = [] @@ -28,3 +32,146 @@ test('primary session reads preserve the empty-list fallback', async () => { assert.deepEqual(result, { sessions: [], total: 0, profile_totals: {} }) }) + +test('remote session reads split oversized sidebar windows into API-safe pages', async () => { + const calls: Array<{ profile: string | null; path: string }> = [] + const rows = Array.from({ length: 250 }, (_, index) => ({ id: `session-${index}` })) + + const result = await fetchRemoteProfileSessions( + 'remote-work', + new URLSearchParams({ profile: 'remote-work', limit: '300', offset: '0', order: 'updated' }), + async (profile, path) => { + calls.push({ profile, path }) + const url = new URL(path, 'http://desktop.test') + const limit = Number(url.searchParams.get('limit')) + const offset = Number(url.searchParams.get('offset')) + + if (limit > 100) { + throw new Error(`remote /api/sessions rejects limit ${limit}`) + } + + return { + sessions: rows.slice(offset, offset + limit), + total: rows.length, + limit, + offset + } + } + ) + + assert.deepEqual(calls, [ + { profile: 'remote-work', path: '/api/sessions?limit=100&offset=0&order=updated' }, + { profile: 'remote-work', path: '/api/sessions?limit=100&offset=100&order=updated' }, + { profile: 'remote-work', path: '/api/sessions?limit=50&offset=200&order=updated' } + ]) + assert.equal(result.sessions.length, 250) + assert.equal(result.total, 250) + assert.equal(result.limit, 300) + assert.equal(result.offset, 0) + assert.deepEqual( + result.sessions.map(row => (row as { id: string }).id), + rows.map(row => row.id) + ) +}) + +test('remote paging preserves offsets and deduplicates pinned backfill rows', async () => { + const calls: string[] = [] + + const rows = Array.from({ length: 240 }, (_, index) => ({ + id: `session-${index}`, + pinned: index === 20 || index === 200 + })) + + const pinned = rows.filter(row => row.pinned) + + const result = await fetchRemoteProfileSessions( + 'remote-work', + new URLSearchParams({ profile: 'remote-work', limit: '150', offset: '80' }), + async (_profile, path) => { + calls.push(path) + const url = new URL(path, 'http://desktop.test') + const limit = Number(url.searchParams.get('limit')) + const offset = Number(url.searchParams.get('offset')) + const window = rows.slice(offset, offset + limit) + const windowIds = new Set(window.map(row => row.id)) + + return { + sessions: [...window, ...pinned.filter(row => !windowIds.has(row.id))], + total: rows.length, + limit, + offset + } + } + ) + + assert.deepEqual(calls, ['/api/sessions?limit=100&offset=80', '/api/sessions?limit=50&offset=180']) + assert.deepEqual( + result.sessions.map(row => (row as { id: string }).id), + [...rows.slice(80, 230).map(row => row.id), 'session-20'] + ) +}) + +test('remote paging treats malformed totals as unknown instead of truncating the result', async () => { + const rows = Array.from({ length: 250 }, (_, index) => ({ id: `session-${index}` })) + + for (const malformedTotal of [null, '', false, 100.5]) { + const calls: string[] = [] + + const result = await fetchRemoteProfileSessions( + 'remote-work', + new URLSearchParams({ limit: '300', offset: '0' }), + async (_profile, path) => { + calls.push(path) + const url = new URL(path, 'http://desktop.test') + const limit = Number(url.searchParams.get('limit')) + const offset = Number(url.searchParams.get('offset')) + + return { + sessions: rows.slice(offset, offset + limit), + total: malformedTotal, + limit, + offset + } + } + ) + + assert.deepEqual(calls, [ + '/api/sessions?limit=100&offset=0', + '/api/sessions?limit=100&offset=100', + '/api/sessions?limit=100&offset=200' + ]) + assert.equal(result.sessions.length, 250) + assert.equal(result.total, 250) + } +}) + +test('merged profile windows retain pinned rows outside the recency window', () => { + const rows = [ + { id: 'recent-default', profile: 'default', pinned: false }, + { id: 'shared-id', profile: 'default', pinned: false }, + { id: 'recent-remote', profile: 'remote-work', pinned: false }, + { id: 'shared-id', profile: 'remote-work', pinned: true }, + { id: 'old-remote', profile: 'remote-work', pinned: true }, + { id: 'old-unpinned', profile: 'remote-work', pinned: false } + ] + + assert.deepEqual(mergeProfileSessionWindow(rows, 0, 3), [rows[0], rows[1], rows[2], rows[3], rows[4]]) +}) + +test('remote session reads keep small requests on one call', async () => { + const calls: Array<{ profile: string | null; path: string }> = [] + const expected = { sessions: [{ id: 'session-1' }], total: 1, limit: 20, offset: 0 } + + const result = await fetchRemoteProfileSessions( + 'remote-work', + new URLSearchParams({ profile: 'remote-work', limit: '20', offset: '0' }), + async (profile, path) => { + calls.push({ profile, path }) + + return expected + } + ) + + assert.deepEqual(calls, [{ profile: 'remote-work', path: '/api/sessions?limit=20&offset=0' }]) + assert.equal(result, expected) +}) diff --git a/apps/desktop/electron/profile-session-routing.ts b/apps/desktop/electron/profile-session-routing.ts index f31e22ca52a34..dada65a7615be 100644 --- a/apps/desktop/electron/profile-session-routing.ts +++ b/apps/desktop/electron/profile-session-routing.ts @@ -1,12 +1,82 @@ -export interface ProfileSessionsResponse { +interface SessionListResponse { sessions: unknown[] total: number - profile_totals: Record [key: string]: unknown } +export interface ProfileSessionsResponse extends SessionListResponse { + profile_totals: Record +} + type FetchJsonForProfile = (profile: string | null, path: string) => Promise +const REMOTE_SESSION_PAGE_LIMIT = 100 + +function rowsOf(data: unknown): unknown[] { + if (!data || typeof data !== 'object' || !('sessions' in data)) { + return [] + } + + return Array.isArray(data.sessions) ? data.sessions : [] +} + +function sessionId(row: unknown): string | null { + if (!row || typeof row !== 'object' || !('id' in row)) { + return null + } + + return typeof row.id === 'string' ? row.id : null +} + +function nonNegativeNumber(value: unknown): number | null { + return typeof value === 'number' && Number.isInteger(value) && value >= 0 ? value : null +} + +function isPinned(row: unknown): boolean { + return Boolean(row && typeof row === 'object' && 'pinned' in row && row.pinned) +} + +function profileSessionId(row: unknown): string | null { + const id = sessionId(row) + + if (!id) { + return null + } + + const profile = + row && typeof row === 'object' && 'profile' in row && typeof row.profile === 'string' ? row.profile : '' + + return `${profile}\0${id}` +} + +export function mergeProfileSessionWindow(rows: unknown[], offset: number, limit: number): unknown[] { + const window = rows.slice(offset, offset + limit) + const seenRows = new Set(window) + const seenIds = new Set(window.map(profileSessionId).filter((id): id is string => id !== null)) + + for (const row of rows.slice(offset + limit)) { + if (!isPinned(row)) { + continue + } + + const id = profileSessionId(row) + + if ((id && seenIds.has(id)) || (!id && seenRows.has(row))) { + continue + } + + if (id) { + seenIds.add(id) + } else { + seenRows.add(row) + } + + window.push(row) + } + + return window +} + export async function fetchPrimaryProfileSessions( searchParams: URLSearchParams, fetchJsonForProfile: FetchJsonForProfile @@ -17,3 +87,105 @@ export async function fetchPrimaryProfileSessions( return { sessions: [], total: 0, profile_totals: {} } } } + +export async function fetchRemoteProfileSessions( + profile: string, + searchParams: URLSearchParams, + fetchJsonForProfile: FetchJsonForProfile +): Promise { + const params = new URLSearchParams(searchParams) + params.delete('profile') // the remote serves its own database + + const requestedLimit = Number(params.get('limit')) + const requestedOffset = Number(params.get('offset') || '0') + + const needsPaging = + Number.isInteger(requestedLimit) && + requestedLimit > REMOTE_SESSION_PAGE_LIMIT && + Number.isInteger(requestedOffset) && + requestedOffset >= 0 + + if (!needsPaging) { + return (await fetchJsonForProfile(profile, `/api/sessions?${params}`)) as SessionListResponse + } + + const sessions: unknown[] = [] + const backfilled: unknown[] = [] + const seenIds = new Set() + const backfilledIds = new Set() + let firstPage: SessionListResponse | null = null + let pageOffset = requestedOffset + let targetOffset = requestedOffset + requestedLimit + + while (pageOffset < targetOffset) { + const pageParams = new URLSearchParams(params) + const pageLimit = Math.min(REMOTE_SESSION_PAGE_LIMIT, targetOffset - pageOffset) + pageParams.set('limit', String(pageLimit)) + pageParams.set('offset', String(pageOffset)) + + const page = (await fetchJsonForProfile(profile, `/api/sessions?${pageParams}`)) as SessionListResponse + firstPage ??= page + + const total = nonNegativeNumber(page.total) + const pageRows = rowsOf(page) + + const windowedCount = + total !== null ? Math.min(pageLimit, Math.max(0, total - pageOffset)) : Math.min(pageLimit, pageRows.length) + + // /api/sessions appends pinned rows that fall outside the requested + // window. Keep those aside until all ordinary pages have been joined so + // pagination preserves the same order as one larger request. + for (const row of pageRows.slice(0, windowedCount)) { + const id = sessionId(row) + + if (id && seenIds.has(id)) { + continue + } + + if (id) { + seenIds.add(id) + backfilledIds.delete(id) + } + + sessions.push(row) + } + + for (const row of pageRows.slice(windowedCount)) { + const id = sessionId(row) + + if ((id && seenIds.has(id)) || (id && backfilledIds.has(id))) { + continue + } + + if (id) { + backfilledIds.add(id) + } + + backfilled.push(row) + } + + if (total !== null) { + targetOffset = Math.min(targetOffset, total) + } + + pageOffset += pageLimit + } + + for (const row of backfilled) { + const id = sessionId(row) + + if (!id || backfilledIds.has(id)) { + sessions.push(row) + } + } + + const total = nonNegativeNumber(firstPage?.total) + + return { + ...(firstPage || {}), + sessions, + total: total ?? sessions.length, + limit: requestedLimit, + offset: requestedOffset + } +} From 8edb4626ca228fece2bcc3ef8ec3e71dd4d1750f Mon Sep 17 00:00:00 2001 From: A9 Date: Fri, 14 Aug 2026 11:37:39 -0400 Subject: [PATCH 023/376] fix(desktop): refresh sessions on profile switch Re-run the foreground session-list refresh whenever the active gateway profile changes, preventing rows from the previously selected profile from persisting in the sidebar. --- .../hooks/use-background-sync.test.tsx | 61 +++++++++++++++++++ .../app/contrib/hooks/use-background-sync.ts | 2 +- 2 files changed, 62 insertions(+), 1 deletion(-) create mode 100644 apps/desktop/src/app/contrib/hooks/use-background-sync.test.tsx diff --git a/apps/desktop/src/app/contrib/hooks/use-background-sync.test.tsx b/apps/desktop/src/app/contrib/hooks/use-background-sync.test.tsx new file mode 100644 index 0000000000000..c11a40bc1518d --- /dev/null +++ b/apps/desktop/src/app/contrib/hooks/use-background-sync.test.tsx @@ -0,0 +1,61 @@ +import { act, cleanup, renderHook } from '@testing-library/react' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +import { $changeEventsAvailable, $cronChangeTick, $sessionsChangeTick } from '@/store/live-sync' +import { $activeSessionId } from '@/store/session' + +import { useBackgroundSync } from './use-background-sync' + +const noop = () => undefined +const requestGateway = async () => ({ sessions: [] }) + +function render(activeGatewayProfile: string, refreshSessions: () => Promise) { + return renderHook( + ({ profile }: { profile: string }) => { + useBackgroundSync({ + activeGatewayProfile: profile, + activeIsMessaging: false, + activeSessionId: null, + freshDraftReady: false, + gatewayState: 'open', + refreshActiveMessagingTranscript: noop, + refreshCronJobs: noop, + refreshCurrentModel: noop, + refreshHermesConfig: noop, + refreshMessagingSessions: noop, + refreshSessions, + requestGateway + }) + }, + { initialProps: { profile: activeGatewayProfile } } + ) +} + +describe('useBackgroundSync profile-scoped session refresh', () => { + beforeEach(() => { + vi.useFakeTimers() + $activeSessionId.set(null) + $changeEventsAvailable.set(false) + $cronChangeTick.set(0) + $sessionsChangeTick.set(0) + }) + + afterEach(() => { + cleanup() + vi.useRealTimers() + }) + + it('refreshes the session list after the active gateway profile changes', async () => { + const refreshSessions = vi.fn(async () => undefined) + const hook = render('default', refreshSessions) + + await act(async () => undefined) + expect(refreshSessions).toHaveBeenCalledTimes(1) + refreshSessions.mockClear() + + hook.rerender({ profile: 'nova' }) + + await act(async () => undefined) + expect(refreshSessions).toHaveBeenCalledTimes(1) + }) +}) diff --git a/apps/desktop/src/app/contrib/hooks/use-background-sync.ts b/apps/desktop/src/app/contrib/hooks/use-background-sync.ts index 2105bb050c9aa..e42084999c17d 100644 --- a/apps/desktop/src/app/contrib/hooks/use-background-sync.ts +++ b/apps/desktop/src/app/contrib/hooks/use-background-sync.ts @@ -411,7 +411,7 @@ export function useBackgroundSync({ }) .catch(() => undefined) } - }, [gatewayState, refreshCurrentModel, refreshSessions, requestGateway]) + }, [activeGatewayProfile, gatewayState, refreshCurrentModel, refreshSessions, requestGateway]) // A reconnect loses renderer-only working/attention atoms while the backend // keeps the actual turns alive. Re-seed from the gateway's in-memory session From 077c04755ecea50b85fecaa7819fc3c2767fcd68 Mon Sep 17 00:00:00 2001 From: Trevor Nash-Keller <4496266+trevornk@users.noreply.github.com> Date: Fri, 14 Aug 2026 16:57:44 -0500 Subject: [PATCH 024/376] fix(desktop): sync active profile after reconnect --- .../src/store/gateway-shared-remote.test.ts | 53 +++++++++++++++++-- apps/desktop/src/store/gateway.ts | 7 ++- 2 files changed, 55 insertions(+), 5 deletions(-) diff --git a/apps/desktop/src/store/gateway-shared-remote.test.ts b/apps/desktop/src/store/gateway-shared-remote.test.ts index 9d1481ae07c6e..c74b312a10127 100644 --- a/apps/desktop/src/store/gateway-shared-remote.test.ts +++ b/apps/desktop/src/store/gateway-shared-remote.test.ts @@ -14,21 +14,36 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' const gatewayMocks = vi.hoisted(() => ({ connect: vi.fn(async (_wsUrl: string): Promise => { throw new Error('dialed a socket for a shared-primary profile') - }) + }), + setConnection: vi.fn() })) vi.mock('@/hermes', () => ({ HermesGateway: class { connectionState = 'closed' - connect = gatewayMocks.connect + connect = async (wsUrl: string): Promise => { + await gatewayMocks.connect(wsUrl) + this.connectionState = 'open' + } + close = vi.fn() onEvent = vi.fn(() => () => {}) onState = vi.fn(() => () => {}) } })) -vi.mock('@/store/session', () => ({ setGatewayState: vi.fn() })) +vi.mock('@/store/session', () => ({ + setConnection: gatewayMocks.setConnection, + setGatewayState: vi.fn() +})) vi.mock('@/store/notify-baseline', () => ({ markNativeNotifyBaseline: vi.fn() })) -const { $gateway, configureGatewayRegistry, ensureGatewayForProfile, setPrimaryGateway } = await import('./gateway') +const { + $gateway, + closeSecondaryGateways, + configureGatewayRegistry, + ensureActiveGatewayOpen, + ensureGatewayForProfile, + setPrimaryGateway +} = await import('./gateway') type DesktopStub = { getConnection: ReturnType } @@ -49,6 +64,7 @@ beforeEach(() => { }) afterEach(() => { + closeSecondaryGateways() vi.clearAllMocks() delete (window as unknown as { hermesDesktop?: unknown }).hermesDesktop }) @@ -95,4 +111,33 @@ describe('ensureGatewayForProfile under a shared global remote', () => { expect(gatewayMocks.connect).toHaveBeenCalledWith(remoteWsUrl) expect($gateway.get()).not.toBe(primary) }) + + it('refreshes the active connection after a pooled profile reconnect succeeds', async () => { + const connection = { + authMode: 'token', + baseUrl: 'https://worker.invalid', + mode: 'remote', + profile: 'worker', + token: 'fake-test-token', + wsUrl: 'wss://worker.invalid/api/ws?token=fake-test-token' + } + + const getConnection = vi.fn(async () => connection) + + setPrimaryGateway(makePrimary() as never, 'default') + installDesktop({ getConnection }) + + gatewayMocks.connect + .mockRejectedValueOnce(new Error('temporarily offline')) + .mockResolvedValueOnce(undefined) + + await ensureGatewayForProfile('worker') + + expect(gatewayMocks.setConnection).not.toHaveBeenCalled() + + await ensureActiveGatewayOpen() + + expect(gatewayMocks.setConnection).toHaveBeenCalledOnce() + expect(gatewayMocks.setConnection).toHaveBeenCalledWith(connection) + }) }) diff --git a/apps/desktop/src/store/gateway.ts b/apps/desktop/src/store/gateway.ts index 6c815a78834fa..8e73a1bfbf10e 100644 --- a/apps/desktop/src/store/gateway.ts +++ b/apps/desktop/src/store/gateway.ts @@ -4,7 +4,7 @@ import { atom } from 'nanostores' import { HermesGateway } from '@/hermes' import { reconnectBackoffDelayMs } from '@/lib/reconnect-backoff' import { markNativeNotifyBaseline } from '@/store/notify-baseline' -import { setGatewayState } from '@/store/session' +import { setConnection, setGatewayState } from '@/store/session' // ── Multi-profile gateway routing ────────────────────────────────────────── // Concurrent sessions across profiles need concurrent sockets: the renderer's @@ -177,6 +177,11 @@ async function openSecondary(entry: Secondary): Promise { const conn = await desktop.getConnection(entry.profile) const wsUrl = await resolveGatewayWsUrl(desktop, conn) await entry.gateway.connect(wsUrl) + + if (g.activeKey === entry.profile) { + setConnection(conn) + } + void desktop.touchBackend?.(entry.profile).catch(() => undefined) } From f9f5c14f9aaf2ce273fe4dae81e1082dfd0a72b0 Mon Sep 17 00:00:00 2001 From: KBANTH Date: Fri, 14 Aug 2026 22:36:01 +0700 Subject: [PATCH 025/376] fix(desktop): keep live gateway across profile switches --- .../desktop/src/app/contrib/surfaces.test.tsx | 67 +++++++++++++++++++ apps/desktop/src/app/contrib/surfaces.tsx | 18 ++--- 2 files changed, 72 insertions(+), 13 deletions(-) create mode 100644 apps/desktop/src/app/contrib/surfaces.test.tsx diff --git a/apps/desktop/src/app/contrib/surfaces.test.tsx b/apps/desktop/src/app/contrib/surfaces.test.tsx new file mode 100644 index 0000000000000..274da802683c3 --- /dev/null +++ b/apps/desktop/src/app/contrib/surfaces.test.tsx @@ -0,0 +1,67 @@ +import { act, cleanup, render, screen } from '@testing-library/react' +import { atom } from 'nanostores' +import { MemoryRouter } from 'react-router' +import { afterEach, describe, expect, it, vi } from 'vitest' + +import type { HermesGateway } from '@/hermes' +import { $gateway } from '@/store/gateway' +import { $activeGatewayProfile } from '@/store/profile' + +import { ChatRoutesSurface } from './surfaces' +import type { WiringActions } from './types' + +vi.mock('@/contrib/react/use-contributions', () => ({ useContributions: vi.fn() })) +vi.mock('@/store/gateway', () => ({ $gateway: atom(null) })) +vi.mock('@/store/profile', () => ({ $activeGatewayProfile: atom('default') })) +vi.mock('@/store/session', () => ({ + $freshDraftReady: atom(false), + $gatewayState: atom('open') +})) +vi.mock('../chat', () => ({ + ChatView: ({ gateway }: { gateway: { id?: string } | null }) =>
{gateway?.id}
+})) +vi.mock('../chat/sidebar', () => ({ ChatSidebar: () => null })) +vi.mock('../right-sidebar/terminal/chrome', () => ({ TerminalPaneChrome: () => null })) +vi.mock('../shell/hooks/use-status-snapshot', () => ({ useStatusSnapshot: () => ({}) })) +vi.mock('../shell/hooks/use-statusbar-items', () => ({ useStatusbarItems: () => ({ leftStatusbarItems: [], statusbarItems: [] }) })) +vi.mock('../shell/statusbar-controls', () => ({ StatusbarControls: () => null })) +vi.mock('../routes', () => ({ + contributedRoutes: () => [], + NEW_CHAT_ROUTE: '/new', + ROUTES_AREA: 'routes', + sessionRoute: (id: string) => `/${id}` +})) +vi.mock('./latest-actions', () => ({ latestChatActions: () => ({}), latestSidebarActions: () => ({}) })) +vi.mock('./panes', () => ({ setStatusbarItemGroup: vi.fn(), useStatusbarContributions: () => [] })) +vi.mock('../shell/model-menu-panel', () => ({ ModelMenuPanel: () => null })) + +afterEach(() => { + cleanup() + $gateway.set(null) + $activeGatewayProfile.set('default') +}) + +describe('ChatRoutesSurface', () => { + it('passes the live gateway after an open-to-open profile switch', () => { + const gatewayA = { id: 'a' } as unknown as HermesGateway + const gatewayB = { id: 'b' } as unknown as HermesGateway + + $gateway.set(gatewayA) + const actions = { getGateway: () => $gateway.get() } as unknown as WiringActions + + render( + + + + ) + + expect(screen.getByTestId('gateway').textContent).toBe('a') + + act(() => { + $gateway.set(gatewayB) + $activeGatewayProfile.set('other') + }) + + expect(screen.getByTestId('gateway').textContent).toBe('b') + }) +}) diff --git a/apps/desktop/src/app/contrib/surfaces.tsx b/apps/desktop/src/app/contrib/surfaces.tsx index fb5b32a25708c..abd3fb04d0b51 100644 --- a/apps/desktop/src/app/contrib/surfaces.tsx +++ b/apps/desktop/src/app/contrib/surfaces.tsx @@ -13,6 +13,7 @@ import { Navigate, Route, Routes, useParams } from 'react-router' import { ContribBoundary, ContribRender } from '@/contrib/react/boundary' import { useContributions } from '@/contrib/react/use-contributions' +import { $gateway } from '@/store/gateway' import { $activeGatewayProfile } from '@/store/profile' import { $freshDraftReady, $gatewayState } from '@/store/session' @@ -102,10 +103,9 @@ export const StatusbarSurface = memo(function StatusbarSurface({ }) /** The workspace pane: the real route table (chat + full-page views + plugin - * routes). Subscribes to `$gatewayState` and ROUTES_AREA itself; the gateway - * instance + voice cap arrive as props so a reconnect/config load re-renders - * only this surface. ChatView subscribes to its own session atoms, so - * streaming never round-trips through the controller. */ + * routes). Subscribes to the gateway instance/state and ROUTES_AREA itself; + * the voice cap arrives as a prop. ChatView subscribes to its own session + * atoms, so streaming never round-trips through the controller. */ export const ChatRoutesSurface = memo(function ChatRoutesSurface({ actions, maxVoiceRecordingSeconds @@ -114,19 +114,11 @@ export const ChatRoutesSurface = memo(function ChatRoutesSurface({ maxVoiceRecordingSeconds?: number }) { const activeGatewayProfile = useStore($activeGatewayProfile) + const gateway = useStore($gateway) const gatewayState = useStore($gatewayState) useContributions(ROUTES_AREA) const routeContributions = contributedRoutes() - // Recapture the live gateway instance whenever the connection state flips. - // getGateway reads a controller ref, so gatewayState is the intentional - // re-eval trigger (not a value the computation itself reads). - const gateway = useMemo( - () => actions.getGateway(), - // eslint-disable-next-line react-hooks/exhaustive-deps - [actions, gatewayState] - ) - const modelMenuContent = useMemo( () => gatewayState === 'open' ? ( From 4faf12b9ce0207213d98ad1bc2d6ceef2dfae59a Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 18:47:02 -0700 Subject: [PATCH 026/376] chore: map salvaged contributor emails (75day, KBANTH) --- contributors/emails/a9@A9deMac-mini.local | 1 + contributors/emails/kritcha.b+github@dgtpsn.com | 1 + 2 files changed, 2 insertions(+) create mode 100644 contributors/emails/a9@A9deMac-mini.local create mode 100644 contributors/emails/kritcha.b+github@dgtpsn.com diff --git a/contributors/emails/a9@A9deMac-mini.local b/contributors/emails/a9@A9deMac-mini.local new file mode 100644 index 0000000000000..44b4fd4c37e5e --- /dev/null +++ b/contributors/emails/a9@A9deMac-mini.local @@ -0,0 +1 @@ +75day diff --git a/contributors/emails/kritcha.b+github@dgtpsn.com b/contributors/emails/kritcha.b+github@dgtpsn.com new file mode 100644 index 0000000000000..2a2647a6fa1f7 --- /dev/null +++ b/contributors/emails/kritcha.b+github@dgtpsn.com @@ -0,0 +1 @@ +KBANTH From a351c17d4abe379c176553cf3ed454e3be7d6838 Mon Sep 17 00:00:00 2001 From: RelaxJonh <92573950+RelaxJonh@users.noreply.github.com> Date: Wed, 29 Jul 2026 11:51:02 +0700 Subject: [PATCH 027/376] fix(desktop): normalise timeout/error subagent statuses to terminal (#73728) The backend emits terminal statuses including 'timeout' and 'error' in subagent.complete payloads, but asStatus() only recognised 'completed', 'failed', 'interrupted', and 'queued'. Unrecognised values fell through to 'running', making timed-out subagents immortal in the active status stack. Fix: map timeout/error to 'failed', cancelled/canceled to 'interrupted'. Nonterminal unknown statuses still default to 'running' for forward compatibility. Fixes #73728 --- apps/desktop/src/store/subagents.test.ts | 37 ++++++++++++++++++++++++ apps/desktop/src/store/subagents.ts | 8 +++-- 2 files changed, 43 insertions(+), 2 deletions(-) diff --git a/apps/desktop/src/store/subagents.test.ts b/apps/desktop/src/store/subagents.test.ts index 254a7a22f4760..b6fb278bde219 100644 --- a/apps/desktop/src/store/subagents.test.ts +++ b/apps/desktop/src/store/subagents.test.ts @@ -190,4 +190,41 @@ describe('subagent store', () => { .sort() ).toEqual(['c', 'd']) }) + + // Regression test for #73728: backend terminal statuses like `timeout` and + // `error` were normalised to `running`, making timed-out subagents immortal + // in the active status stack. `cancelled`/`canceled` must also map to + // `interrupted`. + it('normalises backend terminal statuses to recognised SubagentStatus values', () => { + upsertSubagent('s1', { goal: 'a', status: 'running', subagent_id: 'a', task_index: 0 }) + upsertSubagent('s1', { goal: 'b', status: 'running', subagent_id: 'b', task_index: 1 }) + upsertSubagent('s1', { goal: 'c', status: 'running', subagent_id: 'c', task_index: 2 }) + upsertSubagent('s1', { goal: 'd', status: 'running', subagent_id: 'd', task_index: 3 }) + + // Emit terminal events with backend-native status strings + upsertSubagent('s1', { status: 'timeout', subagent_id: 'a', task_index: 0, summary: 'timed out' }, false, 'subagent.complete') + upsertSubagent('s1', { status: 'error', subagent_id: 'b', task_index: 1, summary: 'errored' }, false, 'subagent.complete') + upsertSubagent('s1', { status: 'cancelled', subagent_id: 'c', task_index: 2 }, false, 'subagent.complete') + upsertSubagent('s1', { status: 'canceled', subagent_id: 'd', task_index: 3 }, false, 'subagent.complete') + + const items = listFor('s1') + const byId = Object.fromEntries(items.map(i => [i.id, i])) + + // timeout → failed + expect(byId['a']?.status).toBe('failed') + expect(byId['a']?.currentTool).toBeUndefined() + + // error → failed + expect(byId['b']?.status).toBe('failed') + + // cancelled → interrupted + expect(byId['c']?.status).toBe('interrupted') + + // canceled → interrupted + expect(byId['d']?.status).toBe('interrupted') + + // All four are terminal — prune should remove them all + pruneFinishedSessionSubagents('s1') + expect(listFor('s1')).toHaveLength(0) + }) }) diff --git a/apps/desktop/src/store/subagents.ts b/apps/desktop/src/store/subagents.ts index 6127b15183c25..27402fe4493db 100644 --- a/apps/desktop/src/store/subagents.ts +++ b/apps/desktop/src/store/subagents.ts @@ -55,8 +55,12 @@ const str = (v: unknown) => (isStr(v) ? v : '') const num = (v: unknown) => (typeof v === 'number' && Number.isFinite(v) ? v : undefined) const strList = (v: unknown) => (Array.isArray(v) ? v.filter(isStr) : []) -const asStatus = (v: unknown): SubagentStatus => - v === 'completed' || v === 'failed' || v === 'interrupted' || v === 'queued' ? v : 'running' +const asStatus = (v: unknown): SubagentStatus => { + if (v === 'completed' || v === 'failed' || v === 'interrupted' || v === 'queued') return v + if (v === 'timeout' || v === 'error') return 'failed' + if (v === 'cancelled' || v === 'canceled') return 'interrupted' + return 'running' +} const compact = (text: string, max = PREVIEW_MAX) => { const line = text.replace(/\s+/g, ' ').trim() From 09b1726a1d3a246fdbc90790aeb369fa59cb42f8 Mon Sep 17 00:00:00 2001 From: David Metcalfe <80915+DavidMetcalfe@users.noreply.github.com> Date: Thu, 13 Aug 2026 11:17:33 -0700 Subject: [PATCH 028/376] fix(desktop): fail closed on unrecognized subagent.complete statuses; surface timeout reason MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Follow-up to the #73728 normalization fix (supersedes the event-agnostic fallback the maintainers flagged as incomplete): - subagent.complete is terminal by definition — an unrecognized status on it now renders as 'failed' instead of falling through to 'running', which would recreate the immortal false-active row for any future backend status (the keep_open request on #73859). - Live events keep the lenient 'running' fallback. - Synthesize a 'Timed out after Xs' summary from duration_seconds when the backend completes with status 'timeout' and no summary, so the failed row explains itself. - Tests: timeout reason synthesis + pruning, event-aware fail-closed vs lenient live fallback (13 total). --- apps/desktop/src/store/subagents.test.ts | 48 ++++++++++++++++++++++++ apps/desktop/src/store/subagents.ts | 40 ++++++++++++++++---- 2 files changed, 81 insertions(+), 7 deletions(-) diff --git a/apps/desktop/src/store/subagents.test.ts b/apps/desktop/src/store/subagents.test.ts index b6fb278bde219..57045fe5f0473 100644 --- a/apps/desktop/src/store/subagents.test.ts +++ b/apps/desktop/src/store/subagents.test.ts @@ -227,4 +227,52 @@ describe('subagent store', () => { pruneFinishedSessionSubagents('s1') expect(listFor('s1')).toHaveLength(0) }) + + // The backend completes subagents with status "timeout" (hard child timeout, + // delegation.child_timeout_seconds) and no summary — synthesize the reason + // so the failed row explains itself instead of rendering as a bare failure. + it('maps backend timeout status to a terminal failure with a synthesized reason', () => { + upsertSubagent('s1', { goal: 'scan files', status: 'running', subagent_id: 't1', task_index: 0 }) + upsertSubagent( + 's1', + { status: 'timeout', subagent_id: 't1', task_index: 0, duration_seconds: 612.3 }, + false, + 'subagent.complete' + ) + + const item = listFor('s1')[0] + expect(item?.status).toBe('failed') + expect(item?.durationSeconds).toBe(612.3) + expect(item?.summary).toBe('Timed out after 612.3s') + + // A timed-out row must be pruned at the next message.start boundary like + // any other finished row — it must not linger as a live spinner. + pruneFinishedSessionSubagents('s1') + expect(listFor('s1')).toHaveLength(0) + }) + + // Fail-closed guard: subagent.complete is terminal by definition, so an + // unrecognized status on it must not resurrect a row as 'running'. Live + // events keep the lenient fallback (a status we don't know is still active). + it('fails closed on unrecognized completion statuses but stays lenient for live events', () => { + upsertSubagent('s1', { goal: 'scan files', status: 'running', subagent_id: 'u1', task_index: 0 }) + upsertSubagent( + 's1', + { status: 'some_future_terminal_status', subagent_id: 'u1', task_index: 0 }, + false, + 'subagent.complete' + ) + expect(listFor('s1')[0]?.status).toBe('failed') + expect(activeSubagentCount(listFor('s1'))).toBe(0) + + upsertSubagent('s1', { goal: 'scan files', status: 'running', subagent_id: 'u2', task_index: 1 }) + upsertSubagent( + 's1', + { status: 'some_future_live_status', subagent_id: 'u2', task_index: 1, text: 'still working' }, + false, + 'subagent.progress' + ) + expect(listFor('s1')[1]?.status).toBe('running') + expect(activeSubagentCount(listFor('s1'))).toBe(1) + }) }) diff --git a/apps/desktop/src/store/subagents.ts b/apps/desktop/src/store/subagents.ts index 27402fe4493db..b962a64501cca 100644 --- a/apps/desktop/src/store/subagents.ts +++ b/apps/desktop/src/store/subagents.ts @@ -55,10 +55,27 @@ const str = (v: unknown) => (isStr(v) ? v : '') const num = (v: unknown) => (typeof v === 'number' && Number.isFinite(v) ? v : undefined) const strList = (v: unknown) => (Array.isArray(v) ? v.filter(isStr) : []) -const asStatus = (v: unknown): SubagentStatus => { - if (v === 'completed' || v === 'failed' || v === 'interrupted' || v === 'queued') return v - if (v === 'timeout' || v === 'error') return 'failed' - if (v === 'cancelled' || v === 'canceled') return 'interrupted' +const asStatus = (v: unknown, terminalEvent = false): SubagentStatus => { + if (v === 'completed' || v === 'failed' || v === 'interrupted' || v === 'queued') { + return v + } + + if (v === 'timeout' || v === 'error') { + return 'failed' + } + + if (v === 'cancelled' || v === 'canceled') { + return 'interrupted' + } + + // Fail closed on completion: a subagent.complete event is terminal by + // definition, so an unrecognized status must render as a failure rather + // than leave a dead row spinning as 'running' forever. Live events keep + // the lenient 'running' fallback. + if (terminalEvent) { + return 'failed' + } + return 'running' } @@ -110,6 +127,15 @@ const appendStream = (stream: SubagentStreamEntry[], entry: SubagentStreamEntry) return [...stream, entry].slice(-MAX_STREAM) } +// The backend sends no summary on a hard child timeout (only a preview like +// "Timed out after 612.3s" + duration_seconds). Synthesize it so the terminal +// row explains why it failed instead of rendering as a bare failure. +const timeoutSummary = (payload: SubagentPayload): string => { + const seconds = num(payload.duration_seconds) + + return str(payload.status) === 'timeout' ? `Timed out after ${seconds ?? '?'}s` : '' +} + function streamFromPayload( payload: SubagentPayload, status: SubagentStatus, @@ -141,7 +167,7 @@ function streamFromPayload( out.push({ at, kind: 'thinking', text }) } - const summary = compact(str(payload.summary) || str(payload.text)) + const summary = compact(str(payload.summary) || str(payload.text) || timeoutSummary(payload)) if (TERMINAL.has(status) && summary) { out.push({ at, isError: status === 'failed', kind: 'summary', text: summary }) @@ -152,7 +178,7 @@ function streamFromPayload( function toProgress(payload: SubagentPayload, prev: SubagentProgress | undefined, eventType = ''): SubagentProgress { const at = Date.now() - const status = asStatus(payload.status) + const status = asStatus(payload.status, eventType === 'subagent.complete') const tool = str(payload.tool_name) const stream = streamFromPayload(payload, status, eventType, at).reduce(appendStream, prev?.stream ?? []) const filesRead = strList(payload.files_read) @@ -177,7 +203,7 @@ function toProgress(payload: SubagentPayload, prev: SubagentProgress | undefined filesRead: filesRead.length ? filesRead : (prev?.filesRead ?? []), filesWritten: filesWritten.length ? filesWritten : (prev?.filesWritten ?? []), stream, - summary: str(payload.summary) || prev?.summary, + summary: str(payload.summary) || prev?.summary || timeoutSummary(payload) || undefined, currentTool: TERMINAL.has(status) ? undefined : tool || prev?.currentTool } } From d3fee89993a87b045a381f9a8bd4b62e5d0e6df9 Mon Sep 17 00:00:00 2001 From: David Metcalfe <80915+DavidMetcalfe@users.noreply.github.com> Date: Thu, 13 Aug 2026 11:18:56 -0700 Subject: [PATCH 029/376] fix(desktop): prefer synthesized timeout summary over stale progress text Review feedback: prev?.summary could shadow the 'Timed out after Xs' reason when a live event had populated it. timeoutSummary() now wins for raw timeout status; add coverage for the missing-duration placeholder. --- apps/desktop/src/store/subagents.test.ts | 7 +++++++ apps/desktop/src/store/subagents.ts | 2 +- 2 files changed, 8 insertions(+), 1 deletion(-) diff --git a/apps/desktop/src/store/subagents.test.ts b/apps/desktop/src/store/subagents.test.ts index 57045fe5f0473..e486cbca0a464 100644 --- a/apps/desktop/src/store/subagents.test.ts +++ b/apps/desktop/src/store/subagents.test.ts @@ -251,6 +251,13 @@ describe('subagent store', () => { expect(listFor('s1')).toHaveLength(0) }) + it('falls back to a placeholder when timeout duration is missing', () => { + upsertSubagent('s1', { goal: 'scan files', status: 'running', subagent_id: 't2', task_index: 0 }) + upsertSubagent('s1', { status: 'timeout', subagent_id: 't2', task_index: 0 }, false, 'subagent.complete') + + expect(listFor('s1')[0]?.summary).toBe('Timed out after ?s') + }) + // Fail-closed guard: subagent.complete is terminal by definition, so an // unrecognized status on it must not resurrect a row as 'running'. Live // events keep the lenient fallback (a status we don't know is still active). diff --git a/apps/desktop/src/store/subagents.ts b/apps/desktop/src/store/subagents.ts index b962a64501cca..3f7b15bbce19f 100644 --- a/apps/desktop/src/store/subagents.ts +++ b/apps/desktop/src/store/subagents.ts @@ -203,7 +203,7 @@ function toProgress(payload: SubagentPayload, prev: SubagentProgress | undefined filesRead: filesRead.length ? filesRead : (prev?.filesRead ?? []), filesWritten: filesWritten.length ? filesWritten : (prev?.filesWritten ?? []), stream, - summary: str(payload.summary) || prev?.summary || timeoutSummary(payload) || undefined, + summary: str(payload.summary) || timeoutSummary(payload) || prev?.summary || undefined, currentTool: TERMINAL.has(status) ? undefined : tool || prev?.currentTool } } From a01c7bb43bb21e0a74121c6612c1c9afa13c9b7f Mon Sep 17 00:00:00 2001 From: Sebastian Mause Date: Fri, 14 Aug 2026 18:48:27 -0700 Subject: [PATCH 030/376] test(desktop): completion events with still-active payload statuses settle as failed MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Folded-in coverage from PR #85995 (smause): a subagent.complete event whose payload still says 'running' or 'queued' must settle the row as failed — the completion event itself is the source of truth that the child is done. --- apps/desktop/src/store/subagents.test.ts | 21 +++++++++++++++++++++ 1 file changed, 21 insertions(+) diff --git a/apps/desktop/src/store/subagents.test.ts b/apps/desktop/src/store/subagents.test.ts index e486cbca0a464..606597b3e8c6a 100644 --- a/apps/desktop/src/store/subagents.test.ts +++ b/apps/desktop/src/store/subagents.test.ts @@ -282,4 +282,25 @@ describe('subagent store', () => { expect(listFor('s1')[1]?.status).toBe('running') expect(activeSubagentCount(listFor('s1'))).toBe(1) }) + + // Folded in from PR #85995: a subagent.complete carrying a still-active + // payload status ('running'/'queued') must also settle as failed — the + // event itself is the source of truth that the child is done. + it.each(['running', 'queued'] as const)( + 'treats a completion event with %s payload status as terminal failure', + status => { + upsertSubagent( + 's1', + { goal: 'inconsistent completion', status: 'running', subagent_id: 'ic1', task_index: 0, tool_name: 'search_files' }, + true, + 'subagent.start' + ) + upsertSubagent('s1', { status, subagent_id: 'ic1', task_index: 0 }, false, 'subagent.complete') + + const items = listFor('s1') + expect(items[0]?.status).toBe('failed') + expect(items[0]?.currentTool).toBeUndefined() + expect(activeSubagentCount(items)).toBe(0) + } + ) }) From e3c3d0895dc02015153538aadbc34f928a01d1f8 Mon Sep 17 00:00:00 2001 From: Michael Gannotti Date: Fri, 14 Aug 2026 18:48:49 -0700 Subject: [PATCH 031/376] test(desktop): late progress events must not revive a timed-out subagent row MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Folded-in coverage from PR #80045 (gannotti, #80018): after a terminal subagent.complete, a stray late 'running' progress event must not restart the spinner — the upsert guard keeps the settled failed status. --- apps/desktop/src/store/subagents.test.ts | 15 +++++++++++++++ 1 file changed, 15 insertions(+) diff --git a/apps/desktop/src/store/subagents.test.ts b/apps/desktop/src/store/subagents.test.ts index 606597b3e8c6a..4922376ecc0ac 100644 --- a/apps/desktop/src/store/subagents.test.ts +++ b/apps/desktop/src/store/subagents.test.ts @@ -303,4 +303,19 @@ describe('subagent store', () => { expect(activeSubagentCount(items)).toBe(0) } ) + + // Folded in from PR #80045 (#80018): a late progress event must not revive + // the spinner after a terminal completion — the row stays settled. + it('does not regress to running when a late running event arrives after timeout', () => { + upsertSubagent('s1', { goal: 'task', status: 'running', subagent_id: 'late1', task_index: 0 }) + upsertSubagent( + 's1', + { goal: 'task', status: 'timeout', subagent_id: 'late1', summary: 'Timed out', task_index: 0 }, + true, + 'subagent.complete' + ) + upsertSubagent('s1', { goal: 'task', status: 'running', subagent_id: 'late1', task_index: 0, text: 'late' }) + + expect(listFor('s1')[0]?.status).toBe('failed') + }) }) From 1c80085f04cc4064ee61991fd0230c90a7f2031d Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 18:49:02 -0700 Subject: [PATCH 032/376] fix(desktop): widen terminal-status handling to sibling sites Two sibling sites still treated only 'completed'/'failed' as terminal: - delegate-model.ts settled result rows as 'completed' for ANY status other than 'failed', so a delegate result row with status 'timeout' or 'error' (the statuses tools/delegate_tool.py actually emits on child timeout or crash) rendered behind a green check. Settled rows now map ok/completed to completed and everything else to failed. - subagents.ts asStatus accepted a literal 'queued' payload status even on a subagent.complete event, leaving the row active forever. The fail-closed branch now runs before the queued fallback, so completion events always settle. --- .../assistant-ui/tool/delegate-model.test.ts | 19 +++++++++++++++++++ .../assistant-ui/tool/delegate-model.ts | 10 +++++++++- apps/desktop/src/store/subagents.ts | 10 +++++----- 3 files changed, 33 insertions(+), 6 deletions(-) diff --git a/apps/desktop/src/components/assistant-ui/tool/delegate-model.test.ts b/apps/desktop/src/components/assistant-ui/tool/delegate-model.test.ts index 975ad90627bc9..14882cdf73e07 100644 --- a/apps/desktop/src/components/assistant-ui/tool/delegate-model.test.ts +++ b/apps/desktop/src/components/assistant-ui/tool/delegate-model.test.ts @@ -53,6 +53,25 @@ describe('delegateRowsFromCall', () => { expect(rows[0]).toMatchObject({ activity: ['found it'], durationSeconds: 12, model: 'anthropic/claude-opus-5' }) }) + // #73728 / #85492: the delegate tool settles rows with 'ok', 'error' or + // 'timeout' — anything that is not a success must render as failed instead + // of hiding behind a green 'completed' check. + it('renders timeout/error settled results as failed, ok as completed', () => { + const rows = delegateRowsFromCall( + { tasks: [{ goal: 'A' }, { goal: 'B' }, { goal: 'C' }, { goal: 'D' }] }, + { + results: [ + { status: 'ok', summary: 'done' }, + { status: 'timeout', error: 'Timed out after 600s' }, + { status: 'error', error: 'boom' }, + { status: 'failure' } + ] + } + ) + + expect(rows.map(r => r.status)).toEqual(['completed', 'failed', 'failed', 'failed']) + }) + it('still lists a background dispatch whose goals only survive in the result', () => { expect(delegateRowsFromCall({}, { status: 'dispatched', goals: ['A', 'B'] }).map(r => r.goal)).toEqual(['A', 'B']) }) diff --git a/apps/desktop/src/components/assistant-ui/tool/delegate-model.ts b/apps/desktop/src/components/assistant-ui/tool/delegate-model.ts index 7f1c4b48d07e9..eff61ec17d45c 100644 --- a/apps/desktop/src/components/assistant-ui/tool/delegate-model.ts +++ b/apps/desktop/src/components/assistant-ui/tool/delegate-model.ts @@ -52,6 +52,14 @@ function resultRows(result: unknown): Record[] { return results.map(parseMaybeObject) } +// The delegate tool settles result rows with statuses like 'ok', 'error', +// 'timeout', 'failed'/'failure' (tools/delegate_tool.py). Anything that is +// not a success must render as failed — mapping unknown statuses to +// 'completed' hid timed-out children behind a green check (#73728, #85492). +function settledRowStatus(status: string): DelegateRowStatus { + return status === '' || status === 'ok' || status === 'completed' ? 'completed' : 'failed' +} + function dispatchedGoals(result: unknown): string[] { const record = parseMaybeObject(result) @@ -87,7 +95,7 @@ export function delegateRowsFromCall(args: unknown, result: unknown, toolCallId goal, id: `${toolCallId}:${index}`, model: entry ? field(entry, 'model') || undefined : undefined, - status: entry ? (field(entry, 'status') === 'failed' ? 'failed' : 'completed') : idle + status: entry ? settledRowStatus(field(entry, 'status')) : idle } }) } diff --git a/apps/desktop/src/store/subagents.ts b/apps/desktop/src/store/subagents.ts index 3f7b15bbce19f..7196f14a471e0 100644 --- a/apps/desktop/src/store/subagents.ts +++ b/apps/desktop/src/store/subagents.ts @@ -56,7 +56,7 @@ const num = (v: unknown) => (typeof v === 'number' && Number.isFinite(v) ? v : u const strList = (v: unknown) => (Array.isArray(v) ? v.filter(isStr) : []) const asStatus = (v: unknown, terminalEvent = false): SubagentStatus => { - if (v === 'completed' || v === 'failed' || v === 'interrupted' || v === 'queued') { + if (v === 'completed' || v === 'failed' || v === 'interrupted') { return v } @@ -69,14 +69,14 @@ const asStatus = (v: unknown, terminalEvent = false): SubagentStatus => { } // Fail closed on completion: a subagent.complete event is terminal by - // definition, so an unrecognized status must render as a failure rather - // than leave a dead row spinning as 'running' forever. Live events keep - // the lenient 'running' fallback. + // definition, so an unrecognized (or still-active 'queued'/'running') + // status must render as a failure rather than leave a dead row spinning + // as 'running' forever. Live events keep the lenient fallback. if (terminalEvent) { return 'failed' } - return 'running' + return v === 'queued' ? v : 'running' } const compact = (text: string, max = PREVIEW_MAX) => { From b9482484f7c20603c80e5ccaa561641f43fa3c38 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 18:49:18 -0700 Subject: [PATCH 033/376] chore: map contributor email for attribution audit --- contributors/emails/sebastian@mause.online | 1 + 1 file changed, 1 insertion(+) create mode 100644 contributors/emails/sebastian@mause.online diff --git a/contributors/emails/sebastian@mause.online b/contributors/emails/sebastian@mause.online new file mode 100644 index 0000000000000..12212b5e79bbf --- /dev/null +++ b/contributors/emails/sebastian@mause.online @@ -0,0 +1 @@ +smause From 3995fd434cb457c85dfacfeead04f69c65bad092 Mon Sep 17 00:00:00 2001 From: StanleyStetson Date: Wed, 12 Aug 2026 17:58:49 +0300 Subject: [PATCH 034/376] fix(tui_gateway): stop replaying live-turn user text after redirect A mid-turn correction (Desktop session.redirect / busy-input interrupt redirect) must not leave a server-queue self-copy of the live inflight user prompt. Otherwise post-turn _drain_queued_prompt restarts that original text as a fresh agent turn after Q completes (#84417). Scrub text-only self-duplicates of inflight_turn.user on successful redirect/steer, refuse admitting them in _enqueue_prompt, rewrite merged "{P}\n\n{Q}" slots to Q-only, bump _queued_prompt_generation on compression session rotation, and restore the claimed queue envelope when generation cancels mid-drain. Stabilize profile-scoped agent-build unit tests under CI load. Fixes #84417 --- tests/test_tui_gateway_queue_on_busy.py | 264 +++++++++++++++++++++++- tests/test_tui_gateway_server.py | 110 +++++++++- tui_gateway/methods_session.py | 7 + tui_gateway/server.py | 120 +++++++++++ 4 files changed, 494 insertions(+), 7 deletions(-) diff --git a/tests/test_tui_gateway_queue_on_busy.py b/tests/test_tui_gateway_queue_on_busy.py index e6780e2976400..33c0e5ab6f1ba 100644 --- a/tests/test_tui_gateway_queue_on_busy.py +++ b/tests/test_tui_gateway_queue_on_busy.py @@ -79,6 +79,237 @@ def test_busy_interrupt_mode_redirects_active_turn(monkeypatch): assert session.get("queued_prompt") is None +def test_successful_redirect_drops_queued_duplicate_of_inflight_user(monkeypatch): + """#84417: correcting a live turn must not re-fire the original prompt from queue. + + When the live turn's original user text is also sitting in the server queue + (e.g. a second prompt.submit of the same text while redirect was not yet + possible), a later successful redirect of a *new* correction Q must purge + that self-duplicate. Otherwise post-turn ``_drain_queued_prompt`` starts a + second agent turn with the old prompt P after Q has already been handled. + """ + monkeypatch.setattr(server, "_load_busy_input_mode", lambda: "interrupt") + agent = types.SimpleNamespace( + _supports_active_turn_redirect=True, + redirect=lambda text: True, + interrupt=lambda *a, **k: (_ for _ in ()).throw( + AssertionError("redirect must not hard-interrupt") + ), + ) + session = _session(agent=agent, running=True) + original = "deepseek released a new flash model — I changed all settings to flash" + session["inflight_turn"] = { + "user": original, + "assistant": "partial", + "streaming": True, + "error": "", + } + # Stale self-duplicate of the live turn (would re-fire after settle). + session["queued_prompt"] = {"text": original, "transport": "ws-1"} + session["queued_prompts"] = [ + {"text": original, "transport": "ws-1"}, + {"text": "unrelated later task", "transport": "ws-1"}, + ] + + resp = server._handle_busy_submit( + "r1", "sid", session, "what about the pricing instead?", "ws-1" + ) + + assert resp["result"]["status"] == "redirected" + # Self-duplicates of the live original must be gone. + assert session.get("queued_prompt") == { + "text": "unrelated later task", + "transport": "ws-1", + } + assert not session.get("queued_prompts") + + +def test_successful_redirect_preserves_unrelated_queued_followups(monkeypatch): + """A legitimate next-turn queue entry must survive a mid-turn redirect.""" + monkeypatch.setattr(server, "_load_busy_input_mode", lambda: "interrupt") + agent = types.SimpleNamespace( + _supports_active_turn_redirect=True, + redirect=lambda text: True, + interrupt=lambda *a, **k: (_ for _ in ()).throw( + AssertionError("redirect must not hard-interrupt") + ), + ) + session = _session(agent=agent, running=True) + session["inflight_turn"] = { + "user": "live turn P", + "assistant": "", + "streaming": True, + "error": "", + } + session["queued_prompt"] = {"text": "run this after", "transport": "ws-1"} + + resp = server._handle_busy_submit("r1", "sid", session, "correction Q", "ws-1") + + assert resp["result"]["status"] == "redirected" + assert session.get("queued_prompt") == { + "text": "run this after", + "transport": "ws-1", + } + + +def test_enqueue_skips_text_duplicate_of_inflight_user(): + """#84417 defense: do not admit a self-duplicate of the live user prompt.""" + session = _session() + session["inflight_turn"] = { + "user": "live turn P", + "assistant": "", + "streaming": True, + "error": "", + } + + server._enqueue_prompt(session, "live turn P", "ws-1") + assert session.get("queued_prompt") is None + + server._enqueue_prompt(session, "different follow-up", "ws-1") + assert session["queued_prompt"] == { + "text": "different follow-up", + "transport": "ws-1", + } + + +def test_enqueue_followup_does_not_merge_stale_inflight_self_duplicate(): + """#84417: scrub P before merging so drain cannot re-fire ``P\\n\\nQ``.""" + session = _session() + session["inflight_turn"] = { + "user": "P", + "assistant": "", + "streaming": True, + "error": "", + } + # Pre-existing stale self-duplicate (e.g. admitted before inflight was set). + session["queued_prompt"] = {"text": "P", "transport": "ws-1"} + + server._enqueue_prompt(session, "Q", "ws-1") + + assert session.get("queued_prompt") == {"text": "Q", "transport": "ws-1"} + assert not session.get("queued_prompts") + + +def test_drop_rewrites_merged_inflight_prefix_to_followup_only(): + """Already-merged ``P\\n\\nQ`` slots keep Q and drop the live original.""" + session = _session() + session["inflight_turn"] = { + "user": "P", + "assistant": "", + "streaming": True, + "error": "", + } + session["queued_prompt"] = {"text": "P\n\nQ", "transport": "ws-1"} + + server._drop_queued_duplicates_of_inflight_user(session) + + assert session.get("queued_prompt") == {"text": "Q", "transport": "ws-1"} + + +def test_hard_interrupt_queue_path_scrubs_stale_inflight_self_duplicate(monkeypatch): + """#84417: interrupt+queue of Q must not leave P ahead of Q in the FIFO.""" + monkeypatch.setattr(server, "_load_busy_input_mode", lambda: "interrupt") + interrupts = [] + agent = types.SimpleNamespace( + _supports_active_turn_redirect=True, + redirect=lambda text: False, # force hard-interrupt fallback + interrupt=lambda *a, **k: interrupts.append(True), + ) + session = _session(agent=agent, running=True) + session["inflight_turn"] = { + "user": "P", + "assistant": "", + "streaming": True, + "error": "", + } + session["queued_prompt"] = {"text": "P", "transport": "ws-1"} + + resp = server._handle_busy_submit("r1", "sid", session, "Q", "ws-1") + + assert resp["result"]["status"] == "queued" + assert session.get("queued_prompt") == {"text": "Q", "transport": "ws-1"} + assert not session.get("queued_prompts") + # Interrupt is async-threaded; policy still enqueued Q after scrubbing P. + + +def test_redirect_then_drain_does_not_re_fire_original_p(monkeypatch): + """#84417 drain-level: after redirect(Q), settle must not start a second P.""" + monkeypatch.setattr(server, "_load_busy_input_mode", lambda: "interrupt") + fired = [] + agent = types.SimpleNamespace( + _supports_active_turn_redirect=True, + redirect=lambda text: True, + interrupt=lambda *a, **k: (_ for _ in ()).throw( + AssertionError("redirect must not hard-interrupt") + ), + ) + session = _session(agent=agent, running=True) + session["inflight_turn"] = { + "user": "P", + "assistant": "partial", + "streaming": True, + "error": "", + } + session["queued_prompt"] = {"text": "P", "transport": "ws-1"} + + resp = server._handle_busy_submit("r1", "sid", session, "Q", "ws-1") + assert resp["result"]["status"] == "redirected" + assert session.get("queued_prompt") is None + + # Turn settles (running cleared in finally) — drain must be a no-op. + session["running"] = False + monkeypatch.setattr( + server, + "_run_prompt_submit", + lambda rid, sid, session, text, **kwargs: fired.append(text), + ) + monkeypatch.setattr(server, "_session_uses_compute_host", lambda _s: False) + + assert server._drain_queued_prompt("r2", "sid", session) is False + assert fired == [] + + +def test_compress_session_rotation_bumps_queued_prompt_generation(monkeypatch): + """#84417 belt: rotation invalidates in-flight drain claims on the parent key. + + Queue *contents* survive (a legitimate follow-up must still run after + compression); only the generation counter advances so a drain that claimed + under the pre-rotation key cannot dispatch after re-anchor. + """ + monkeypatch.setattr(server, "_transfer_active_session_slot", lambda *a, **k: True) + monkeypatch.setattr(server, "_restart_slash_worker", lambda *a, **k: None) + agent = types.SimpleNamespace(session_id="child-after-rotation") + session = _session(agent=agent, session_key="parent-before-rotation") + session["_queued_prompt_generation"] = 3 + session["queued_prompt"] = {"text": "run after compress", "transport": "ws-1"} + + server._sync_session_key_after_compress("sid", session, clear_pending_title=False) + + assert session["session_key"] == "child-after-rotation" + assert session["_queued_prompt_generation"] == 4 + # Follow-up kept — only the claim generation bumped. + assert session["queued_prompt"] == { + "text": "run after compress", + "transport": "ws-1", + } + + +def test_compress_no_rotation_does_not_bump_queue_generation(monkeypatch): + """No-op when agent.session_id already matches session_key.""" + monkeypatch.setattr( + server, + "_transfer_active_session_slot", + lambda *a, **k: (_ for _ in ()).throw(AssertionError("no transfer")), + ) + agent = types.SimpleNamespace(session_id="same-key") + session = _session(agent=agent, session_key="same-key") + session["_queued_prompt_generation"] = 2 + + server._sync_session_key_after_compress("sid", session) + + assert session["_queued_prompt_generation"] == 2 + + @@ -299,7 +530,36 @@ def _boom(*a, **k): def test_drain_does_not_dispatch_a_prompt_cancelled_after_claim(monkeypatch): - session = _session(queued_prompt={"text": "B", "transport": None}) + """Generation cancel aborts dispatch but must restore the claimed head. + + Compress re-anchor / Stop bump generation between claim and check. Dropping + the envelope would silently lose a legitimate follow-up (#84417 belt). + """ + session = _session( + queued_prompt={"text": "B", "transport": "ws-1"}, + queued_prompts=[{"text": "C", "transport": "ws-1"}], + ) + monkeypatch.setattr( + server, + "_session_uses_compute_host", + lambda _session: session.__setitem__("_queued_prompt_generation", 1) or False, + ) + monkeypatch.setattr( + server, + "_run_prompt_submit", + lambda *args, **kwargs: (_ for _ in ()).throw(AssertionError("must not dispatch")), + ) + + assert server._drain_queued_prompt("r1", "sid", session) is True + assert session["running"] is False + # Claimed B restored first; C that advanced into the slot is behind it. + assert session.get("queued_prompt") == {"text": "B", "transport": "ws-1"} + assert session.get("queued_prompts") == [{"text": "C", "transport": "ws-1"}] + + +def test_drain_restores_claimed_prompt_when_generation_bumps_mid_claim(monkeypatch): + """Single-item queue: generation cancel must not empty the queue.""" + session = _session(queued_prompt={"text": "follow-up Q", "transport": None}) monkeypatch.setattr( server, "_session_uses_compute_host", @@ -313,6 +573,8 @@ def test_drain_does_not_dispatch_a_prompt_cancelled_after_claim(monkeypatch): assert server._drain_queued_prompt("r1", "sid", session) is True assert session["running"] is False + assert session.get("queued_prompt") == {"text": "follow-up Q", "transport": None} + assert not session.get("queued_prompts") def test_drain_does_not_clear_stop_after_its_final_generation_check(monkeypatch): diff --git a/tests/test_tui_gateway_server.py b/tests/test_tui_gateway_server.py index 6b1f11150eacb..25a29a4d06b28 100644 --- a/tests/test_tui_gateway_server.py +++ b/tests/test_tui_gateway_server.py @@ -695,6 +695,7 @@ def test_profile_scoped_agent_build_starts_mcp_discovery_in_profile_home( ): """Agent construction must start MCP discovery under the selected profile.""" import threading + import uuid from hermes_constants import get_hermes_home @@ -720,19 +721,26 @@ def test_profile_scoped_agent_build_starts_mcp_discovery_in_profile_home( monkeypatch.setattr(server, "_SlashWorker", lambda *args: None) monkeypatch.setattr(server, "_attach_worker", lambda *args: None) monkeypatch.setattr(server, "_config_model_target", lambda: ("", "")) + # CI runs this huge file serially under load; a prior session's _build can + # still be finishing (session.info emit) when the next test starts, so a + # 2s Event wait flakes. Unique sid + longer bound; still fail closed. + monkeypatch.setattr(server, "_start_notification_poller", lambda *a, **k: None) + monkeypatch.setattr(server, "_schedule_mcp_late_refresh", lambda *a, **k: None) + monkeypatch.setattr(server, "_emit", lambda *a, **k: None) ready = threading.Event() - sid = "test-sid" + sid = f"test-sid-{uuid.uuid4().hex[:8]}" session = { "agent_ready": ready, - "session_key": "test-key", + "session_key": f"test-key-{uuid.uuid4().hex[:8]}", "profile_home": str(profile_home), } server._sessions[sid] = session try: server._start_agent_build(sid, session) - assert built.wait(timeout=2) + assert built.wait(timeout=15), "agent build thread never called _make_agent" + assert ready.wait(timeout=5), "agent_ready never set after build" finally: server._sessions.pop(sid, None) @@ -747,6 +755,7 @@ def test_profile_scoped_agent_build_installs_secret_scope(monkeypatch, tmp_path) .env (#67605 item 2). """ import threading + import uuid from agent.secret_scope import current_secret_scope @@ -775,19 +784,25 @@ def _fake_make_agent(*args, **kwargs): monkeypatch.setattr(server, "_SlashWorker", lambda *args: None) monkeypatch.setattr(server, "_attach_worker", lambda *args: None) monkeypatch.setattr(server, "_config_model_target", lambda: ("", "")) + # Same CI flake class as the MCP profile-home test: bound wait + less work + # on the build thread (no poller / late MCP refresh / session.info emit). + monkeypatch.setattr(server, "_start_notification_poller", lambda *a, **k: None) + monkeypatch.setattr(server, "_schedule_mcp_late_refresh", lambda *a, **k: None) + monkeypatch.setattr(server, "_emit", lambda *a, **k: None) ready = threading.Event() - sid = "test-secret-sid" + sid = f"test-secret-sid-{uuid.uuid4().hex[:8]}" session = { "agent_ready": ready, - "session_key": "test-secret-key", + "session_key": f"test-secret-key-{uuid.uuid4().hex[:8]}", "profile_home": str(profile_home), } server._sessions[sid] = session try: server._start_agent_build(sid, session) - assert built.wait(timeout=2) + assert built.wait(timeout=15), "agent build thread never called _make_agent" + assert ready.wait(timeout=5), "agent_ready never set after build" finally: server._sessions.pop(sid, None) @@ -9388,6 +9403,89 @@ def test_session_redirect_calls_capable_core_agent(monkeypatch): assert before is None or session["last_active"] >= before +def test_session_redirect_rpc_drops_queued_duplicate_of_inflight_user(): + """#84417: Desktop ``session.redirect`` must purge stale self-duplicates. + + Production path: renderer steers via ``session.redirect`` (not + ``prompt.submit``). A self-copy of the live original user text already in + the server queue must not survive a successful redirect — otherwise + post-turn ``_drain_queued_prompt`` restarts prompt P after Q is handled. + Unrelated next-turn envelopes stay. + """ + original = "deepseek released a new flash model — I changed all settings to flash" + agent = types.SimpleNamespace( + _supports_active_turn_redirect=True, + redirect=lambda text: True, + ) + session = _session(agent=agent, running=True) + session["inflight_turn"] = { + "user": original, + "assistant": "partial", + "streaming": True, + "error": "", + } + session["queued_prompt"] = {"text": original, "transport": "ws-1"} + session["queued_prompts"] = [ + {"text": original, "transport": "ws-1"}, + {"text": "unrelated later task", "transport": "ws-1"}, + ] + server._sessions["sid"] = session + try: + resp = server.handle_request( + { + "id": "1", + "method": "session.redirect", + "params": { + "session_id": "sid", + "text": "what about the pricing instead?", + }, + } + ) + finally: + server._sessions.pop("sid", None) + + assert resp["result"]["status"] == "redirected" + assert session["inflight_turn"]["user"] == original + assert session["inflight_turn"]["corrections"] == [ + "what about the pricing instead?" + ] + # Self-duplicates of the live original are gone; legitimate follow-up kept. + assert session.get("queued_prompt") == { + "text": "unrelated later task", + "transport": "ws-1", + } + assert not session.get("queued_prompts") + + +def test_session_redirect_build_window_scrubs_stale_p_when_queuing_q(): + """#84417: build-window queue of Q must not leave P ahead of Q.""" + original = "live original P" + session = _session(running=True) + session["agent"] = None # async agent build window + session["inflight_turn"] = { + "user": original, + "assistant": "", + "streaming": True, + "error": "", + } + session["queued_prompt"] = {"text": original, "transport": "ws-1"} + server._sessions["sid"] = session + try: + resp = server.handle_request( + { + "id": "1", + "method": "session.redirect", + "params": {"session_id": "sid", "text": "correction Q"}, + } + ) + finally: + server._sessions.pop("sid", None) + + assert resp["result"] == {"status": "queued", "text": "correction Q"} + assert session["queued_prompt"]["text"] == "correction Q" + assert not session.get("queued_prompts") + + def test_session_redirect_records_correction_without_erasing_prompt(): """A redirect must not overwrite the turn's original user text. diff --git a/tui_gateway/methods_session.py b/tui_gateway/methods_session.py index d0dc6d88d701b..32b202f539258 100644 --- a/tui_gateway/methods_session.py +++ b/tui_gateway/methods_session.py @@ -3247,6 +3247,10 @@ def _(rid, params: dict) -> dict: # text has no user bubble — the "my message vanished on reload" loss. with session["history_lock"]: _record_inflight_correction(session, text) + # #84417: steer does not cancel the live original, but a server + # queue self-copy of that original must still not re-fire after + # settle (same class as redirect). + _drop_queued_duplicates_of_inflight_user(session) session["last_active"] = time.time() return _ok(rid, {"status": "queued" if accepted else "rejected", "text": text}) @@ -3283,6 +3287,9 @@ def _(rid, params: dict) -> dict: if accepted: with session["history_lock"]: _record_inflight_correction(session, text) + # #84417: purge server-queue self-duplicates of the live original + # so post-turn drain cannot restart the pre-correction prompt. + _drop_queued_duplicates_of_inflight_user(session) session["last_active"] = time.time() return _ok( rid, diff --git a/tui_gateway/server.py b/tui_gateway/server.py index 3d0b8f7129c22..c45100cf30ff4 100644 --- a/tui_gateway/server.py +++ b/tui_gateway/server.py @@ -5120,6 +5120,16 @@ def _sync_session_key_after_compress( # don't keep targeting the ended row. session["session_key"] = new_session_id + # #84417 (belt): invalidate any in-flight ``_drain_queued_prompt`` claim + # that captured generation under the pre-rotation session_key. A raced + # drain must not dispatch on the continuation with a stale claim; the + # claimed envelope is restored to the queue (see ``_drain_queued_prompt``) + # so legitimate follow-ups still survive. Complements self-duplicate + # scrubbing on redirect. + session["_queued_prompt_generation"] = int( + session.get("_queued_prompt_generation", 0) + ) + 1 + if clear_pending_title: session["pending_title"] = None if restart_slash_worker: @@ -7675,6 +7685,20 @@ def _enqueue_prompt( sent it even if the session transport is rebound meanwhile. """ image_paths = list(image_paths or []) + # #84417: scrub any live-turn self-duplicates first so the consecutive-text + # merge below cannot glue "{original}\\n\\n{later}" and re-fire original + # on drain after a later correction settles. + _drop_queued_duplicates_of_inflight_user(session) + # Never queue a text-only self-copy of the live inflight user prompt. The + # live turn already owns that text; draining it after settle would restart + # the same user turn as a fresh agent invocation. + if not image_paths and isinstance(text, str): + turn = session.get("inflight_turn") + original = ( + str(turn.get("user") or "").strip() if isinstance(turn, dict) else "" + ) + if original and text.strip() == original: + return queued = {"text": text, "transport": transport} if image_paths: queued["image_paths"] = image_paths @@ -7696,6 +7720,82 @@ def _enqueue_prompt( session["queued_prompt"] = queued +def _sanitize_queued_entry_vs_inflight_user( + entry: Any, original: str +) -> dict | None: + """Drop or rewrite a queue envelope that re-carries the live user text. + + Returns ``None`` to drop the envelope, or a (possibly rewritten) dict to + keep. Text-only self-duplicates of ``original`` are dropped. A merged + slot ``"{original}\\n\\n{later}"`` (from ``_enqueue_prompt``'s consecutive + text merge) is rewritten to just ``later`` so a later correction is not + lost and the original is not re-fired (#84417). Image-bearing envelopes + are left alone — their chronology/ownership is load-bearing. + """ + if not original or not isinstance(entry, dict): + return entry if isinstance(entry, dict) else None + if entry.get("image_paths"): + return entry + text = entry.get("text") + if not isinstance(text, str): + return entry + stripped = text.strip() + if not stripped: + return None + if stripped == original: + return None + # Lossless text-merge glued the live original onto a later follow-up. + for sep in ("\n\n", "\n"): + prefix = original + sep + if text.startswith(prefix): + rest = text[len(prefix) :].strip() + if not rest or rest == original: + return None + cleaned = dict(entry) + cleaned["text"] = rest + return cleaned + return entry + + +def _drop_queued_duplicates_of_inflight_user(session: dict) -> None: + """Remove server-queue copies of the live turn's original user text. + + A mid-turn ``prompt.submit`` of the same text can land in + ``queued_prompt`` when redirect is not yet available (model not active, + build window, tool boundary). If the user then corrects the turn with a + different prompt via redirect, that stale self-duplicate must not + ``_drain_queued_prompt`` after the redirected turn completes — otherwise + the original prompt restarts as a fresh agent turn (#84417). + + Unrelated follow-ups (different text, image-bearing envelopes) stay. + Merged ``original + later`` slots are rewritten to ``later`` only. + """ + turn = session.get("inflight_turn") + if not isinstance(turn, dict): + return + original = str(turn.get("user") or "").strip() + if not original: + return + + head = session.get("queued_prompt") + rest = list(session.get("queued_prompts") or []) + kept: list[dict] = [] + for entry in ([head] if head else []) + rest: + cleaned = _sanitize_queued_entry_vs_inflight_user(entry, original) + if cleaned is not None: + kept.append(cleaned) + + if not kept: + session["queued_prompt"] = None + session.pop("queued_prompts", None) + return + session["queued_prompt"] = kept[0] + if len(kept) > 1: + session["queued_prompts"] = kept[1:] + else: + session.pop("queued_prompts", None) + + def _interrupt_busy_session(sid: str, session: dict, agent: Any) -> None: """Interrupt a busy turn without blocking the RPC reader or session lock. @@ -7774,6 +7874,8 @@ def _handle_busy_submit( try: if agent.steer(plain_text): with session["history_lock"]: + _record_inflight_correction(session, plain_text) + _drop_queued_duplicates_of_inflight_user(session) session["last_active"] = time.time() return _ok(rid, {"status": "steered"}) except Exception: @@ -7793,6 +7895,9 @@ def _handle_busy_submit( if agent.redirect(plain_text): with session["history_lock"]: _record_inflight_correction(session, plain_text) + # #84417: do not re-fire the live turn's original user text + # from a stale server-queue self-duplicate after settle. + _drop_queued_duplicates_of_inflight_user(session) session["last_active"] = time.time() return _ok(rid, {"status": "redirected"}) except Exception: @@ -7837,6 +7942,21 @@ def _drain_queued_prompt(rid, sid: str, session: dict) -> bool: use_compute_host = _session_uses_compute_host(session) with session["history_lock"]: if int(session.get("_queued_prompt_generation", 0)) != queue_generation: + # Generation cancelled the claim (Stop, compress re-anchor, …). + # Do not dispatch — but put the claimed envelope back so a + # legitimate follow-up is not silently dropped. Order: claimed + # head first, then whatever advanced into the slot while we held + # the claim (#84417 belt accuracy). + rest: list = [] + advanced = session.get("queued_prompt") + if advanced: + rest.append(advanced) + rest.extend(session.get("queued_prompts") or []) + session["queued_prompt"] = queued + if rest: + session["queued_prompts"] = rest + else: + session.pop("queued_prompts", None) session["running"] = False return True dispatch_failed = False From 4ac12d53fb73ea6d8e6b8d2a484cbdc7369b9746 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 18:53:42 -0700 Subject: [PATCH 035/376] fix(tui_gateway): keep steer-mode fall-through bursts queued instead of interrupting Desktop/TUI busy-input `steer` mode escalated any fall-through message (steer() rejected, raised, or a non-steerable multimodal payload) into a hard interrupt of the live turn. AIAgent.interrupt() also clears the pending steer buffer, so a burst of user messages sent while the agent was busy could be silently destroyed: earlier successfully-steered messages were dropped from the buffer and the live turn was killed. Steer-mode fall-throughs now keep pure queue semantics: preserved FIFO in queued_prompt/queued_prompts and drained on turn end, per the existing steer contract. Only explicit `interrupt` mode still fires _interrupt_busy_session. No synthetic user messages are injected mid-loop; accepted steers continue through the sanctioned OOB steer-marker path. Regression tests cover: rejected steer queues without interrupting, steer exception falls back to queue, multimodal payload queues, a mixed burst preserves accepted steers plus the queued fall-through, and a fall-through burst drains all texts FIFO after turn end. Fixes #86134 --- tests/test_tui_gateway_queue_on_busy.py | 142 ++++++++++++++++++++++++ tui_gateway/server.py | 11 +- 2 files changed, 152 insertions(+), 1 deletion(-) diff --git a/tests/test_tui_gateway_queue_on_busy.py b/tests/test_tui_gateway_queue_on_busy.py index 33c0e5ab6f1ba..4b42a2d6de909 100644 --- a/tests/test_tui_gateway_queue_on_busy.py +++ b/tests/test_tui_gateway_queue_on_busy.py @@ -357,6 +357,148 @@ def test_busy_steer_mode_injects_when_accepted(monkeypatch): assert session.get("queued_prompt") is None +# ── steer-mode burst preservation (#86134) ───────────────────────────────── + +def test_busy_steer_fallthrough_queues_without_interrupting(monkeypatch): + """A steer-mode fall-through must keep queue semantics, never interrupt. + + #86134: ``AIAgent.interrupt()`` drops the pending steer buffer, so a hard + interrupt fired for a fall-through message destroyed the earlier + (successfully steered) messages of a burst AND killed the live turn. + """ + monkeypatch.setattr(server, "_load_busy_input_mode", lambda: "steer") + interrupted = threading.Event() + agent = types.SimpleNamespace( + steer=lambda text: False, # steer rejected → falls through to queue + interrupt=lambda *a, **k: interrupted.set(), + ) + session = _session(agent=agent, running=True) + + resp = server._handle_busy_submit("r1", "sid", session, "follow-up", "ws-1") + + assert resp["result"]["status"] == "queued" + assert session["queued_prompt"]["text"] == "follow-up" + # _interrupt_busy_session runs on a worker thread — give it a beat. + assert not interrupted.wait(0.2), "steer-mode fall-through must not hard-interrupt" + + +def test_busy_steer_exception_falls_back_to_queue_without_interrupting(monkeypatch): + monkeypatch.setattr(server, "_load_busy_input_mode", lambda: "steer") + interrupted = threading.Event() + agent = types.SimpleNamespace( + steer=lambda text: (_ for _ in ()).throw(RuntimeError("boom")), + interrupt=lambda *a, **k: interrupted.set(), + ) + session = _session(agent=agent, running=True) + + resp = server._handle_busy_submit("r1", "sid", session, "still here?", "ws-1") + + assert resp["result"]["status"] == "queued" + assert session["queued_prompt"]["text"] == "still here?" + assert not interrupted.wait(0.2), "steer failure must not escalate to interrupt" + + +def test_busy_steer_mode_multimodal_payload_queues_without_interrupting(monkeypatch): + """Image-bearing payloads are not steerable; they must queue, not kill.""" + monkeypatch.setattr(server, "_load_busy_input_mode", lambda: "steer") + rich = [ + {"type": "text", "text": "look at this"}, + {"type": "image_url", "image_url": {"url": "data:image/png;base64,abc"}}, + ] + steered = [] + interrupted = threading.Event() + agent = types.SimpleNamespace( + steer=lambda text: steered.append(text) or True, + interrupt=lambda *a, **k: interrupted.set(), + ) + session = _session(agent=agent, running=True) + + resp = server._handle_busy_submit("r1", "sid", session, rich, "ws-1") + + assert resp["result"]["status"] == "queued" + assert steered == [] + assert session["queued_prompt"]["text"] == rich + assert not interrupted.wait(0.2), "multimodal steer fall-through must not interrupt" + + +def test_busy_steer_burst_mix_preserves_accepted_steers_and_queue(monkeypatch): + """Burst of N messages: accepted steers survive a later fall-through. + + Models the real ``AIAgent`` contract: ``steer()`` concatenates into a + pending buffer that ``interrupt()`` would clear. A rejected message later + in the burst must not clear the buffer or stop the turn (#86134). + """ + monkeypatch.setattr(server, "_load_busy_input_mode", lambda: "steer") + + class _Agent: + def __init__(self): + self._pending_steer = None + self.accept = True + self.interrupted = threading.Event() + + def steer(self, text): + if not self.accept: + return False + self._pending_steer = ( + f"{self._pending_steer}\n{text}" if self._pending_steer else text + ) + return True + + def interrupt(self, *a, **k): + self._pending_steer = None # what the real interrupt() does + self.interrupted.set() + + agent = _Agent() + session = _session(agent=agent, running=True) + session["inflight_turn"] = {"user": "original ask"} + + r1 = server._handle_busy_submit("r1", "sid", session, "first note", "ws-1") + r2 = server._handle_busy_submit("r2", "sid", session, "second note", "ws-1") + agent.accept = False # third message loses the steer race + r3 = server._handle_busy_submit("r3", "sid", session, "third note", "ws-1") + + assert r1["result"]["status"] == "steered" + assert r2["result"]["status"] == "steered" + assert r3["result"]["status"] == "queued" + # No hard interrupt fired for the fall-through message... + assert not agent.interrupted.wait(0.2), "burst fall-through must not hard-interrupt" + # ...so earlier steers are preserved, distinct, in order. + assert agent._pending_steer == "first note\nsecond note" + # Fall-through preserved for the turn-end drain. + assert session["queued_prompt"]["text"] == "third note" + + +def test_busy_steer_fallthrough_burst_drains_all_texts_fifo(monkeypatch): + """Every fall-through text of a burst reaches the model after turn end.""" + monkeypatch.setattr(server, "_load_busy_input_mode", lambda: "steer") + interrupted = threading.Event() + agent = types.SimpleNamespace( + steer=lambda text: False, + interrupt=lambda *a, **k: interrupted.set(), + ) + session = _session(agent=agent, running=True) + for text in ("msg A", "msg B", "msg C"): + resp = server._handle_busy_submit("r", "sid", session, text, "ws-1") + assert resp["result"]["status"] == "queued" + assert not interrupted.wait(0.2), "queue fall-through burst must not interrupt" + + dispatched = [] + monkeypatch.setattr( + server, + "_run_prompt_submit", + lambda rid, sid, _session, text, **kwargs: dispatched.append(text), + ) + session["running"] = False + while server._drain_queued_prompt("drain", "sid", session): + session["running"] = False + if not session.get("queued_prompt"): + break + + joined = "\n".join(str(t) for t in dispatched) + for text in ("msg A", "msg B", "msg C"): + assert text in joined, f"burst message dropped: {text!r}" + + diff --git a/tui_gateway/server.py b/tui_gateway/server.py index c45100cf30ff4..7e5fa9dfeffa9 100644 --- a/tui_gateway/server.py +++ b/tui_gateway/server.py @@ -7915,7 +7915,16 @@ def _handle_busy_submit( # Attachments need a separate model invocation. Queue them without # cancelling the active turn so the user gets both results in order. - if mode != "queue" and not image_paths: + # + # #86134: ``steer`` mode must NEVER escalate to a hard interrupt. A burst + # of user messages while the agent is busy can land as a mix of accepted + # steers (stashed in ``AIAgent._pending_steer``) and fall-through queue + # envelopes (payload not steerable, ``steer()`` rejected/raised). A hard + # interrupt here kills the live turn AND ``AIAgent.interrupt()`` drops + # the pending steer buffer — silently destroying the earlier messages of + # the burst. Steer-mode fall-throughs keep queue semantics: preserved + # FIFO in ``queued_prompt``/``queued_prompts`` and drained on turn end. + if mode == "interrupt" and not image_paths: _interrupt_busy_session(sid, session, agent) return _ok(rid, {"status": "queued"}) From 801fd0b3d8dabd6baf2ee8c8aa8eedffd01eac2c Mon Sep 17 00:00:00 2001 From: Olympusbuildz Date: Wed, 12 Aug 2026 09:10:22 -0700 Subject: [PATCH 036/376] fix(desktop): interrupt-first after Stop so edit/resend avoids session-busy Stop clears frontend busy immediately while the gateway may still wind down. Edit/restore then passed interruptFirst=false and raced 4009 session busy. Keep a short per-session cooldown after cancel so rewind still interrupt-first, and expire the submit-in-flight lock so a hung submit cannot block the session forever. Fixes #83855 Co-authored-by: Olympusbuildz Signed-off-by: Olympusbuildz --- .../src/app/chat/session-tile-actions.ts | 23 ++++- .../session/hooks/use-prompt-actions/index.ts | 23 ++++- .../hooks/use-prompt-actions/submit.ts | 8 +- .../hooks/use-prompt-actions/utils.test.ts | 79 +++++++++++++++- .../session/hooks/use-prompt-actions/utils.ts | 91 ++++++++++++++++++- 5 files changed, 211 insertions(+), 13 deletions(-) diff --git a/apps/desktop/src/app/chat/session-tile-actions.ts b/apps/desktop/src/app/chat/session-tile-actions.ts index 8a35d14124c16..48faa43cf2139 100644 --- a/apps/desktop/src/app/chat/session-tile-actions.ts +++ b/apps/desktop/src/app/chat/session-tile-actions.ts @@ -48,7 +48,11 @@ import { truncateSubmitParams } from '../session/hooks/use-prompt-actions/rewind' import { useSubmitPrompt } from '../session/hooks/use-prompt-actions/submit' -import { type SubmitTextOptions } from '../session/hooks/use-prompt-actions/utils' +import { + markSessionRecentlyInterrupted, + shouldInterruptBeforeRewind, + type SubmitTextOptions +} from '../session/hooks/use-prompt-actions/utils' import { upsertOptimisticSession } from '../session/hooks/use-session-actions/utils' import type { ComposerScope } from './composer/scope' @@ -258,6 +262,9 @@ export function useSessionTileActions({ runtimeId, scope, storedSessionId }: Ses const cancelRun = useCallback(async () => { const sessionId = runtimeIdRef.current + // Frontend busy clears immediately; gateway wind-down can lag (#83855). + markSessionRecentlyInterrupted(sessionId) + update(state => ({ ...state, messages: finalizeInterruptedMessages(state.messages, state.streamId), @@ -456,13 +463,16 @@ export function useSessionTileActions({ runtimeId, scope, storedSessionId }: Ses resetSessionBackground(sessionId) clearPreviewArtifacts(sessionId) - const wasBusy = readState()?.busy ?? false + const interruptFirst = shouldInterruptBeforeRewind({ + busy: readState()?.busy ?? false, + sessionId + }) update(state => applyRewindOptimistic(state, plan.sourceIndex)) try { applySurvivorRowIds( - await submitRewind(plan.text, plan.truncateOrdinal, wasBusy, plan.truncateMessageId, plan.truncateRowId) + await submitRewind(plan.text, plan.truncateOrdinal, interruptFirst, plan.truncateMessageId, plan.truncateRowId) ) } catch (err) { update(state => ({ ...state, busy: false, awaitingResponse: false, messages })) @@ -487,13 +497,16 @@ export function useSessionTileActions({ runtimeId, scope, storedSessionId }: Ses resetSessionBackground(sessionId) clearPreviewArtifacts(sessionId) - const wasBusy = readState()?.busy ?? false + const interruptFirst = shouldInterruptBeforeRewind({ + busy: readState()?.busy ?? false, + sessionId + }) update(state => applyRewindOptimistic(state, plan.sourceIndex, plan.editedMessage)) try { applySurvivorRowIds( - await submitRewind(plan.text, plan.truncateOrdinal, wasBusy, plan.truncateMessageId, plan.truncateRowId) + await submitRewind(plan.text, plan.truncateOrdinal, interruptFirst, plan.truncateMessageId, plan.truncateRowId) ) } catch (err) { update(state => ({ ...state, busy: false, awaitingResponse: false, messages })) diff --git a/apps/desktop/src/app/session/hooks/use-prompt-actions/index.ts b/apps/desktop/src/app/session/hooks/use-prompt-actions/index.ts index fdd32fdf7479b..709ea7026ddc1 100644 --- a/apps/desktop/src/app/session/hooks/use-prompt-actions/index.ts +++ b/apps/desktop/src/app/session/hooks/use-prompt-actions/index.ts @@ -70,8 +70,10 @@ import { friendlyRemoteAttachError, type GatewayRequest, inlineErrorMessage, + markSessionRecentlyInterrupted, readFileDataUrlForAttach, readImageForRemoteAttach, + shouldInterruptBeforeRewind, type SubmitTextOptions, withSessionNotFoundResume } from './utils' @@ -649,6 +651,10 @@ export function usePromptActions({ return } + // Frontend busy clears immediately; gateway wind-down can lag. Mark so a + // fast edit/resend still interrupt-first instead of racing 4009 (#83855). + markSessionRecentlyInterrupted(sessionId) + updateSessionState(sessionId, state => { const streamId = state.streamId const messages = finalizeInterruptedMessages(state.messages, streamId) @@ -908,6 +914,13 @@ export function usePromptActions({ resetSessionBackground(sessionId) clearPreviewArtifacts(sessionId) + // Capture before optimistic busy=true — otherwise interruptFirst is always + // true and idle restores wrongly interrupt (and Stop→edit misses cooldown). + const interruptFirst = shouldInterruptBeforeRewind({ + busy: busyRef.current || $busy.get(), + sessionId + }) + clearNotifications() setMutableRef(busyRef, true) setBusy(true) @@ -920,7 +933,7 @@ export function usePromptActions({ plan.text, plan.truncateOrdinal, plan.truncateMessageId, - busyRef.current || $busy.get(), + interruptFirst, plan.truncateRowId ) @@ -965,6 +978,12 @@ export function usePromptActions({ resetSessionBackground(sessionId) clearPreviewArtifacts(sessionId) + // Before optimistic busy=true — see restoreToMessage (#83855). + const interruptFirst = shouldInterruptBeforeRewind({ + busy: busyRef.current || $busy.get(), + sessionId + }) + clearNotifications() setMutableRef(busyRef, true) setBusy(true) @@ -977,7 +996,7 @@ export function usePromptActions({ plan.text, plan.truncateOrdinal, plan.truncateMessageId, - busyRef.current || $busy.get(), + interruptFirst, plan.truncateRowId ) diff --git a/apps/desktop/src/app/session/hooks/use-prompt-actions/submit.ts b/apps/desktop/src/app/session/hooks/use-prompt-actions/submit.ts index be6264748eaa5..422d79ef64ff7 100644 --- a/apps/desktop/src/app/session/hooks/use-prompt-actions/submit.ts +++ b/apps/desktop/src/app/session/hooks/use-prompt-actions/submit.ts @@ -38,12 +38,13 @@ import { resolveSessionProfile } from '../use-session-actions/utils' import { finalizeInterruptedMessages } from './rewind' import { - _submitInFlight, + acquireSubmitInFlight, type GatewayRequest, inlineErrorMessage, isProviderSetupError, isSessionBusyError, isTargetSessionBusy, + releaseSubmitInFlight, SessionRecoveryAborted, type SubmitTextOptions, withSessionBusyRetry, @@ -296,17 +297,16 @@ export function useSubmitPrompt(deps: SubmitPromptDeps) { // session switch; this per-session lock makes that safe. const submitLockKey = targetStoredSessionId || sessionId || startingActiveSessionId || '__pending_new__' - if (_submitInFlight.has(submitLockKey)) { + if (!acquireSubmitInFlight(submitLockKey)) { return false } - _submitInFlight.add(submitLockKey) let submitLockReleased = false const releaseSubmitLock = () => { if (!submitLockReleased) { submitLockReleased = true - _submitInFlight.delete(submitLockKey) + releaseSubmitInFlight(submitLockKey) } } diff --git a/apps/desktop/src/app/session/hooks/use-prompt-actions/utils.test.ts b/apps/desktop/src/app/session/hooks/use-prompt-actions/utils.test.ts index c2b1ac231c5ed..e00efecf66e35 100644 --- a/apps/desktop/src/app/session/hooks/use-prompt-actions/utils.test.ts +++ b/apps/desktop/src/app/session/hooks/use-prompt-actions/utils.test.ts @@ -1,11 +1,14 @@ import type { AppendMessage } from '@assistant-ui/react' -import { describe, expect, it, vi } from 'vitest' +import { afterEach, describe, expect, it, vi } from 'vitest' import type { ChatMessage } from '@/lib/chat-messages' import { + acquireSubmitInFlight, appendText, base64FromDataUrl, + clearSessionRecentlyInterrupted, + clearSubmitInFlight, friendlyRemoteAttachError, type GatewayRequest, imageFilenameFromPath, @@ -13,15 +16,89 @@ import { isSessionBusyError, isSessionIdCandidate, isSessionNotFoundError, + isSessionRecentlyInterrupted, + isSubmitInFlight, + markSessionRecentlyInterrupted, + RECENT_INTERRUPT_COOLDOWN_MS, readFileDataUrlForAttach, + releaseSubmitInFlight, renderRpcResult, SessionRecoveryAborted, + shouldInterruptBeforeRewind, slashStatusText, + SUBMIT_IN_FLIGHT_TTL_MS, visibleUserIndexAtOrdinal, visibleUserOrdinal, withSessionNotFoundResume } from './utils' +afterEach(() => { + clearSessionRecentlyInterrupted() + clearSubmitInFlight() +}) + +describe('recent interrupt cooldown', () => { + it('is true within the cooldown and false after expiry', () => { + const sessionId = 'sess-cooldown' + const t0 = 1_000_000 + + markSessionRecentlyInterrupted(sessionId, t0) + + expect(isSessionRecentlyInterrupted(sessionId, t0)).toBe(true) + expect(isSessionRecentlyInterrupted(sessionId, t0 + RECENT_INTERRUPT_COOLDOWN_MS - 1)).toBe(true) + expect(isSessionRecentlyInterrupted(sessionId, t0 + RECENT_INTERRUPT_COOLDOWN_MS)).toBe(false) + }) + + it('returns false after mark + elapsed past cooldown', () => { + const sessionId = 'sess-elapsed' + const t0 = 5_000_000 + + markSessionRecentlyInterrupted(sessionId, t0) + expect(isSessionRecentlyInterrupted(sessionId, t0 + RECENT_INTERRUPT_COOLDOWN_MS + 1)).toBe(false) + }) + + it('shouldInterruptBeforeRewind is true when recently interrupted even if not busy', () => { + const sessionId = 'sess-edit-after-stop' + const t0 = 9_000_000 + + markSessionRecentlyInterrupted(sessionId, t0) + + expect(shouldInterruptBeforeRewind({ busy: false, sessionId, now: t0 + 500 })).toBe(true) + expect(shouldInterruptBeforeRewind({ busy: false, sessionId, now: t0 + RECENT_INTERRUPT_COOLDOWN_MS + 1 })).toBe( + false + ) + }) + + it('shouldInterruptBeforeRewind stays false for idle sessions with no recent interrupt', () => { + expect(shouldInterruptBeforeRewind({ busy: false, sessionId: 'idle-sess' })).toBe(false) + expect(shouldInterruptBeforeRewind({ busy: true, sessionId: 'busy-sess' })).toBe(true) + }) +}) + +describe('submit in-flight TTL', () => { + it('blocks a second acquire while fresh and frees after TTL without explicit release', () => { + const key = 'lock-ttl' + const t0 = 2_000_000 + + expect(acquireSubmitInFlight(key, t0)).toBe(true) + expect(isSubmitInFlight(key, t0 + 1)).toBe(true) + expect(acquireSubmitInFlight(key, t0 + 1)).toBe(false) + + expect(isSubmitInFlight(key, t0 + SUBMIT_IN_FLIGHT_TTL_MS)).toBe(false) + expect(acquireSubmitInFlight(key, t0 + SUBMIT_IN_FLIGHT_TTL_MS)).toBe(true) + }) + + it('release clears the lock immediately', () => { + const key = 'lock-release' + const t0 = 3_000_000 + + expect(acquireSubmitInFlight(key, t0)).toBe(true) + releaseSubmitInFlight(key) + expect(isSubmitInFlight(key, t0 + 1)).toBe(false) + expect(acquireSubmitInFlight(key, t0 + 1)).toBe(true) + }) +}) + describe('isSessionIdCandidate', () => { it('accepts the timestamped and hex id forms', () => { expect(isSessionIdCandidate('20260101_120000_abc123')).toBe(true) diff --git a/apps/desktop/src/app/session/hooks/use-prompt-actions/utils.ts b/apps/desktop/src/app/session/hooks/use-prompt-actions/utils.ts index aba343355bc36..77227229e52ca 100644 --- a/apps/desktop/src/app/session/hooks/use-prompt-actions/utils.ts +++ b/apps/desktop/src/app/session/hooks/use-prompt-actions/utils.ts @@ -259,12 +259,101 @@ export async function withSessionBusyRetry(call: () => Promise): Promise() + +export function markSessionRecentlyInterrupted(sessionId: string, now = Date.now()): void { + if (!sessionId) { + return + } + + _recentlyInterruptedUntil.set(sessionId, now + RECENT_INTERRUPT_COOLDOWN_MS) +} + +export function isSessionRecentlyInterrupted(sessionId: string, now = Date.now()): boolean { + const until = _recentlyInterruptedUntil.get(sessionId) + + if (until === undefined) { + return false + } + + if (now >= until) { + _recentlyInterruptedUntil.delete(sessionId) + + return false + } + + return true +} + +export function clearSessionRecentlyInterrupted(sessionId?: string): void { + if (sessionId) { + _recentlyInterruptedUntil.delete(sessionId) + + return + } + + _recentlyInterruptedUntil.clear() +} + +/** Whether a rewind/edit should interrupt before submit — busy OR recent Stop. */ +export function shouldInterruptBeforeRewind(opts: { + busy: boolean + sessionId: string + now?: number +}): boolean { + return opts.busy || isSessionRecentlyInterrupted(opts.sessionId, opts.now) +} + // Hard guard: at most one prompt.submit in flight per session. Every submit // path — user Enter, queue drain, busy-retry, slash fallthrough — funnels // through submitPromptText. Without this, a stalled turn (e.g. a context-bloated // session whose first call hangs) let the SAME prompt launch several real turns // at once (the "message stacked 5×" bug). Keyed by stored/active session id. -export const _submitInFlight = new Set() +// Entries expire so a hung submit cannot permanently block the session (#83855). +export const SUBMIT_IN_FLIGHT_TTL_MS = 30_000 + +const _submitInFlightAt = new Map() + +export function isSubmitInFlight(key: string, now = Date.now()): boolean { + const acquiredAt = _submitInFlightAt.get(key) + + if (acquiredAt === undefined) { + return false + } + + if (now - acquiredAt >= SUBMIT_IN_FLIGHT_TTL_MS) { + _submitInFlightAt.delete(key) + + return false + } + + return true +} + +/** Returns true when the lock was acquired; false when another fresh hold blocks. */ +export function acquireSubmitInFlight(key: string, now = Date.now()): boolean { + if (isSubmitInFlight(key, now)) { + return false + } + + _submitInFlightAt.set(key, now) + + return true +} + +export function releaseSubmitInFlight(key: string): void { + _submitInFlightAt.delete(key) +} + +export function clearSubmitInFlight(): void { + _submitInFlightAt.clear() +} export function base64FromDataUrl(dataUrl: string): string { const comma = dataUrl.indexOf(',') From 33b39f3f4b7aeb15e7f641ff5f4dcd59a041fd31 Mon Sep 17 00:00:00 2001 From: "Simplicio, Wesley (ext)" Date: Mon, 10 Aug 2026 17:18:45 -0300 Subject: [PATCH 037/376] fix(desktop): prove submit target belongs to selected session from both directions (#65328) Fail closed on missing ownership cache entries and prove runtime ownership forward+reverse against runtimeIdByStoredSessionIdRef before prompt.submit. Thread the ownership cache through main wiring and session-tile submit. Adds regression tests for forward mismatch, reverse-only proof, positive map control, and cache-miss resume. Closes #65328. --- .../src/app/chat/session-tile-actions.ts | 6 + apps/desktop/src/app/contrib/wiring.tsx | 1 + .../hooks/use-prompt-actions/index.test.tsx | 217 ++++++++++++++++++ .../session/hooks/use-prompt-actions/index.ts | 3 + .../hooks/use-prompt-actions/submit.ts | 36 +++ 5 files changed, 263 insertions(+) diff --git a/apps/desktop/src/app/chat/session-tile-actions.ts b/apps/desktop/src/app/chat/session-tile-actions.ts index 48faa43cf2139..112a805f95084 100644 --- a/apps/desktop/src/app/chat/session-tile-actions.ts +++ b/apps/desktop/src/app/chat/session-tile-actions.ts @@ -111,6 +111,11 @@ export function useSessionTileActions({ runtimeId, scope, storedSessionId }: Ses runtimeIdRef.current = runtimeId const storedIdRef = useRef(storedSessionId) storedIdRef.current = storedSessionId + // A tile IS its session (see the comment on the useSubmitPrompt call below) + // A tile owns one stable stored/runtime pair, so seed the shared ownership + // cache explicitly rather than relying on the primary route cache. + const runtimeIdByStoredSessionIdRef = useRef(new Map([[storedSessionId, runtimeId]])) + runtimeIdByStoredSessionIdRef.current.set(storedSessionId, runtimeId) // Tile busy tracks the SESSION state, never the global $busy — and it must // read LIVE. A render-time snapshot goes stale (this hook's host doesn't @@ -223,6 +228,7 @@ export function useSessionTileActions({ runtimeId, scope, storedSessionId }: Ses // token is a stable constant (the guard never trips for a tile). getRouteToken: () => runtimeId, requestGateway, + runtimeIdByStoredSessionIdRef, // Tile ids are always bound before this hook mounts, so routed recovery is // unreachable here; keep the shared submit contract explicit. resumeStoredSession: () => undefined, diff --git a/apps/desktop/src/app/contrib/wiring.tsx b/apps/desktop/src/app/contrib/wiring.tsx index 7eeaeaa47978b..b670d831455eb 100644 --- a/apps/desktop/src/app/contrib/wiring.tsx +++ b/apps/desktop/src/app/contrib/wiring.tsx @@ -584,6 +584,7 @@ export function ContribWiring({ children }: { children: ReactNode }) { refreshSessions, requestGateway, resumeStoredSession: resumeSession, + runtimeIdByStoredSessionIdRef, selectedStoredSessionIdRef, startFreshSessionDraft, sttEnabled, diff --git a/apps/desktop/src/app/session/hooks/use-prompt-actions/index.test.tsx b/apps/desktop/src/app/session/hooks/use-prompt-actions/index.test.tsx index 7070f90d2bd99..2796d3edda78e 100644 --- a/apps/desktop/src/app/session/hooks/use-prompt-actions/index.test.tsx +++ b/apps/desktop/src/app/session/hooks/use-prompt-actions/index.test.tsx @@ -103,6 +103,7 @@ function Harness({ refreshSessions, requestGateway, resumeStoredSession, + runtimeIdByStoredSessionIdRef: runtimeIdByStoredSessionIdRefProp, seedMessages, selectedStoredSessionIdRef: selectedStoredSessionIdRefProp, storedSessionId, @@ -125,6 +126,7 @@ function Harness({ refreshSessions: () => Promise requestGateway: (method: string, params?: Record, timeoutMs?: number) => Promise resumeStoredSession?: (storedSessionId: string) => Promise | void + runtimeIdByStoredSessionIdRef?: MutableRefObject> seedMessages?: unknown[] selectedStoredSessionIdRef?: MutableRefObject storedSessionId?: null | string @@ -141,6 +143,16 @@ function Harness({ current: storedSessionId === undefined ? RUNTIME_SESSION_ID : storedSessionId } + const defaultStoredSessionId = storedSessionId === undefined ? RUNTIME_SESSION_ID : storedSessionId + const defaultRuntimeSessionId = activeSessionId === undefined ? RUNTIME_SESSION_ID : activeSessionId + const runtimeIdByStoredSessionIdRef: MutableRefObject> = + runtimeIdByStoredSessionIdRefProp ?? { + current: + defaultStoredSessionId && defaultRuntimeSessionId + ? new Map([[defaultStoredSessionId, defaultRuntimeSessionId]]) + : new Map() + } + const localBusyRef = busyRef ?? { current: false } const stateRef = useRef({ @@ -164,6 +176,7 @@ function Harness({ refreshSessions, requestGateway, resumeStoredSession: resumeStoredSession ?? (() => undefined), + runtimeIdByStoredSessionIdRef, selectedStoredSessionIdRef, startFreshSessionDraft: () => undefined, sttEnabled: false, @@ -3347,6 +3360,7 @@ describe('usePromptActions sleep/wake session recovery', () => { refreshSessions={async () => undefined} requestGateway={requestGateway} resumeStoredSession={resumeStoredSession} + runtimeIdByStoredSessionIdRef={{ current: new Map([[STORED_SESSION_ID, RECOVERED_SESSION_ID]]) }} selectedStoredSessionIdRef={selectedStoredSessionIdRef} storedSessionId={STORED_SESSION_ID} /> @@ -3378,6 +3392,7 @@ describe('usePromptActions sleep/wake session recovery', () => { refreshSessions={async () => undefined} requestGateway={requestGateway} resumeStoredSession={resumeStoredSession} + runtimeIdByStoredSessionIdRef={{ current: new Map([[STORED_SESSION_ID, RECOVERED_SESSION_ID]]) }} selectedStoredSessionIdRef={selectedStoredSessionIdRef} storedSessionId={STORED_SESSION_ID} /> @@ -4226,6 +4241,208 @@ describe('usePromptActions busy-gateway churn tolerance (#64327)', () => { }) }) +describe('usePromptActions submit entry-time runtime ownership proof (#64789/#65328)', () => { + const STORED_SESSION_B = 'stored-project-b' + const RUNTIME_SESSION_A = 'rt-session-a' + const RUNTIME_SESSION_B_RESUMED = 'rt-session-b-resumed' + + afterEach(() => { + cleanup() + vi.restoreAllMocks() + }) + + it('does not submit to runtime A when the cache proves B is bound to a different runtime (forward mismatch)', async () => { + const calls: { method: string; params?: Record }[] = [] + + const selectedStoredSessionIdRef: MutableRefObject = { current: STORED_SESSION_B } + const activeSessionIdRef: MutableRefObject = { current: RUNTIME_SESSION_A } + + const runtimeIdByStoredSessionIdRef: MutableRefObject> = { + current: new Map([[STORED_SESSION_B, 'rt-session-b-known']]) + } + + const requestGateway = vi.fn(async (method: string, params?: Record) => { + calls.push({ method, params }) + + if (method === 'session.resume') { + return { session_id: RUNTIME_SESSION_B_RESUMED } as never + } + + return {} as never + }) + + let handle: HarnessHandle | null = null + render( + (handle = h)} + refreshSessions={async () => undefined} + requestGateway={requestGateway} + runtimeIdByStoredSessionIdRef={runtimeIdByStoredSessionIdRef} + selectedStoredSessionIdRef={selectedStoredSessionIdRef} + storedSessionId={STORED_SESSION_B} + /> + ) + await waitFor(() => expect(handle).not.toBeNull()) + + await handle!.submitText('ordinary text for the selected project B session') + + expect(calls.find(c => c.method === 'session.resume')?.params).toEqual({ + session_id: STORED_SESSION_B, + source: 'desktop' + }) + expect( + calls.find(c => c.method === 'prompt.submit' && c.params?.session_id === RUNTIME_SESSION_A) + ).toBeUndefined() + expect( + calls.find(c => c.method === 'prompt.submit' && c.params?.session_id === RUNTIME_SESSION_B_RESUMED) + ).toBeDefined() + }) + + it('does not submit to runtime A when the cache proves A belongs to a DIFFERENT stored session (reverse proof, no forward entry for B)', async () => { + // The failure mode a one-directional (stored -> runtime) lookup misses: + // the cache has no entry for B at all (a forward miss looks like "no + // conflict"), but A is definitively known to belong to some OTHER + // stored session. A real-world trigger: the user is mid-conversation in + // an old session (whose runtime is A, cached under stored-project-old), + // then creates a fresh session B — if activeSessionIdRef hasn't been + // re-homed to B's own runtime yet by the time submit fires, A must not + // be accepted just because B itself was never cached. + const calls: { method: string; params?: Record }[] = [] + + const selectedStoredSessionIdRef: MutableRefObject = { current: STORED_SESSION_B } + const activeSessionIdRef: MutableRefObject = { current: RUNTIME_SESSION_A } + + const runtimeIdByStoredSessionIdRef: MutableRefObject> = { + current: new Map([['stored-project-old', RUNTIME_SESSION_A]]) + } + + const requestGateway = vi.fn(async (method: string, params?: Record) => { + calls.push({ method, params }) + + if (method === 'session.resume') { + return { session_id: RUNTIME_SESSION_B_RESUMED } as never + } + + return {} as never + }) + + let handle: HarnessHandle | null = null + render( + (handle = h)} + refreshSessions={async () => undefined} + requestGateway={requestGateway} + runtimeIdByStoredSessionIdRef={runtimeIdByStoredSessionIdRef} + selectedStoredSessionIdRef={selectedStoredSessionIdRef} + storedSessionId={STORED_SESSION_B} + /> + ) + await waitFor(() => expect(handle).not.toBeNull()) + + await handle!.submitText('ordinary text for the freshly created project B session') + + expect(calls.find(c => c.method === 'session.resume')?.params).toEqual({ + session_id: STORED_SESSION_B, + source: 'desktop' + }) + expect( + calls.find(c => c.method === 'prompt.submit' && c.params?.session_id === RUNTIME_SESSION_A) + ).toBeUndefined() + expect( + calls.find(c => c.method === 'prompt.submit' && c.params?.session_id === RUNTIME_SESSION_B_RESUMED) + ).toBeDefined() + }) + + it('still submits directly when the cache positively maps the selected session to the runtime', async () => { + // Direct submit is safe only when the cache explicitly proves the selected + // stored session owns the active runtime. + const calls: { method: string; params?: Record }[] = [] + + const selectedStoredSessionIdRef: MutableRefObject = { current: STORED_SESSION_B } + const activeSessionIdRef: MutableRefObject = { current: RUNTIME_SESSION_A } + const runtimeIdByStoredSessionIdRef: MutableRefObject> = { + current: new Map([[STORED_SESSION_B, RUNTIME_SESSION_A]]) + } + + const requestGateway = vi.fn(async (method: string, params?: Record) => { + calls.push({ method, params }) + + return {} as never + }) + + let handle: HarnessHandle | null = null + render( + (handle = h)} + refreshSessions={async () => undefined} + requestGateway={requestGateway} + runtimeIdByStoredSessionIdRef={runtimeIdByStoredSessionIdRef} + selectedStoredSessionIdRef={selectedStoredSessionIdRef} + storedSessionId={STORED_SESSION_B} + /> + ) + await waitFor(() => expect(handle).not.toBeNull()) + + await handle!.submitText('first message in a genuinely fresh session') + + expect(calls.some(c => c.method === 'session.resume')).toBe(false) + expect( + calls.find(c => c.method === 'prompt.submit' && c.params?.session_id === RUNTIME_SESSION_A) + ).toBeDefined() + }) + + it('resumes the selected session when its ownership cache entry is missing', async () => { + const calls: { method: string; params?: Record }[] = [] + const selectedStoredSessionIdRef: MutableRefObject = { current: STORED_SESSION_B } + const activeSessionIdRef: MutableRefObject = { current: RUNTIME_SESSION_A } + const runtimeIdByStoredSessionIdRef: MutableRefObject> = { current: new Map() } + + const requestGateway = vi.fn(async (method: string, params?: Record) => { + calls.push({ method, params }) + + if (method === 'session.resume') { + return { session_id: RUNTIME_SESSION_B_RESUMED } as never + } + + return {} as never + }) + + let handle: HarnessHandle | null = null + render( + (handle = h)} + refreshSessions={async () => undefined} + requestGateway={requestGateway} + runtimeIdByStoredSessionIdRef={runtimeIdByStoredSessionIdRef} + selectedStoredSessionIdRef={selectedStoredSessionIdRef} + storedSessionId={STORED_SESSION_B} + /> + ) + await waitFor(() => expect(handle).not.toBeNull()) + + await handle!.submitText('message after an ownership-cache miss') + + expect(calls.find(c => c.method === 'session.resume')?.params).toEqual({ + session_id: STORED_SESSION_B, + source: 'desktop' + }) + expect( + calls.find(c => c.method === 'prompt.submit' && c.params?.session_id === RUNTIME_SESSION_A) + ).toBeUndefined() + expect( + calls.find(c => c.method === 'prompt.submit' && c.params?.session_id === RUNTIME_SESSION_B_RESUMED) + ).toBeDefined() + }) +}) + describe('usePromptActions eager attachment upload (drop-time)', () => { afterEach(() => { cleanup() diff --git a/apps/desktop/src/app/session/hooks/use-prompt-actions/index.ts b/apps/desktop/src/app/session/hooks/use-prompt-actions/index.ts index 709ea7026ddc1..3fabb514c7bda 100644 --- a/apps/desktop/src/app/session/hooks/use-prompt-actions/index.ts +++ b/apps/desktop/src/app/session/hooks/use-prompt-actions/index.ts @@ -235,6 +235,7 @@ interface PromptActionsOptions { refreshSessions: () => Promise requestGateway: (method: string, params?: Record, timeoutMs?: number) => Promise resumeStoredSession: (storedSessionId: string) => Promise | void + runtimeIdByStoredSessionIdRef: MutableRefObject> selectedStoredSessionIdRef: MutableRefObject startFreshSessionDraft: () => void sttEnabled: boolean @@ -266,6 +267,7 @@ export function usePromptActions({ refreshSessions, requestGateway, resumeStoredSession, + runtimeIdByStoredSessionIdRef, selectedStoredSessionIdRef, startFreshSessionDraft, sttEnabled, @@ -482,6 +484,7 @@ export function usePromptActions({ getRuntimeIdForStoredSession, getRouteToken, requestGateway, + runtimeIdByStoredSessionIdRef, resumeStoredSession, selectedStoredSessionIdRef, syncAttachmentsForSubmit, diff --git a/apps/desktop/src/app/session/hooks/use-prompt-actions/submit.ts b/apps/desktop/src/app/session/hooks/use-prompt-actions/submit.ts index 422d79ef64ff7..52976138e1cc9 100644 --- a/apps/desktop/src/app/session/hooks/use-prompt-actions/submit.ts +++ b/apps/desktop/src/app/session/hooks/use-prompt-actions/submit.ts @@ -60,6 +60,7 @@ interface SubmitPromptDeps { getRuntimeIdForStoredSession: (storedSessionId: string) => null | string getRouteToken: () => string requestGateway: GatewayRequest + runtimeIdByStoredSessionIdRef: MutableRefObject> resumeStoredSession: (storedSessionId: string) => Promise | void selectedStoredSessionIdRef: MutableRefObject syncAttachmentsForSubmit: ( @@ -105,6 +106,7 @@ export function useSubmitPrompt(deps: SubmitPromptDeps) { getRuntimeIdForStoredSession, getRouteToken, requestGateway, + runtimeIdByStoredSessionIdRef, resumeStoredSession, selectedStoredSessionIdRef, syncAttachmentsForSubmit, @@ -429,6 +431,39 @@ export function useSubmitPrompt(deps: SubmitPromptDeps) { sessionId = null } + // Entry-time consistency check (#64789/#65328): activeSessionId is a + // render-closure value that can already be stale relative to the + // currently selected stored session by the time submit fires (e.g. a + // fast reselect, or a new-chat draft's active ref not yet re-homed). + // The #54527 drift guard only catches divergence that happens AFTER + // this point, so an already-diverged runtime/stored pair sails + // through it. Prove membership from BOTH directions against the same + // cache rather than trusting an absent forward entry as "no + // conflict" — a bare forward miss can't rule out the runtime being + // known to belong to a DIFFERENT stored session (the failure mode a + // one-directional check misses): if either direction disagrees, + // activeSessionId is not trustworthy and the resume-by-stored-id path + // below re-establishes the correct runtime id instead of silently + // sending to the wrong one. + const ownershipStoredSessionId = options?.sessionId ? null : targetStoredSessionId + + if (sessionId && ownershipStoredSessionId) { + const provenRuntimeId = runtimeIdByStoredSessionIdRef.current.get(ownershipStoredSessionId) + // A selected stored session requires positive ownership proof. A cache + // miss is therefore unsafe too: the active runtime may belong to an + // entirely different stored session, so resume the selected id instead + // of sending to an unverified runtime. + const knownMismatch = provenRuntimeId !== sessionId + + const runtimeOwnedByOtherStored = Array.from(runtimeIdByStoredSessionIdRef.current.entries()).some( + ([storedId, runtimeId]) => runtimeId === sessionId && storedId !== ownershipStoredSessionId + ) + + if (knownMismatch || runtimeOwnedByOtherStored) { + sessionId = null + } + } + if (sessionId) { seedOptimistic(sessionId) } else if (targetIsCurrentView()) { @@ -743,6 +778,7 @@ export function useSubmitPrompt(deps: SubmitPromptDeps) { getRuntimeIdForStoredSession, getRouteToken, requestGateway, + runtimeIdByStoredSessionIdRef, resumeStoredSession, scope, selectedStoredSessionIdRef, From 898cc871250f0390adcf339e4509d8fda1b74e4a Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 18:55:42 -0700 Subject: [PATCH 038/376] =?UTF-8?q?fix(desktop):=20salvage=20session-race?= =?UTF-8?q?=20sequencing=20cluster=20=E2=80=94=20widen=20Stop=20cooldown?= =?UTF-8?q?=20to=20tile=20interrupts,=20prove=20submit=20ownership=20both?= =?UTF-8?q?=20ways?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Follow-ups on top of the two salvaged commits: - widen the #83855 recently-interrupted cooldown to the session-tile interrupt path (use-session-tile-delegate.interruptSession) — same race class, sibling call site: a tile Stop also clears busy before the gateway settles, so a quick tile edit/resend raced 4009 session busy. The recovered runtime id is marked too. - regression test for the tile cooldown. - refresh three #65328 ownership-proof assertions to tolerate the omit_messages flag main now sends on session.resume (toMatchObject). - eslint import-order fix in utils.test.ts. --- .../hooks/use-session-tile-delegate.test.ts | 26 +++++++++++++++++++ .../hooks/use-session-tile-delegate.ts | 16 ++++++++++-- .../hooks/use-prompt-actions/index.test.tsx | 6 ++--- .../hooks/use-prompt-actions/utils.test.ts | 2 +- 4 files changed, 44 insertions(+), 6 deletions(-) diff --git a/apps/desktop/src/app/contrib/hooks/use-session-tile-delegate.test.ts b/apps/desktop/src/app/contrib/hooks/use-session-tile-delegate.test.ts index 4a236aa2b45b4..c00fa22255803 100644 --- a/apps/desktop/src/app/contrib/hooks/use-session-tile-delegate.test.ts +++ b/apps/desktop/src/app/contrib/hooks/use-session-tile-delegate.test.ts @@ -100,3 +100,29 @@ describe('useSessionTileDelegate resumeTile', () => { }) }) }) + +describe('useSessionTileDelegate interruptSession', () => { + beforeEach(() => { + setSessions([]) + }) + + afterEach(async () => { + setSessions([]) + const { clearSessionRecentlyInterrupted } = await import('../../session/hooks/use-prompt-actions/utils') + clearSessionRecentlyInterrupted() + }) + + it('marks the session recently interrupted so a quick tile edit/resend still interrupt-firsts (#83855)', async () => { + const { isSessionRecentlyInterrupted } = await import('../../session/hooks/use-prompt-actions/utils') + + const requestGateway = vi.fn(async () => ({}) as never) + + renderTile(requestGateway) + await sessionTileDelegate()!.interruptSession('runtime-tile-1') + + expect(requestGateway).toHaveBeenCalledWith('session.interrupt', { session_id: 'runtime-tile-1' }) + // Same 3s cooldown the primary chat's Stop sets: busy reads false while the + // gateway winds down, so the rewind path must still interrupt-first. + expect(isSessionRecentlyInterrupted('runtime-tile-1')).toBe(true) + }) +}) diff --git a/apps/desktop/src/app/contrib/hooks/use-session-tile-delegate.ts b/apps/desktop/src/app/contrib/hooks/use-session-tile-delegate.ts index 0fd18f15b7815..86936bb84884e 100644 --- a/apps/desktop/src/app/contrib/hooks/use-session-tile-delegate.ts +++ b/apps/desktop/src/app/contrib/hooks/use-session-tile-delegate.ts @@ -6,7 +6,7 @@ import { publishSessionState, setSessionTileDelegate } from '@/store/session-sta import type { SessionResumeResponse } from '@/types/hermes' import type { usePromptActions } from '../../session/hooks/use-prompt-actions' -import { withSessionNotFoundResume } from '../../session/hooks/use-prompt-actions/utils' +import { markSessionRecentlyInterrupted, withSessionNotFoundResume } from '../../session/hooks/use-prompt-actions/utils' import { resolveSessionProfile } from '../../session/hooks/use-session-actions/utils' import type { useSessionStateCache } from '../../session/hooks/use-session-state-cache' import type { GatewayRequester } from '../types' @@ -85,11 +85,23 @@ export function useSessionTileDelegate({ await executeSlashCommand(rawCommand, { sessionId }) }, interruptSession: async runtimeId => { + // Same cooldown as the primary chat's Stop (#83855): the gateway may + // still be winding down after this interrupt, so a quick edit/resend + // on the tile must go interrupt-first even though busy already reads + // false. Mark the runtime id (and any recovered id) before the RPC so + // the window covers the whole wind-down. + markSessionRecentlyInterrupted(runtimeId) await withSessionNotFoundResume( runtimeId, storedSessionIdForRuntime(runtimeId), liveId => requestGateway('session.interrupt', { session_id: liveId }), - { requestGateway, onRecovered: rebindTileRuntime(runtimeId) } + { + requestGateway, + onRecovered: recoveredId => { + markSessionRecentlyInterrupted(recoveredId) + rebindTileRuntime(runtimeId)(recoveredId) + } + } ) }, resumeTile: async storedSessionId => { diff --git a/apps/desktop/src/app/session/hooks/use-prompt-actions/index.test.tsx b/apps/desktop/src/app/session/hooks/use-prompt-actions/index.test.tsx index 2796d3edda78e..f8c41b76253ae 100644 --- a/apps/desktop/src/app/session/hooks/use-prompt-actions/index.test.tsx +++ b/apps/desktop/src/app/session/hooks/use-prompt-actions/index.test.tsx @@ -4288,7 +4288,7 @@ describe('usePromptActions submit entry-time runtime ownership proof (#64789/#65 await handle!.submitText('ordinary text for the selected project B session') - expect(calls.find(c => c.method === 'session.resume')?.params).toEqual({ + expect(calls.find(c => c.method === 'session.resume')?.params).toMatchObject({ session_id: STORED_SESSION_B, source: 'desktop' }) @@ -4345,7 +4345,7 @@ describe('usePromptActions submit entry-time runtime ownership proof (#64789/#65 await handle!.submitText('ordinary text for the freshly created project B session') - expect(calls.find(c => c.method === 'session.resume')?.params).toEqual({ + expect(calls.find(c => c.method === 'session.resume')?.params).toMatchObject({ session_id: STORED_SESSION_B, source: 'desktop' }) @@ -4430,7 +4430,7 @@ describe('usePromptActions submit entry-time runtime ownership proof (#64789/#65 await handle!.submitText('message after an ownership-cache miss') - expect(calls.find(c => c.method === 'session.resume')?.params).toEqual({ + expect(calls.find(c => c.method === 'session.resume')?.params).toMatchObject({ session_id: STORED_SESSION_B, source: 'desktop' }) diff --git a/apps/desktop/src/app/session/hooks/use-prompt-actions/utils.test.ts b/apps/desktop/src/app/session/hooks/use-prompt-actions/utils.test.ts index e00efecf66e35..c21e38f9c9174 100644 --- a/apps/desktop/src/app/session/hooks/use-prompt-actions/utils.test.ts +++ b/apps/desktop/src/app/session/hooks/use-prompt-actions/utils.test.ts @@ -19,8 +19,8 @@ import { isSessionRecentlyInterrupted, isSubmitInFlight, markSessionRecentlyInterrupted, - RECENT_INTERRUPT_COOLDOWN_MS, readFileDataUrlForAttach, + RECENT_INTERRUPT_COOLDOWN_MS, releaseSubmitInFlight, renderRpcResult, SessionRecoveryAborted, From 62dbd87b1b726bcb66b2370a4e0be43e0c02dfdf Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 18:55:56 -0700 Subject: [PATCH 039/376] chore: map contributor email for salvaged PR attribution --- contributors/emails/Olympus.roots@outlook.com | 1 + 1 file changed, 1 insertion(+) create mode 100644 contributors/emails/Olympus.roots@outlook.com diff --git a/contributors/emails/Olympus.roots@outlook.com b/contributors/emails/Olympus.roots@outlook.com new file mode 100644 index 0000000000000..f01752d16b229 --- /dev/null +++ b/contributors/emails/Olympus.roots@outlook.com @@ -0,0 +1 @@ +olympusbuildz From f703e7061869fd6af9efb599ee3cfa435c49c551 Mon Sep 17 00:00:00 2001 From: VooDoo Pixels Date: Tue, 11 Aug 2026 19:01:27 -0400 Subject: [PATCH 040/376] fix: make desktop approval routing reliable Correlate approval requests, reject stale responses, replay pending approvals after reconnect or session resume, and preserve fail-closed timeout behavior. --- .../hooks/use-message-stream/gateway-event.ts | 20 +++++-- .../components/assistant-ui/tool/approval.tsx | 7 ++- apps/desktop/src/lib/gateway-events.test.ts | 8 ++- apps/desktop/src/lib/gateway-events.ts | 14 +++++ apps/desktop/src/store/prompts.test.ts | 57 ++++++++++++++++++ apps/desktop/src/store/prompts.ts | 53 +++++++++++++++- tests/tools/test_approval.py | 58 ++++++++++++++++++ tests/tui_gateway/test_protocol.py | 60 +++++++++++++++++++ tools/approval.py | 35 +++++++++-- tui_gateway/methods_prompt.py | 33 ++++++++++ 10 files changed, 330 insertions(+), 15 deletions(-) diff --git a/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event.ts b/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event.ts index 46b992a68d42f..a9cd1c8267703 100644 --- a/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event.ts +++ b/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event.ts @@ -12,7 +12,7 @@ import { translateNow } from '@/i18n' import { type GatewayEventPayload, textPart } from '@/lib/chat-messages' import { coerceGatewayText, coerceThinkingText, normalizePersonalityValue } from '@/lib/chat-runtime' import { playCompletionSound } from '@/lib/completion-sound' -import { resolveGatewayEventSessionId } from '@/lib/gateway-events' +import { approvalReplaySessionId, resolveGatewayEventSessionId } from '@/lib/gateway-events' import { triggerHaptic } from '@/lib/haptics' import { modelOptionsQueryKey } from '@/lib/model-options' import { isProviderSetupErrorMessage } from '@/lib/provider-setup-errors' @@ -42,7 +42,13 @@ import { revealDesktopPane } from '@/store/pane-focus' import { flashPetActivity, markPetUnread, setPetActivity } from '@/store/pet' import { $activeGatewayProfile, normalizeProfileKey } from '@/store/profile' import { followActiveSessionCwd } from '@/store/projects' -import { clearAllPrompts, setApprovalRequest, setSecretRequest, setSudoRequest } from '@/store/prompts' +import { + clearAllPrompts, + receiveApprovalRequest, + replayPendingApproval, + setSecretRequest, + setSudoRequest +} from '@/store/prompts' import { recordAgentReaction } from '@/store/reactions-local' import { $currentCwd, @@ -316,6 +322,11 @@ export function useGatewayEventHandler(deps: GatewayEventDeps) { const sessionId = route.sessionId const isActiveEvent = !!sessionId && sessionId === activeSessionIdRef.current + const replaySessionId = approvalReplaySessionId(event.type, activeSessionIdRef.current, sessionId) + if (replaySessionId) { + void replayPendingApproval($gateway.get(), replaySessionId).catch(() => undefined) + } + // Mid-turn compaction does not emit another message.start. The first // model output or tool event proves summarization has finished and the // turn has resumed, so retire the phase label without waiting for the @@ -1024,7 +1035,7 @@ export function useGatewayEventHandler(deps: GatewayEventDeps) { const command = typeof payload?.command === 'string' ? payload.command : '' const description = typeof payload?.description === 'string' ? payload.description : 'dangerous command' - setApprovalRequest({ + void receiveApprovalRequest($gateway.get(), { // false only when a tirith warning forbids it; backend omits the field otherwise. allowPermanent: payload?.allow_permanent !== false, choices: Array.isArray(payload?.choices) @@ -1032,9 +1043,10 @@ export function useGatewayEventHandler(deps: GatewayEventDeps) { : undefined, command, description, + requestId: typeof payload?.request_id === 'string' ? payload.request_id : undefined, sessionId: sessionId ?? null, smartDenied: payload?.smart_denied === true - }) + }).catch(() => undefined) if (sessionId) { updateSessionState(sessionId, state => ({ ...state, needsInput: true })) diff --git a/apps/desktop/src/components/assistant-ui/tool/approval.tsx b/apps/desktop/src/components/assistant-ui/tool/approval.tsx index 50de684f9d98a..d0b3647837e54 100644 --- a/apps/desktop/src/components/assistant-ui/tool/approval.tsx +++ b/apps/desktop/src/components/assistant-ui/tool/approval.tsx @@ -24,6 +24,7 @@ import { type ApprovalRequest, clearApprovalRequest, registerApprovalInlineAnchor, + replayPendingApproval, sessionApprovalInlineVisible, sessionApprovalRequest } from '@/store/prompts' @@ -144,16 +145,18 @@ const ApprovalBar: FC<{ request: ApprovalRequest; surface: 'floating' | 'inline' try { await gateway.request<{ resolved?: boolean }>('approval.respond', { choice, + request_id: request.requestId, session_id: request.sessionId ?? undefined }) triggerHaptic(choice === 'deny' ? 'cancel' : 'submit') - clearApprovalRequest(request.sessionId) + clearApprovalRequest(request.sessionId, request.requestId) + void replayPendingApproval(gateway, request.sessionId).catch(() => undefined) } catch (error) { notifyError(error, copy.sendFailed) setSubmitting(null) } }, - [busy, copy.gatewayDisconnected, copy.sendFailed, gateway, request.sessionId] + [busy, copy.gatewayDisconnected, copy.sendFailed, gateway, request.requestId, request.sessionId] ) // ⌘/Ctrl+Enter → Run, Esc → Reject. diff --git a/apps/desktop/src/lib/gateway-events.test.ts b/apps/desktop/src/lib/gateway-events.test.ts index 02c3f643ca6a6..907acbdd80ad0 100644 --- a/apps/desktop/src/lib/gateway-events.test.ts +++ b/apps/desktop/src/lib/gateway-events.test.ts @@ -1,8 +1,14 @@ import { describe, expect, it } from 'vitest' -import { gatewayEventRequiresSessionId, resolveGatewayEventSessionId } from './gateway-events' +import { approvalReplaySessionId, gatewayEventRequiresSessionId, resolveGatewayEventSessionId } from './gateway-events' describe('gateway event routing', () => { + it('rehydrates pending approvals on reconnect ready and resumed session info', () => { + expect(approvalReplaySessionId('gateway.ready', 'active-1', null)).toBe('active-1') + expect(approvalReplaySessionId('session.info', 'active-1', 'routed-1')).toBe('routed-1') + expect(approvalReplaySessionId('message.delta', 'active-1', 'routed-1')).toBeNull() + }) + it('drops only unscoped subagent events (genuinely background work)', () => { expect(gatewayEventRequiresSessionId('subagent.progress')).toBe(true) expect(gatewayEventRequiresSessionId('subagent.start')).toBe(true) diff --git a/apps/desktop/src/lib/gateway-events.ts b/apps/desktop/src/lib/gateway-events.ts index 4b09ba30d7ed4..0dd2a6cc27d5c 100644 --- a/apps/desktop/src/lib/gateway-events.ts +++ b/apps/desktop/src/lib/gateway-events.ts @@ -70,6 +70,20 @@ export interface GatewayEventSessionRoute { sessionId: null | string } +export function approvalReplaySessionId( + eventType: string | undefined, + activeSessionId: null | string, + routedSessionId: null | string +): null | string { + if (eventType === 'gateway.ready') { + return activeSessionId + } + if (eventType === 'session.info') { + return routedSessionId + } + return null +} + /** * Resolve which runtime session owns a gateway event. * diff --git a/apps/desktop/src/store/prompts.test.ts b/apps/desktop/src/store/prompts.test.ts index a761adf6f25a7..1ca6052ce151a 100644 --- a/apps/desktop/src/store/prompts.test.ts +++ b/apps/desktop/src/store/prompts.test.ts @@ -10,6 +10,8 @@ import { clearApprovalRequest, clearSecretRequest, clearSudoRequest, + receiveApprovalRequest, + replayPendingApproval, setApprovalRequest, setSecretRequest, setSudoRequest @@ -67,6 +69,61 @@ describe('approval prompt store', () => { expect($approvalRequest.get()?.allowPermanent).toBe(false) }) + + it('correlates clearing to the exact approval request id', () => { + setApprovalRequest({ command: 'x', description: 'd', requestId: 'r1', sessionId: 's1' }) + + clearApprovalRequest('s1', 'stale') + expect($approvalRequest.get()?.requestId).toBe('r1') + clearApprovalRequest('s1', 'r1') + expect($approvalRequest.get()).toBeNull() + }) + + it('acknowledges an approval only after parking it', async () => { + const calls: Array<[string, Record]> = [] + const gateway = { + request: async (method: string, params: Record) => { + calls.push([method, params]) + return { acknowledged: true } + } + } + + await receiveApprovalRequest(gateway, { + command: 'x', + description: 'd', + requestId: 'r1', + sessionId: 's1' + }) + + expect($approvalRequest.get()?.requestId).toBe('r1') + expect(calls).toEqual([['approval.received', { request_id: 'r1', session_id: 's1' }]]) + }) + + it('replays and acknowledges the oldest unresolved approval after reconnect', async () => { + const calls: Array<[string, Record]> = [] + const gateway = { + request: async (method: string, params: Record) => { + calls.push([method, params]) + if (method === 'approval.pending') { + return { + approvals: [ + { command: 'first', description: 'd1', request_id: 'r1' }, + { command: 'second', description: 'd2', request_id: 'r2' } + ] + } + } + return { acknowledged: true } + } + } + + await replayPendingApproval(gateway, 's1') + + expect($approvalRequest.get()?.requestId).toBe('r1') + expect(calls).toEqual([ + ['approval.pending', { session_id: 's1' }], + ['approval.received', { request_id: 'r1', session_id: 's1' }] + ]) + }) }) describe('sudo prompt store', () => { diff --git a/apps/desktop/src/store/prompts.ts b/apps/desktop/src/store/prompts.ts index 0efe97515a993..6e7d7bc6c92ff 100644 --- a/apps/desktop/src/store/prompts.ts +++ b/apps/desktop/src/store/prompts.ts @@ -67,18 +67,32 @@ function keyedPromptStore(): PromptStore { } } -// Approval is session-keyed on the backend (one in-flight approval per session, -// resolved via approval.respond {choice, session_id}). It carries no request_id, -// unlike sudo/secret which are _block()-style request/response. +// Approval is session-keyed on the backend and correlated by `request_id` when +// available (legacy ID-free responses remain FIFO-compatible). Resolved via +// approval.respond {choice, request_id, session_id}. export interface ApprovalRequest extends KeyedPrompt { // false when the backend won't honor a permanent allow (tirith warning) → hide "Always allow". allowPermanent?: boolean choices?: string[] command: string description: string + requestId?: string smartDenied?: boolean } +interface ApprovalGateway { + request: (method: string, params: Record) => Promise +} + +interface PendingApprovalPayload { + allow_permanent?: boolean + choices?: unknown + command?: unknown + description?: unknown + request_id?: unknown + smart_denied?: boolean +} + export interface SudoRequest extends KeyedPrompt { requestId: string } @@ -101,6 +115,39 @@ export const $approvalRequest = approval.$active export const setApprovalRequest = approval.set export const clearApprovalRequest = approval.clear +export async function receiveApprovalRequest(gateway: ApprovalGateway | null, request: ApprovalRequest): Promise { + setApprovalRequest(request) + if (gateway && request.requestId && request.sessionId) { + await gateway.request('approval.received', { + request_id: request.requestId, + session_id: request.sessionId + }) + } +} + +export async function replayPendingApproval(gateway: ApprovalGateway | null, sessionId: string | null): Promise { + if (!gateway || !sessionId) { + return + } + const rawResult = await gateway.request('approval.pending', { + session_id: sessionId + }) + const result = rawResult && typeof rawResult === 'object' ? (rawResult as { approvals?: PendingApprovalPayload[] }) : {} + const pending = Array.isArray(result?.approvals) ? result.approvals[0] : undefined + if (!pending || typeof pending.request_id !== 'string') { + return + } + await receiveApprovalRequest(gateway, { + allowPermanent: pending.allow_permanent !== false, + choices: Array.isArray(pending.choices) ? pending.choices.filter(choice => typeof choice === 'string') : undefined, + command: typeof pending.command === 'string' ? pending.command : '', + description: typeof pending.description === 'string' ? pending.description : 'dangerous command', + requestId: pending.request_id, + sessionId, + smartDenied: pending.smart_denied === true + }) +} + /** The prompt request for one specific session — the tile counterpart of the * active-session `$*Request` views (same map, fixed key). */ export const sessionApprovalRequest = (sessionId: string | null) => diff --git a/tests/tools/test_approval.py b/tests/tools/test_approval.py index 8892a089fc419..1a75403436d99 100644 --- a/tests/tools/test_approval.py +++ b/tests/tools/test_approval.py @@ -1311,6 +1311,64 @@ def _fail_notify(_data): ] assert hook_calls[-1][1]["choice"] == "notify_failed" + def test_pending_approval_is_replayable_and_acknowledged(self, monkeypatch): + from tools import approval as mod + + self._force_short_timeout(monkeypatch, seconds=2) + notified = [] + mod.register_gateway_notify(self.SESSION_KEY, lambda data: notified.append(data)) + result_holder = {} + + thread = threading.Thread( + target=lambda: result_holder.setdefault( + "result", mod.check_all_command_guards("rm -rf .git", "local") + ) + ) + thread.start() + for _ in range(200): + if notified: + break + time.sleep(0.005) + + request_id = notified[0]["request_id"] + assert request_id + assert mod.list_gateway_approvals(self.SESSION_KEY) == [notified[0]] + assert mod.ack_gateway_approval(self.SESSION_KEY, request_id) is True + assert mod.resolve_gateway_approval( + self.SESSION_KEY, "once", request_id=request_id + ) == 1 + thread.join(timeout=5) + assert result_holder["result"]["approved"] is True + + def test_stale_request_id_cannot_resolve_current_approval(self, monkeypatch): + from tools import approval as mod + + self._force_short_timeout(monkeypatch, seconds=2) + notified = [] + mod.register_gateway_notify(self.SESSION_KEY, lambda data: notified.append(data)) + result_holder = {} + thread = threading.Thread( + target=lambda: result_holder.setdefault( + "result", mod.check_all_command_guards("rm -rf .git", "local") + ) + ) + thread.start() + for _ in range(200): + if notified: + break + time.sleep(0.005) + + request_id = notified[0]["request_id"] + assert mod.resolve_gateway_approval( + self.SESSION_KEY, "once", request_id="stale-request" + ) == 0 + assert mod.list_gateway_approvals(self.SESSION_KEY) + assert mod.resolve_gateway_approval( + self.SESSION_KEY, "deny", request_id=request_id + ) == 1 + thread.join(timeout=5) + assert result_holder["result"]["approved"] is False + class TestTirithImportErrorFailOpenPolicy: """Regression guard for #20733. diff --git a/tests/tui_gateway/test_protocol.py b/tests/tui_gateway/test_protocol.py index 08934c7201ad4..5a1dd764c2c9f 100644 --- a/tests/tui_gateway/test_protocol.py +++ b/tests/tui_gateway/test_protocol.py @@ -274,6 +274,66 @@ def test_late_prompt_response_is_idempotent(server, method, value_key): assert response["result"] == {"status": "expired"} +def test_approval_pending_replays_unresolved_requests(server, monkeypatch): + from tools import approval + + server._sessions["ui-1"] = {"session_key": "agent-1", "history": []} + pending = [{"request_id": "req-1", "command": "danger"}] + monkeypatch.setattr(approval, "list_gateway_approvals", lambda key: pending if key == "agent-1" else []) + + response = server.handle_request( + {"id": "r1", "method": "approval.pending", "params": {"session_id": "ui-1"}} + ) + + assert response["result"] == {"approvals": pending} + + +def test_approval_received_acknowledges_exact_request(server, monkeypatch): + from tools import approval + + server._sessions["ui-1"] = {"session_key": "agent-1", "history": []} + calls = [] + monkeypatch.setattr( + approval, + "ack_gateway_approval", + lambda key, request_id: calls.append((key, request_id)) or True, + ) + + response = server.handle_request( + { + "id": "r2", + "method": "approval.received", + "params": {"session_id": "ui-1", "request_id": "req-1"}, + } + ) + + assert response["result"] == {"acknowledged": True} + assert calls == [("agent-1", "req-1")] + + +def test_approval_response_correlates_request_id(server, monkeypatch): + from tools import approval + + server._sessions["ui-1"] = {"session_key": "agent-1", "history": []} + calls = [] + monkeypatch.setattr( + approval, + "resolve_gateway_approval", + lambda key, choice, **kwargs: calls.append((key, choice, kwargs)) or 1, + ) + + response = server.handle_request( + { + "id": "r3", + "method": "approval.respond", + "params": {"session_id": "ui-1", "request_id": "req-1", "choice": "once"}, + } + ) + + assert response["result"] == {"resolved": 1} + assert calls == [("agent-1", "once", {"resolve_all": False, "request_id": "req-1"})] + + def test_clear_pending(server): ev = threading.Event() # _pending values are (sid, Event) tuples diff --git a/tools/approval.py b/tools/approval.py index db0595747cd64..f178a4af13929 100644 --- a/tools/approval.py +++ b/tools/approval.py @@ -22,6 +22,7 @@ import threading import time import unicodedata +import uuid from typing import Optional from hermes_cli.config import cfg_get @@ -2562,11 +2563,13 @@ def _denial_breaker_addendum(session_key: str) -> str: class _ApprovalEntry: """One pending dangerous-command approval inside a gateway session.""" - __slots__ = ("event", "data", "result", "reason") + __slots__ = ("event", "data", "result", "reason", "acknowledged") def __init__(self, data: dict): self.event = threading.Event() - self.data = data # command, description, pattern_keys, … + self.data = dict(data) + self.data.setdefault("request_id", uuid.uuid4().hex) + self.acknowledged = False self.result: Optional[str] = None # "once"|"session"|"always"|"deny" # Optional free-text reason supplied with an explicit deny # (``/deny ``) so the agent can adapt instead of only @@ -2605,7 +2608,8 @@ def unregister_gateway_notify(session_key: str) -> None: def resolve_gateway_approval(session_key: str, choice: str, resolve_all: bool = False, - reason: Optional[str] = None) -> int: + reason: Optional[str] = None, + request_id: Optional[str] = None) -> int: """Called by the gateway's /approve or /deny handler to unblock waiting agent thread(s). @@ -2623,7 +2627,12 @@ def resolve_gateway_approval(session_key: str, choice: str, queue = _gateway_queues.get(session_key) if not queue: return 0 - if resolve_all: + if request_id: + targets = [entry for entry in queue if entry.data.get("request_id") == request_id] + if not targets: + return 0 + queue[:] = [entry for entry in queue if entry not in targets] + elif resolve_all: targets = list(queue) queue.clear() else: @@ -2639,6 +2648,22 @@ def resolve_gateway_approval(session_key: str, choice: str, return len(targets) +def list_gateway_approvals(session_key: str) -> list[dict]: + """Return replay-safe snapshots of unresolved approvals for one session.""" + with _lock: + return [dict(entry.data) for entry in _gateway_queues.get(session_key, [])] + + +def ack_gateway_approval(session_key: str, request_id: str) -> bool: + """Record that a client received a particular pending approval request.""" + with _lock: + for entry in _gateway_queues.get(session_key, []): + if entry.data.get("request_id") == request_id: + entry.acknowledged = True + return True + return False + + def has_blocking_approval(session_key: str) -> bool: """Check if a session has one or more blocking gateway approvals waiting.""" with _lock: @@ -3930,7 +3955,7 @@ def _drop_entry() -> None: # Notify the user (bridges sync agent thread → async gateway) try: - notify_cb(approval_data) + notify_cb(dict(entry.data)) except Exception as exc: logger.warning("Gateway approval notify failed: %s", exc) _drop_entry() diff --git a/tui_gateway/methods_prompt.py b/tui_gateway/methods_prompt.py index a1c0860a60465..e6c1f236c0200 100644 --- a/tui_gateway/methods_prompt.py +++ b/tui_gateway/methods_prompt.py @@ -1347,6 +1347,38 @@ def _(rid, params: dict) -> dict: return _respond(rid, params, "value", allow_expired=True) +@method("approval.pending") +def _(rid, params: dict) -> dict: + session, err = _sess(params, rid) + if err: + return err + try: + from tools.approval import list_gateway_approvals + + return _ok(rid, {"approvals": list_gateway_approvals(session["session_key"])}) + except Exception as e: + return _err(rid, 5004, str(e)) + + +@method("approval.received") +def _(rid, params: dict) -> dict: + session, err = _sess(params, rid) + if err: + return err + request_id = params.get("request_id") + if not isinstance(request_id, str) or not request_id: + return _err(rid, 4006, "request_id required") + try: + from tools.approval import ack_gateway_approval + + return _ok( + rid, + {"acknowledged": ack_gateway_approval(session["session_key"], request_id)}, + ) + except Exception as e: + return _err(rid, 5004, str(e)) + + @method("approval.respond") def _(rid, params: dict) -> dict: session, err = _sess(params, rid) @@ -1362,6 +1394,7 @@ def _(rid, params: dict) -> dict: session["session_key"], params.get("choice", "deny"), resolve_all=params.get("all", False), + request_id=params.get("request_id"), ) }, ) From 38b9005b957f6ef3841072a1b4e0012f479abf30 Mon Sep 17 00:00:00 2001 From: konsisumer Date: Sun, 9 Aug 2026 01:48:03 +0200 Subject: [PATCH 041/376] fix(desktop): replay pending approvals after reconnect --- .../hooks/use-session-actions/index.ts | 23 +++++++++++ apps/desktop/src/types/hermes.ts | 10 +++++ tests/tui_gateway/test_protocol.py | 38 ++++++++++++++++++- tools/approval.py | 16 ++++++++ tui_gateway/server.py | 34 +++++++++++++---- 5 files changed, 113 insertions(+), 8 deletions(-) diff --git a/apps/desktop/src/app/session/hooks/use-session-actions/index.ts b/apps/desktop/src/app/session/hooks/use-session-actions/index.ts index 3db4e8008e366..8edd118c871bc 100644 --- a/apps/desktop/src/app/session/hooks/use-session-actions/index.ts +++ b/apps/desktop/src/app/session/hooks/use-session-actions/index.ts @@ -14,6 +14,7 @@ import { clearQueuedPrompts, migrateQueuedPrompts } from '@/store/composer-queue import { $pinnedSessionIds } from '@/store/layout' import { clearNotifications, notify, notifyError } from '@/store/notifications' import { $activeGatewayProfile, $newChatProfile, ensureGatewayProfile, normalizeProfileKey } from '@/store/profile' +import { setApprovalRequest } from '@/store/prompts' import { beginSessionMutation, endSessionMutation, @@ -192,6 +193,24 @@ interface FreshSessionDraftOptions { workspaceTarget?: NewChatWorkspaceTarget } +function restorePendingApproval(response: SessionResumeResponse, sessionId: string): boolean { + const pending = response.pending_approval + + if (!pending) { + return false + } + + setApprovalRequest({ + allowPermanent: pending.allow_permanent !== false, + choices: pending.choices, + command: pending.command ?? '', + description: pending.description ?? 'dangerous command', + sessionId, + smartDenied: pending.smart_denied === true + }) + return true +} + function normalizeNewChatWorkspaceTarget(target: NewChatWorkspaceTarget): NewChatWorkspaceTarget { return typeof target === 'string' ? target.trim() || null : target } @@ -767,6 +786,7 @@ export function useSessionActions({ sessionStateByRuntimeIdRef.current.delete(cachedRuntimeId) dropSessionState(cachedRuntimeId) } else { + const pendingApproval = restorePendingApproval(activated, cachedRuntimeId) const runtimeInfo = applyRuntimeInfo(activated.info) // `omit_messages` means the response carries NO transcript, not @@ -823,6 +843,7 @@ export function useSessionActions({ messages: activatedMessages, busy: running, awaitingResponse: running, + needsInput: pendingApproval || state.needsInput, // Adopting someone else's turn: we'll stream its reply // without ever having received its prompt, so the settle // path must not take the "I saw it all" shortcut. @@ -1039,6 +1060,7 @@ export function useSessionActions({ setActiveSessionId(resumed.session_id) activeSessionIdRef.current = resumed.session_id + const pendingApproval = restorePendingApproval(resumed, resumed.session_id) const runtimeInfo = applyRuntimeInfo(resumed.info) patchSessionWorkspace(storedSessionId, runtimeInfo?.cwd) @@ -1051,6 +1073,7 @@ export function useSessionActions({ messages: messagesForView, busy: resumedRunning, awaitingResponse: resumedRunning && !recoveredInFlightTail, + needsInput: pendingApproval || state.needsInput, adoptedRunningTurn: state.adoptedRunningTurn || resumedRunning, ...(inFlightRecovery.applied ? { diff --git a/apps/desktop/src/types/hermes.ts b/apps/desktop/src/types/hermes.ts index 2afe4e44a794e..3d92f358b4b20 100644 --- a/apps/desktop/src/types/hermes.ts +++ b/apps/desktop/src/types/hermes.ts @@ -616,6 +616,16 @@ export interface SessionResumeResponse { queued?: null | { user?: string } + // The oldest gateway approval still waiting for a response. This is returned + // on resume so a reconnect can restore a prompt whose original event was + // emitted while the client transport was detached. + pending_approval?: { + allow_permanent?: boolean + choices?: string[] + command?: string + description?: string + smart_denied?: boolean + } info?: SessionRuntimeInfo message_count: number messages: SessionMessage[] diff --git a/tests/tui_gateway/test_protocol.py b/tests/tui_gateway/test_protocol.py index 5a1dd764c2c9f..a6dfafd659aa2 100644 --- a/tests/tui_gateway/test_protocol.py +++ b/tests/tui_gateway/test_protocol.py @@ -174,6 +174,43 @@ def test_write_json(capture): assert json.loads(buf.getvalue()) == {"test": True} +def test_live_session_payload_replays_pending_approval(server, monkeypatch): + """A reattached client receives the approval that was emitted while detached.""" + from tools import approval + + session = { + "agent": types.SimpleNamespace(), + "cols": 80, + "created_at": 1.0, + "history": [], + "history_lock": threading.Lock(), + "running": True, + "session_key": "stored-session", + } + first = { + "choices": ["once", "deny"], + "command": "rm -rf /tmp/example", + "description": "recursive delete", + } + second = {"command": "rm -rf /tmp/later", "description": "later"} + saved_queue = approval._gateway_queues.pop("stored-session", None) + approval._gateway_queues["stored-session"] = [ + approval._ApprovalEntry(first), + approval._ApprovalEntry(second), + ] + monkeypatch.setattr(server, "_approval_request_payload", lambda data: dict(data or {})) + + try: + payload = server._live_session_payload("runtime-session", session) + finally: + approval._gateway_queues.pop("stored-session", None) + if saved_queue is not None: + approval._gateway_queues["stored-session"] = saved_queue + + assert payload["pending_approval"] == first + assert payload["pending_approval"] is not first + + def test_disable_flush_env_var_actually_wires_to_module_constant(monkeypatch): """End-to-end: setting `HERMES_TUI_GATEWAY_NO_FLUSH=1` and importing `tui_gateway.transport` fresh actually flips `_DISABLE_FLUSH` true. @@ -821,4 +858,3 @@ def test_unregister_live_transport_stops_delivery(capture): assert a.frames == [] # No live transports left → fell back to stdio. assert json.loads(buf.getvalue())["params"]["type"] == "skin.changed" - diff --git a/tools/approval.py b/tools/approval.py index f178a4af13929..ac8b34e44235d 100644 --- a/tools/approval.py +++ b/tools/approval.py @@ -2670,6 +2670,22 @@ def has_blocking_approval(session_key: str) -> bool: return bool(_gateway_queues.get(session_key)) +def get_pending_gateway_approval(session_key: str) -> dict | None: + """Return a copy of the oldest unresolved gateway approval for a session. + + Reconnectable clients use this to restore an approval prompt whose original + notification was sent while their transport was detached. The queue remains + authoritative: this is a read-only snapshot, not a claim on the approval. + """ + if not session_key: + return None + with _lock: + queue = _gateway_queues.get(session_key) + if not queue: + return None + return dict(queue[0].data) + + def submit_pending(session_key: str, approval: dict): """Store a pending approval request for a session.""" with _lock: diff --git a/tui_gateway/server.py b/tui_gateway/server.py index 7e5fa9dfeffa9..f391f7c4ef0c3 100644 --- a/tui_gateway/server.py +++ b/tui_gateway/server.py @@ -1891,13 +1891,8 @@ def _send_compute_host_control( ) -def _emit_approval_request(sid: str, data: dict | None) -> None: - """Emit an ``approval.request`` event to the TUI client with the command - redacted. The approval payload is built from the RAW command string, so a - credential-shaped value Tirith flagged would otherwise be echoed verbatim - to the TUI client (#48456 — third egress transport alongside the chat - platforms and the SSE/API stream fixed in #50767). Reuse the shared gateway - seam so all approval transports redact consistently.""" +def _approval_request_payload(data: dict | None) -> dict: + """Build the client-safe representation of a pending approval.""" payload = dict(data or {}) if "choices" not in payload: if payload.get("smart_denied"): @@ -1910,6 +1905,29 @@ def _emit_approval_request(sid: str, data: dict | None) -> None: from gateway.run import _redact_approval_command payload["command"] = _redact_approval_command(payload.get("command")) + return payload + + +def _pending_approval_request_payload(session_key: str) -> dict | None: + """Read the oldest unresolved approval in a session, if there is one.""" + try: + from tools.approval import get_pending_gateway_approval + + approval = get_pending_gateway_approval(session_key) + except Exception: + logger.debug("failed to read pending approval for %s", session_key, exc_info=True) + return None + return _approval_request_payload(approval) if approval else None + + +def _emit_approval_request(sid: str, data: dict | None) -> None: + """Emit an ``approval.request`` event to the TUI client with the command + redacted. The approval payload is built from the RAW command string, so a + credential-shaped value Tirith flagged would otherwise be echoed verbatim + to the TUI client (#48456 — third egress transport alongside the chat + platforms and the SSE/API stream fixed in #50767). Reuse the shared gateway + seam so all approval transports redact consistently.""" + payload = _approval_request_payload(data) _emit("approval.request", sid, payload) @@ -8473,6 +8491,8 @@ def _live_session_payload( payload["inflight"] = inflight if queued: payload["queued"] = queued + if approval := _pending_approval_request_payload(str(session.get("session_key") or "")): + payload["pending_approval"] = approval return payload From 34d76a1df081aa87916431b4e898a31cab7629a8 Mon Sep 17 00:00:00 2001 From: Richard Howes Date: Thu, 6 Aug 2026 00:04:27 +0200 Subject: [PATCH 042/376] [verified] fix(desktop): reveal active clarify prompts Reveal a blocking clarify card by re-arming the existing thread bottom-scroll bridge after the request row is hydrated. Keep background-session prompts isolated to their needs-input indicator. Refs #53666. --- .../clarify-hydration.test.tsx | 30 +++++++++++++++++++ .../hooks/use-message-stream/gateway-event.ts | 5 ++++ 2 files changed, 35 insertions(+) diff --git a/apps/desktop/src/app/session/hooks/use-message-stream/clarify-hydration.test.tsx b/apps/desktop/src/app/session/hooks/use-message-stream/clarify-hydration.test.tsx index 6af488231fe95..de0819561dffa 100644 --- a/apps/desktop/src/app/session/hooks/use-message-stream/clarify-hydration.test.tsx +++ b/apps/desktop/src/app/session/hooks/use-message-stream/clarify-hydration.test.tsx @@ -6,6 +6,7 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import type { ClientSessionState } from '@/app/types' import { createClientSessionState } from '@/lib/chat-runtime' import { clearClarifyRequest } from '@/store/clarify' +import { onScrollToBottomRequest } from '@/store/thread-scroll' import type { RpcEvent } from '@/types/hermes' import { useMessageStream } from './index' @@ -19,6 +20,9 @@ const SID = 'session-1' let handleEvent: ((event: RpcEvent) => void) | null = null let stateRef: MutableRefObject> | null = null +let stopScrollListener: (() => void) | null = null + +const scrollToBottom = vi.fn() function Harness() { const activeSessionIdRef = useRef(SID) @@ -71,11 +75,15 @@ describe('clarify.request stream hydration', () => { handleEvent = null stateRef = null clearClarifyRequest() + scrollToBottom.mockClear() + stopScrollListener = onScrollToBottomRequest(scrollToBottom) }) afterEach(() => { cleanup() clearClarifyRequest() + stopScrollListener?.() + stopScrollListener = null vi.restoreAllMocks() }) @@ -93,6 +101,28 @@ describe('clarify.request stream hydration', () => { }) }) + it('reveals a clarify prompt raised by the active session', async () => { + await mountStream() + + clarifyRequest({ choices: ['yes', 'no'], question: 'Ship it?', request_id: 'req-reveal' }) + + expect(scrollToBottom).toHaveBeenCalledOnce() + }) + + it('does not move the active thread for a background session clarify', async () => { + await mountStream() + + act(() => + handleEvent!({ + payload: { choices: ['yes', 'no'], question: 'Ship it?', request_id: 'req-background' }, + session_id: 'session-background', + type: 'clarify.request' + }) + ) + + expect(scrollToBottom).not.toHaveBeenCalled() + }) + it('merges with the real tool.start row even though its id differs from the request id', async () => { await mountStream() diff --git a/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event.ts b/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event.ts index a9cd1c8267703..19cae17463369 100644 --- a/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event.ts +++ b/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event.ts @@ -75,6 +75,7 @@ import { dropSessionState } from '@/store/session-states' import { pruneDelegateFallbackSubagents, pruneFinishedSessionSubagents, upsertSubagent } from '@/store/subagents' import { reportMcpToolResult } from '@/store/suggestion-providers/repair' import { invalidateSkillSuggestionIndex } from '@/store/suggestion-providers/skill' +import { requestScrollToBottom } from '@/store/thread-scroll' import { clearActiveSessionTodos } from '@/store/todos' import { recordToolDiff } from '@/store/tool-diffs' import { setSessionDraftingTool } from '@/store/tool-drafting' @@ -985,6 +986,10 @@ export function useGatewayEventHandler(deps: GatewayEventDeps) { // "needs input" indicator on its row — works for the active session // too, and survives alt-tab / window blur (unlike a toast). updateSessionState(sessionId, state => ({ ...state, needsInput: true })) + + if (sessionId === activeSessionIdRef.current) { + requestScrollToBottom() + } } dispatchNativeNotification({ From 6cbe5a35b63b0783776a4755f3b95fb7d8731696 Mon Sep 17 00:00:00 2001 From: Thomas Hudspith-Tatham Date: Sun, 2 Aug 2026 16:28:30 +0100 Subject: [PATCH 043/376] fix(desktop): support multi-select clarifications --- .../hooks/use-composer-submit.test.tsx | 8 +- .../clarify-hydration.test.tsx | 28 ++++++- .../hooks/use-message-stream/gateway-event.ts | 12 ++- .../assistant-ui/clarify-tool.test.tsx | 55 +++++++++++++- .../components/assistant-ui/clarify-tool.tsx | 73 ++++++++++++------- apps/desktop/src/lib/chat-messages.ts | 1 + apps/desktop/src/store/clarify.test.ts | 1 + apps/desktop/src/store/clarify.ts | 1 + apps/desktop/src/store/prompts.test.ts | 2 +- 9 files changed, 149 insertions(+), 32 deletions(-) diff --git a/apps/desktop/src/app/chat/composer/hooks/use-composer-submit.test.tsx b/apps/desktop/src/app/chat/composer/hooks/use-composer-submit.test.tsx index cc987e9559a12..e4eae4f7cc9a1 100644 --- a/apps/desktop/src/app/chat/composer/hooks/use-composer-submit.test.tsx +++ b/apps/desktop/src/app/chat/composer/hooks/use-composer-submit.test.tsx @@ -199,7 +199,13 @@ describe('useComposerSubmit with a clarify parked on the session', () => { const parkClarify = (sessionId: string) => { $clarifyRequests.set({ - [sessionId]: { requestId: `req-${sessionId}`, question: 'which one?', choices: ['a', 'b'], sessionId } + [sessionId]: { + requestId: `req-${sessionId}`, + question: 'which one?', + choices: ['a', 'b'], + multiSelect: false, + sessionId + } }) $gateway.set({ request: gatewayRequest } as unknown as ReturnType) } diff --git a/apps/desktop/src/app/session/hooks/use-message-stream/clarify-hydration.test.tsx b/apps/desktop/src/app/session/hooks/use-message-stream/clarify-hydration.test.tsx index de0819561dffa..306c460a5607e 100644 --- a/apps/desktop/src/app/session/hooks/use-message-stream/clarify-hydration.test.tsx +++ b/apps/desktop/src/app/session/hooks/use-message-stream/clarify-hydration.test.tsx @@ -5,7 +5,7 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import type { ClientSessionState } from '@/app/types' import { createClientSessionState } from '@/lib/chat-runtime' -import { clearClarifyRequest } from '@/store/clarify' +import { $clarifyRequests, clearClarifyRequest } from '@/store/clarify' import { onScrollToBottomRequest } from '@/store/thread-scroll' import type { RpcEvent } from '@/types/hermes' @@ -123,6 +123,32 @@ describe('clarify.request stream hydration', () => { expect(scrollToBottom).not.toHaveBeenCalled() }) + it('preserves multi-select through the store and hydrated tool row', async () => { + await mountStream() + + clarifyRequest({ + choices: ['read', 'write'], + multi_select: true, + question: 'Which permissions?', + request_id: 'req-multi' + }) + + expect($clarifyRequests.get()[SID]?.multiSelect).toBe(true) + + const part = clarifyParts()[0] + expect(part?.type).toBe('tool-call') + + if (part?.type !== 'tool-call') { + throw new Error('Expected a hydrated clarify tool call') + } + + expect(part.args).toMatchObject({ + choices: ['read', 'write'], + multi_select: true, + question: 'Which permissions?' + }) + }) + it('merges with the real tool.start row even though its id differs from the request id', async () => { await mountStream() diff --git a/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event.ts b/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event.ts index 19cae17463369..0ba4b2a0db626 100644 --- a/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event.ts +++ b/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event.ts @@ -957,6 +957,7 @@ export function useGatewayEventHandler(deps: GatewayEventDeps) { const question = typeof payload?.question === 'string' ? payload.question : '' const rawChoices = payload?.choices const choices = normalizeChoices(rawChoices) + const multiSelect = payload?.multi_select === true if (requestId && question) { if (rawChoices != null && choices.length === 0) { @@ -967,6 +968,7 @@ export function useGatewayEventHandler(deps: GatewayEventDeps) { requestId, question, choices: choices.length > 0 ? choices : null, + multiSelect, sessionId: sessionId ?? null }) @@ -978,7 +980,15 @@ export function useGatewayEventHandler(deps: GatewayEventDeps) { // choices. Upsert a stable pending clarify tool row from the request // itself so the prompt stays answerable; a real tool.start/complete // with the same request id merges rather than duplicates. - upsertToolCall(sessionId, { args: { choices, question }, name: 'clarify', tool_id: requestId }, 'running') + upsertToolCall( + sessionId, + { + args: { choices, ...(multiSelect ? { multi_select: true } : {}), question }, + name: 'clarify', + tool_id: requestId + }, + 'running' + ) // The transcript only renders the active session, so a background // clarify is otherwise invisible (the row just keeps spinning like diff --git a/apps/desktop/src/components/assistant-ui/clarify-tool.test.tsx b/apps/desktop/src/components/assistant-ui/clarify-tool.test.tsx index 0b80ad68d4d54..4d33a176c26f7 100644 --- a/apps/desktop/src/components/assistant-ui/clarify-tool.test.tsx +++ b/apps/desktop/src/components/assistant-ui/clarify-tool.test.tsx @@ -71,13 +71,14 @@ function liveClarifyProps(choices = ['staging', 'production']): ToolCallMessageP } } -function renderLiveClarify() { +function renderLiveClarify({ multiSelect = false }: { multiSelect?: boolean } = {}) { const request = vi.fn().mockResolvedValue({ ok: true }) $activeSessionId.set('session-1') $gateway.set({ request } as never) setClarifyRequest({ choices: ['staging', 'production'], + multiSelect, question: 'Which deployment target?', requestId: 'request-1', sessionId: 'session-1' @@ -87,6 +88,57 @@ function renderLiveClarify() { return request } +describe('ClarifyTool choice selection', () => { + it('selects independently, deselects and submits multi-select choices as a JSON array', async () => { + const request = renderLiveClarify({ multiSelect: true }) + const staging = screen.getByRole('button', { name: /staging/ }) + const production = screen.getByRole('button', { name: /production/ }) + + fireEvent.click(staging) + fireEvent.click(production) + expect(staging.getAttribute('aria-pressed')).toBe('true') + expect(production.getAttribute('aria-pressed')).toBe('true') + + fireEvent.keyDown(window, { key: 'ArrowDown' }) + expect(staging.getAttribute('aria-pressed')).toBe('true') + expect(production.getAttribute('aria-pressed')).toBe('true') + + fireEvent.click(staging) + expect(staging.getAttribute('aria-pressed')).toBe('false') + fireEvent.click(staging) + + fireEvent.click(screen.getByRole('button', { name: /Continue/ })) + + await waitFor(() => { + expect(request).toHaveBeenCalledWith('clarify.respond', { + answer: JSON.stringify(['production', 'staging']), + request_id: 'request-1' + }) + }) + }) + + it('keeps single-select replacement and plain-string submission', async () => { + const request = renderLiveClarify() + const staging = screen.getByRole('button', { name: /staging/ }) + const production = screen.getByRole('button', { name: /production/ }) + + fireEvent.click(staging) + fireEvent.click(production) + + expect(staging.getAttribute('aria-pressed')).toBe('false') + expect(production.getAttribute('aria-pressed')).toBe('true') + + fireEvent.click(screen.getByRole('button', { name: /Continue/ })) + + await waitFor(() => { + expect(request).toHaveBeenCalledWith('clarify.respond', { + answer: 'production', + request_id: 'request-1' + }) + }) + }) +}) + describe('readClarifyResult', () => { it('reads question + user_response from the tool JSON payload', () => { expect( @@ -348,6 +400,7 @@ describe('ClarifyTool pending marker', () => { $gateway.set({ request: vi.fn().mockResolvedValue({ ok: true }) } as never) setClarifyRequest({ choices: null, + multiSelect: false, question: 'Anything else?', requestId: 'request-1', sessionId: 'session-1' diff --git a/apps/desktop/src/components/assistant-ui/clarify-tool.tsx b/apps/desktop/src/components/assistant-ui/clarify-tool.tsx index f970c6d5727fb..255729de780ab 100644 --- a/apps/desktop/src/components/assistant-ui/clarify-tool.tsx +++ b/apps/desktop/src/components/assistant-ui/clarify-tool.tsx @@ -42,6 +42,7 @@ import { parseMaybeObject } from './tool/fallback-model/format' interface ClarifyArgs { question?: string choices?: string[] | null + multiSelect?: boolean } interface ClarifyResult { @@ -73,7 +74,8 @@ function readClarifyArgs(args: unknown): ClarifyArgs { return { question, - choices: choices.length > 0 ? choices : null + choices: choices.length > 0 ? choices : null, + multiSelect: row.multi_select === true } } @@ -167,7 +169,7 @@ function ChoiceButton({ disabled, keyShortcuts, onClick, - selected = false, + selected, title }: { active?: boolean @@ -192,6 +194,7 @@ function ChoiceButton({ {accessory &&
{accessory}
} From 89fa8298022495c9ad15241ccc78f5bc84716a6a Mon Sep 17 00:00:00 2001 From: Sora-bluesky Date: Tue, 11 Aug 2026 17:19:33 +0900 Subject: [PATCH 049/376] test(desktop): pin the tail-only contract for the loading and stall indicators Regression coverage for #68634. The indicator family mounts only on the thread's any-role tail (ba756333): a running bubble that is not the tail stays silent, even when only a user or system row trails it, and the optimistic-placeholder flow renders exactly one status row, the placeholder's own. Mutation-checked: relaxing the mount gate to a last-assistant walk fails the two silence cases, and removing the gate fails all five. --- .../thread/duplicate-stall-indicator.test.tsx | 260 ++++++++++++++++++ 1 file changed, 260 insertions(+) create mode 100644 apps/desktop/src/components/assistant-ui/thread/duplicate-stall-indicator.test.tsx diff --git a/apps/desktop/src/components/assistant-ui/thread/duplicate-stall-indicator.test.tsx b/apps/desktop/src/components/assistant-ui/thread/duplicate-stall-indicator.test.tsx new file mode 100644 index 0000000000000..024c8eab39149 --- /dev/null +++ b/apps/desktop/src/components/assistant-ui/thread/duplicate-stall-indicator.test.tsx @@ -0,0 +1,260 @@ +// Loading and stall indicators mount only on the thread's last message. +// The tail-only gate from ba756333 keeps non-tail running bubbles silent, +// including assistants followed only by a user or system row. The optimistic +// placeholder flow renders exactly one status row. These contracts pin the +// duplicate-indicator regression tracked in #68634. +import { AssistantRuntimeProvider, type ThreadMessage, useExternalStoreRuntime } from '@assistant-ui/react' +import { act, cleanup, render, screen } from '@testing-library/react' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +import { __resetElapsedTimerRegistryForTests } from '@/components/chat/activity-timer' +import { setSessionCompacting } from '@/store/compaction' +import { $activeSessionId, $turnStartedAt } from '@/store/session' + +import { Thread } from '.' + +// Layout/observer stubs mirrored from streaming.test.tsx. jsdom has no +// ResizeObserver, rAF, or real layout, and the Thread scroll container needs +// non-zero dimensions to mount without throwing. +class TestResizeObserver { + observe() {} + unobserve() {} + disconnect() {} +} + +vi.stubGlobal('ResizeObserver', TestResizeObserver) +vi.stubGlobal('requestAnimationFrame', (callback: FrameRequestCallback) => + window.setTimeout(() => callback(performance.now()), 0) +) +vi.stubGlobal('cancelAnimationFrame', (id: number) => window.clearTimeout(id)) +vi.stubGlobal('CSS', { escape: (str: string) => str }) + +Element.prototype.scrollTo = function scrollTo() {} + +Element.prototype.animate = function animate() { + return { + cancel: () => {}, + finished: Promise.resolve() + } as unknown as Animation +} + +function stubOffsetDimension( + prop: 'offsetHeight' | 'offsetWidth', + clientProp: 'clientHeight' | 'clientWidth', + fallback: number +) { + const previous = Object.getOwnPropertyDescriptor(HTMLElement.prototype, prop) + + Object.defineProperty(HTMLElement.prototype, prop, { + configurable: true, + get() { + return previous?.get?.call(this) || (this as HTMLElement)[clientProp] || fallback + } + }) +} + +stubOffsetDimension('offsetWidth', 'clientWidth', 800) +stubOffsetDimension('offsetHeight', 'clientHeight', 600) + +const createdAt = new Date('2026-05-01T00:00:00.000Z') +const sessionId = 'session-68634' + +function userMessage(id: string, text: string): ThreadMessage { + return { + id, + role: 'user', + content: [{ type: 'text', text }], + attachments: [], + createdAt, + metadata: { custom: {} } + } as ThreadMessage +} + +// This shape mirrors the `/steer` note appended by appendSessionTextMessage +// in apps/desktop/src/app/session/hooks/use-prompt-actions/index.ts. +function systemMessage(id: string, text: string): ThreadMessage { + return { + id, + role: 'system', + content: [{ type: 'text', text }], + createdAt, + metadata: { custom: {} } + } as ThreadMessage +} + +function runningAssistantMessage(id: string, text: string): ThreadMessage { + return { + id, + role: 'assistant', + content: [{ type: 'text', text }], + status: { type: 'running' }, + createdAt, + metadata: { + unstable_state: null, + unstable_annotations: [], + unstable_data: [], + steps: [], + custom: {} + } + } as ThreadMessage +} + +function Harness({ messages, isRunning = false }: { messages: ThreadMessage[]; isRunning?: boolean }) { + // isRunning: false at the runtime level. Per-message `status: {type: + // 'running'}` is what drives StreamStallIndicator mounting. + // Passing isRunning: true makes useExternalStoreRuntime auto-append a + // synthetic empty trailing assistant placeholder whenever the last message + // is not already a running assistant, such as the trailing user prompt + // cases below. That is the real production flow. The isRunning:true tests + // prove that the placeholder is treated as the tail and renders its own + // loading row while the real bubble's stall row stays silent. + const runtime = useExternalStoreRuntime({ + messages, + isRunning, + onNew: async () => {} + }) + + return ( + + + + ) +} + +describe('StreamStallIndicator tail gating (#68634)', () => { + beforeEach(() => { + vi.useFakeTimers() + vi.setSystemTime(new Date('2026-01-01T00:00:00.000Z')) + __resetElapsedTimerRegistryForTests() + $activeSessionId.set(sessionId) + $turnStartedAt.set(Date.now()) + setSessionCompacting(sessionId, true) + }) + + afterEach(() => { + cleanup() + setSessionCompacting(sessionId, false) + $activeSessionId.set(null) + $turnStartedAt.set(null) + __resetElapsedTimerRegistryForTests() + vi.useRealTimers() + }) + + it('renders exactly one indicator, on the later bubble, when two assistant bubbles are running with content', () => { + const { container } = render( + + ) + + act(() => { + vi.advanceTimersByTime(5_000) + }) + + const indicators = screen.getAllByRole('status', { name: 'Summarizing thread' }) + expect(indicators.length).toBe(1) + + const roots = container.querySelectorAll('[data-slot="aui_assistant-message-root"]') + expect(roots.length).toBe(2) + // The second assistant root is also the thread's any-role tail. + expect(roots[0]?.querySelector('[data-slot="aui_stream-stall"]')).toBeNull() + expect(roots[1]?.querySelector('[data-slot="aui_stream-stall"]')).not.toBeNull() + }) + + it('keeps a running assistant silent when a queued user prompt trails it and the runtime is idle', () => { + const { container } = render( + + ) + + act(() => { + vi.advanceTimersByTime(5_000) + }) + + expect(container.querySelector('[data-slot="aui_response-loading"]')).toBeNull() + expect(container.querySelector('[data-slot="aui_stream-stall"]')).toBeNull() + }) + + it('keeps a running assistant silent when a steer system note trails it and the runtime is idle', () => { + const { container } = render( + + ) + + act(() => { + vi.advanceTimersByTime(5_000) + }) + + expect(container.querySelector('[data-slot="aui_response-loading"]')).toBeNull() + expect(container.querySelector('[data-slot="aui_stream-stall"]')).toBeNull() + }) + + // In the production flow, isRunning: true with a trailing queued user prompt + // makes the runtime append an empty optimistic assistant placeholder after + // the real running bubble. The placeholder renders ResponseLoadingIndicator. + // During compaction, that row carries the same accessible label as the stall + // indicator. The real non-tail bubble must remain silent so there is exactly + // one status row. + it('still renders the indicator when the runtime appends an optimistic placeholder (isRunning:true)', () => { + render( + + ) + + act(() => { + vi.advanceTimersByTime(5_000) + }) + + const indicators = screen.getAllByRole('status', { name: 'Summarizing thread' }) + expect(indicators.length).toBe(1) + // The surviving row belongs to the placeholder, while the real running + // bubble's stall row stays silent. + expect(document.querySelectorAll('[data-slot="aui_response-loading"]').length).toBe(1) + expect(document.querySelectorAll('[data-slot="aui_stream-stall"]').length).toBe(0) + }) + + // Outside compaction, the placeholder uses the plain loading label and the + // real bubble's stall row must remain silent after the stall threshold. + it('keeps a single status row for the placeholder outside compaction (isRunning:true)', () => { + setSessionCompacting(sessionId, false) + + render( + + ) + + act(() => { + vi.advanceTimersByTime(5_000) + }) + + expect(document.querySelectorAll('[data-slot="aui_response-loading"]').length).toBe(1) + expect(document.querySelectorAll('[data-slot="aui_stream-stall"]').length).toBe(0) + }) +}) From 0e7151ceae1d7abfec09725df976c12aac08dbc2 Mon Sep 17 00:00:00 2001 From: LeonSGP43 Date: Tue, 9 Jun 2026 23:19:32 +0800 Subject: [PATCH 050/376] fix(todo): keep active step ahead of pending rows --- tests/tools/test_todo_tool.py | 37 ++++++++++++++++++++++++++++++++--- tools/todo_tool.py | 31 +++++++++++++++++++++++++++-- 2 files changed, 63 insertions(+), 5 deletions(-) diff --git a/tests/tools/test_todo_tool.py b/tests/tools/test_todo_tool.py index 1dc19b88c77a4..c7b41f09c7bef 100644 --- a/tests/tools/test_todo_tool.py +++ b/tests/tools/test_todo_tool.py @@ -14,8 +14,9 @@ def test_write_replaces_list(self): ] result = store.write(items) assert len(result) == 2 - assert result[0]["id"] == "1" - assert result[1]["status"] == "in_progress" + assert result[0]["id"] == "2" + assert result[0]["status"] == "in_progress" + assert result[1]["id"] == "1" def test_write_deduplicates_duplicate_ids(self): @@ -26,8 +27,21 @@ def test_write_deduplicates_duplicate_ids(self): {"id": "1", "content": "Latest version", "status": "in_progress"}, ]) assert result == [ - {"id": "2", "content": "Other task", "status": "pending"}, {"id": "1", "content": "Latest version", "status": "in_progress"}, + {"id": "2", "content": "Other task", "status": "pending"}, + ] + + def test_write_moves_active_item_before_earlier_pending_step(self): + store = TodoStore() + result = store.write([ + {"id": "1", "content": "Already done", "status": "completed"}, + {"id": "2", "content": "Verify freed space", "status": "pending"}, + {"id": "3", "content": "Move archives to Trash", "status": "in_progress"}, + ]) + assert result == [ + {"id": "1", "content": "Already done", "status": "completed"}, + {"id": "3", "content": "Move archives to Trash", "status": "in_progress"}, + {"id": "2", "content": "Verify freed space", "status": "pending"}, ] @@ -91,6 +105,23 @@ def test_merge_appends_new(self): items = store.read() assert len(items) == 2 + def test_merge_reorders_active_item_ahead_of_earlier_pending_step(self): + store = TodoStore() + store.write([ + {"id": "1", "content": "Completed", "status": "completed"}, + {"id": "2", "content": "Verify freed space", "status": "pending"}, + {"id": "3", "content": "Move archives to Trash", "status": "pending"}, + ]) + result = store.write( + [{"id": "3", "status": "in_progress"}], + merge=True, + ) + assert result == [ + {"id": "1", "content": "Completed", "status": "completed"}, + {"id": "3", "content": "Move archives to Trash", "status": "in_progress"}, + {"id": "2", "content": "Verify freed space", "status": "pending"}, + ] + class TestTodoToolFunction: def test_read_mode(self): diff --git a/tools/todo_tool.py b/tools/todo_tool.py index 13b5fd4aad939..1eea334f43831 100644 --- a/tools/todo_tool.py +++ b/tools/todo_tool.py @@ -67,7 +67,9 @@ def write(self, todos: List[Dict[str, Any]], merge: bool = False) -> List[Dict[s """ if not merge: # Replace mode: new list entirely - self._items = [self._validate(t) for t in self._dedupe_by_id(todos)] + self._items = self._normalize_order( + [self._validate(t) for t in self._dedupe_by_id(todos)] + ) else: # Merge mode: update existing items by id, append new ones existing = {item["id"]: item for item in self._items} @@ -97,7 +99,7 @@ def write(self, todos: List[Dict[str, Any]], merge: bool = False) -> List[Dict[s if current["id"] not in seen: rebuilt.append(current) seen.add(current["id"]) - self._items = rebuilt + self._items = self._normalize_order(rebuilt) # Bound total item count so a replayed/oversized list can't grow the # re-injection block without limit. Keep the highest-priority head # (list order is priority). @@ -200,6 +202,31 @@ def _dedupe_by_id(todos: List[Dict[str, Any]]) -> List[Dict[str, Any]]: last_index[item_id] = i return [todos[i] for i in sorted(last_index.values())] + @staticmethod + def _normalize_order(items: List[Dict[str, str]]) -> List[Dict[str, str]]: + """Lift the active step ahead of any earlier unfinished placeholders.""" + active_index = next( + (i for i, item in enumerate(items) if item["status"] == "in_progress"), + None, + ) + if active_index is None: + return items + + pending_index = next( + ( + i for i, item in enumerate(items[:active_index]) + if item["status"] == "pending" + ), + None, + ) + if pending_index is None: + return items + + normalized = items.copy() + active_item = normalized.pop(active_index) + normalized.insert(pending_index, active_item) + return normalized + def todo_tool( todos: Optional[List[Dict[str, Any]]] = None, From 8bd83a9f7cf54c7184a913822a05b572a5a8c3f1 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 18:56:23 -0700 Subject: [PATCH 051/376] test(gateway): pin get_update_result in metadata-mirror snapshot test The test compares two _session_info snapshots taken at different times; the background update-check thread can complete between them and flip update_behind (None -> -1), making the equality assertion flaky once the suite runs long enough. Pin the value via monkeypatch. --- tests/test_tui_gateway_server.py | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/tests/test_tui_gateway_server.py b/tests/test_tui_gateway_server.py index 12e9e49e87925..cfb26d270ae41 100644 --- a/tests/test_tui_gateway_server.py +++ b/tests/test_tui_gateway_server.py @@ -402,6 +402,13 @@ def start(self): def test_compute_host_turn_end_updates_metadata_mirror(monkeypatch): + # _session_info embeds get_update_result(), whose value flips whenever the + # background update-check thread happens to finish. This test compares two + # snapshots taken at different times, so pin the value to keep it + # deterministic regardless of how long the preceding tests ran. + import hermes_cli.banner as _banner + + monkeypatch.setattr(_banner, "get_update_result", lambda timeout=0.5: None) session = _session( agent=None, agent_ready=threading.Event(), From cf30d895866b61d463e2ce7677a9a020611fb7f1 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 18:58:05 -0700 Subject: [PATCH 052/376] fix(desktop): import MemoryRouter from react-router in collapsed-indicator test The repo standardized on react-router (see find-bar.test.tsx); react-router-dom is not installed, so the salvaged test failed to transform. --- .../app/chat/composer/status-stack/collapsed-indicator.test.tsx | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/apps/desktop/src/app/chat/composer/status-stack/collapsed-indicator.test.tsx b/apps/desktop/src/app/chat/composer/status-stack/collapsed-indicator.test.tsx index 1b528d3827afd..0006c4ab25b3a 100644 --- a/apps/desktop/src/app/chat/composer/status-stack/collapsed-indicator.test.tsx +++ b/apps/desktop/src/app/chat/composer/status-stack/collapsed-indicator.test.tsx @@ -1,5 +1,5 @@ import { cleanup, fireEvent, render, screen } from '@testing-library/react' -import { MemoryRouter } from 'react-router-dom' +import { MemoryRouter } from 'react-router' import { afterEach, beforeAll, describe, expect, it, vi } from 'vitest' import { $todosBySession } from '@/store/todos' From 99033ab1f0592ccdc394f6cdcbfaec2ff94bb210 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 18:58:24 -0700 Subject: [PATCH 053/376] chore: map contributor email for PINKIIILQWQ --- contributors/emails/pink@macmini-hermes.local | 1 + 1 file changed, 1 insertion(+) create mode 100644 contributors/emails/pink@macmini-hermes.local diff --git a/contributors/emails/pink@macmini-hermes.local b/contributors/emails/pink@macmini-hermes.local new file mode 100644 index 0000000000000..24491e0530f62 --- /dev/null +++ b/contributors/emails/pink@macmini-hermes.local @@ -0,0 +1 @@ +PINKIIILQWQ From f8cc6d082e56eacd62f8391feb89d6ddc8237930 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 19:08:39 -0700 Subject: [PATCH 054/376] test(todo): make JSON-string coercion test order-agnostic The type-coercion test pinned index order of todos, which #42649's _normalize_order intentionally changes (in_progress lifts ahead of earlier pending rows). Assert coercion by id instead of position. --- tests/tools/test_todo_tool_type_coercion.py | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/tests/tools/test_todo_tool_type_coercion.py b/tests/tools/test_todo_tool_type_coercion.py index fa70b8c91ab4d..84d03a6781fdb 100644 --- a/tests/tools/test_todo_tool_type_coercion.py +++ b/tests/tools/test_todo_tool_type_coercion.py @@ -23,8 +23,12 @@ def test_json_string_is_parsed_into_list(self): result = json.loads(todo_tool(todos=todos_str, store=store)) assert "error" not in result assert result["summary"]["total"] == 2 - assert result["todos"][0]["id"] == "t1" - assert result["todos"][1]["status"] == "in_progress" + # Order-agnostic: TodoStore._normalize_order may lift the in_progress + # item ahead of earlier pending rows (#42649); this test only pins + # JSON-string coercion, not ordering. + by_id = {t["id"]: t for t in result["todos"]} + assert set(by_id) == {"t1", "t2"} + assert by_id["t2"]["status"] == "in_progress" def test_non_list_non_string_returns_error(self): From d0295754073a1bb2aa2d228ccd90976e65f35067 Mon Sep 17 00:00:00 2001 From: liuhao1024 Date: Tue, 16 Jun 2026 23:25:55 +0800 Subject: [PATCH 055/376] fix(desktop): sort survivors by last_active in mergeSessionPage to prevent stale sidebar order When multiple sessions are active/settled simultaneously, survivors (sessions the server omitted from the fresh page) were prepended as a block in their old relative order from the previous $sessions array. This caused recently-interacted sessions to appear below older ones. Now survivors are sorted by last_active descending and merged into the incoming array at the correct position using a two-pointer merge, so the sidebar always reflects true recency. Fixes #47203 --- apps/desktop/src/store/session.test.ts | 36 ++++++++++++++++++++++++ apps/desktop/src/store/session.ts | 38 +++++++++++++++++++++++++- 2 files changed, 73 insertions(+), 1 deletion(-) diff --git a/apps/desktop/src/store/session.test.ts b/apps/desktop/src/store/session.test.ts index fc2e8967f5f69..be1ea4adbccd8 100644 --- a/apps/desktop/src/store/session.test.ts +++ b/apps/desktop/src/store/session.test.ts @@ -267,6 +267,42 @@ describe('mergeSessionPage', () => { expect(merged.map(s => s.id)).toEqual(['tip-5']) expect(merged[0]?.last_active).toBe(9_000) }) + + it('sorts survivors by last_active so they interleave with incoming instead of forming a stale block', () => { + // Repro of #47203: two survivors (B and C) have different last_active + // timestamps. B settled more recently than C. Without sorting, survivors + // are prepended in their old order from `previous`, which may be stale. + // With sorting, B (more recent) should appear before C. + const previous = [ + session({ id: 'c', last_active: 100 }), + session({ id: 'b', last_active: 200 }), + session({ id: 'a', last_active: 300 }), + ] + // Server returns A (fresh page, order=recent), omits B and C (min_messages=1) + const incoming = [session({ id: 'a', last_active: 300, message_count: 2 })] + + const merged = mergeSessionPage(previous, incoming, ['b', 'c']) + + // B (last_active 200) should come before C (last_active 100) + expect(merged.map(s => s.id)).toEqual(['a', 'b', 'c']) + }) + + it('places a very recent survivor in correct position among incoming sessions', () => { + // A survivor with last_active between two incoming sessions should be + // interleaved, not prepended as a block. + const previous = [ + session({ id: 'survivor', last_active: 150 }), + session({ id: 'old', last_active: 50 }), + ] + const incoming = [ + session({ id: 'newest', last_active: 200 }), + session({ id: 'older', last_active: 100 }), + ] + + const merged = mergeSessionPage(previous, incoming, ['survivor']) + + // survivor (150) should be between newest (200) and older (100) + expect(merged.map(s => s.id)).toEqual(['newest', 'survivor', 'older']) }) describe('touchSessionActivity', () => { diff --git a/apps/desktop/src/store/session.ts b/apps/desktop/src/store/session.ts index c64a194ef59ad..36c872cc44e5a 100644 --- a/apps/desktop/src/store/session.ts +++ b/apps/desktop/src/store/session.ts @@ -437,7 +437,43 @@ export function mergeSessionPage( (keep.has(session.id) || (session._lineage_root_id != null && keep.has(session._lineage_root_id))) ) - return survivors.length ? [...survivors, ...merged] : merged + if (!survivors.length) { + return merged + } + + // Survivors carry their old relative positions from `previous`, which can be + // stale — the server page is the fresh `order=recent` truth. Sort survivors + // by the same effective-recency key the backend sorts by (last_active, then + // started_at, then id) and interleave them into the title-preserving merged + // rows so a retained session lands where recency puts it instead of the + // whole set forming a stale block at the top of the sidebar (fixes #47203). + const recency = (session: SessionInfo): number => session.last_active || session.started_at || 0 + + const bySessionRecency = (a: SessionInfo, b: SessionInfo): number => + recency(b) - recency(a) || (b.started_at ?? 0) - (a.started_at ?? 0) || (a.id < b.id ? 1 : a.id > b.id ? -1 : 0) + + const sortedSurvivors = [...survivors].sort(bySessionRecency) + const interleaved: SessionInfo[] = [] + let survivorIndex = 0 + let mergedIndex = 0 + + while (survivorIndex < sortedSurvivors.length && mergedIndex < merged.length) { + if (bySessionRecency(sortedSurvivors[survivorIndex], merged[mergedIndex]) <= 0) { + interleaved.push(sortedSurvivors[survivorIndex++]) + } else { + interleaved.push(merged[mergedIndex++]) + } + } + + while (survivorIndex < sortedSurvivors.length) { + interleaved.push(sortedSurvivors[survivorIndex++]) + } + + while (mergedIndex < merged.length) { + interleaved.push(merged[mergedIndex++]) + } + + return interleaved } /** Raise a session in recents on user send (before stream / turn resolve). */ From 9247f4e1a8cddccdf0990c23efc2e2be7e251ed5 Mon Sep 17 00:00:00 2001 From: james47 <220877172+james47kjv@users.noreply.github.com> Date: Fri, 14 Aug 2026 17:11:52 +0000 Subject: [PATCH 056/376] fix(desktop): compare pinned/archived in the session list signature MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `refreshSessions` swaps the session page into `$sessions` only when `sameCronSignature` reports a change, and that signature compared row content — id, lineage root, title, source, profile, preview, message_count, last_active, ended_at — but not row state. A page whose only delta was `pinned` was judged identical and discarded, so the row cached in the atom kept its old flag indefinitely. An idle conversation never moves any of the compared fields again, which is exactly the kind a user goes and unpins. `session-pin-sync` treats that row as authoritative. Its write guard (daeedf67c) is released by a page that CONFIRMS the value it wrote, and falls back to letting the server win once WRITE_GUARD_MS elapses with no confirmation. Because the confirming page was filtered out one layer up, the fallback was the only branch that ever ran: ~10s after an unpin the next reconcile read the frozen `pinned: true` row and called pinSession() again. Adoption marks the id `mirrored`, so the push pass never corrected the backend either — the local pin set and sessions.pinned drifted apart permanently, which is why four of five pins rendered in the sidebar read pinned=0 in state.db. Compare both flags so a pin-only page reaches the atom. That restores the guard's confirm path and makes WRITE_GUARD_MS a backstop again rather than the load-bearing branch. `archived` is included for the same reason: it is row state a consumer reads. Neither flag moves outside a deliberate user action, so the churn the gate exists to prevent is unaffected. The existing `releases the guard once a page confirms the written value` test passes on main because it hands `$sessions` the confirming page directly — the gap was in the pipeline that decides whether such a page is ever delivered. Fixes #76919 Co-Authored-By: Claude Opus 5 (1M context) --- .../src/lib/session-signatures.test.ts | 25 ++++++++++++++++++- apps/desktop/src/lib/session-signatures.ts | 9 ++++++- 2 files changed, 32 insertions(+), 2 deletions(-) diff --git a/apps/desktop/src/lib/session-signatures.test.ts b/apps/desktop/src/lib/session-signatures.test.ts index aeecd19b61fbc..0c4c280c8ef3b 100644 --- a/apps/desktop/src/lib/session-signatures.test.ts +++ b/apps/desktop/src/lib/session-signatures.test.ts @@ -4,7 +4,8 @@ import type { SessionInfo } from '@/hermes' import { sameCronSignature, sessionMessagesSignature } from './session-signatures' -const session = (id: string, title: string | null): SessionInfo => ({ id, title }) as SessionInfo +const session = (id: string, title: string | null, extra: Partial = {}): SessionInfo => + ({ id, title, ...extra }) as SessionInfo describe('sameCronSignature', () => { it('is false when the lengths differ', () => { @@ -28,6 +29,28 @@ describe('sameCronSignature', () => { const b = [session('b', 't'), session('a', 't')] expect(sameCronSignature(a, b)).toBe(false) }) + + // A pin-only page must reach $sessions: session-pin-sync treats the row as + // authoritative and releases its write guard when a page confirms the value + // it wrote. Gating that page out froze the row and re-pinned what the user + // had just unpinned (#76919). + it('is false when only the pinned flag changed', () => { + const a = [session('a', 't', { pinned: true })] + const b = [session('a', 't', { pinned: false })] + expect(sameCronSignature(a, b)).toBe(false) + }) + + it('is false when only the archived flag changed', () => { + const a = [session('a', 't', { archived: false })] + const b = [session('a', 't', { archived: true })] + expect(sameCronSignature(a, b)).toBe(false) + }) + + it('is true when both flags match', () => { + const a = [session('a', 't', { archived: false, pinned: true })] + const b = [session('a', 't', { archived: false, pinned: true })] + expect(sameCronSignature(a, b)).toBe(true) + }) }) describe('sessionMessagesSignature', () => { diff --git a/apps/desktop/src/lib/session-signatures.ts b/apps/desktop/src/lib/session-signatures.ts index 4ef20e1fab19e..fac47254226c9 100644 --- a/apps/desktop/src/lib/session-signatures.ts +++ b/apps/desktop/src/lib/session-signatures.ts @@ -23,7 +23,14 @@ export function sameCronSignature(a: SessionInfo[], b: SessionInfo[]): boolean { session.preview === other.preview && session.message_count === other.message_count && session.last_active === other.last_active && - session.ended_at === other.ended_at + session.ended_at === other.ended_at && + // Row STATE, not just row content: session-pin-sync reconciles the + // sidebar's pins against `pinned` on the rows in this atom, so a page + // whose only delta is a flag has to swap in or the reconciler reads a + // frozen copy forever. An idle conversation never moves any of the + // fields above again, which is exactly when a pin gets toggled (#76919). + session.pinned === other.pinned && + session.archived === other.archived ) }) } From b6d2f15b6cac1ab61cc4d597d0dcb2198edd5cac Mon Sep 17 00:00:00 2001 From: Jreevo Date: Wed, 29 Jul 2026 07:27:10 -0700 Subject: [PATCH 057/376] fix(desktop): dedupe persisted sidebar order ids to stop duplicate repo headers The desktop sidebar persists repo/lane order in localStorage (hermes.desktop.workspaceParentOrder / workspaceOrder). If that saved list ever contains the same id twice, orderByIds() pushes the matching item once per occurrence, rendering the same repo header twice inside a project. reconcileFreshFirst() then preserves the duplicates, so the corruption self-perpetuates across restarts and storage clears. Verified on a live install: the persisted workspaceParentOrder contained the same repo path at two positions, the backend project tree was clean, and the duplicated header matched the duplicated id. Treat persisted UI order as untrusted input: orderByIds() now skips ids it has already emitted, and reconcileFreshFirst() dedupes the retained tail so the next persist writes a clean list (self-healing). --- .../src/app/chat/sidebar/order.test.ts | 9 +++++++ apps/desktop/src/app/chat/sidebar/order.ts | 26 +++++++++++++++---- 2 files changed, 30 insertions(+), 5 deletions(-) diff --git a/apps/desktop/src/app/chat/sidebar/order.test.ts b/apps/desktop/src/app/chat/sidebar/order.test.ts index 37a4b36df96de..0c84107b4dbf7 100644 --- a/apps/desktop/src/app/chat/sidebar/order.test.ts +++ b/apps/desktop/src/app/chat/sidebar/order.test.ts @@ -51,6 +51,11 @@ describe('orderByIds', () => { expect(orderByIds(items, id, ['b', 'a'])).toEqual([{ id: 'fresh' }, { id: 'b' }, { id: 'a' }]) }) + it('never duplicates an item when the persisted order repeats its id', () => { + const items = [{ id: 'a' }, { id: 'b' }] + expect(orderByIds(items, id, ['a', 'a', 'b'])).toEqual([{ id: 'a' }, { id: 'b' }]) + }) + it('keeps a newly-loaded older page below the hand-picked order', () => { // Callers pass recency-sorted lists, so an unknown id BELOW the ordered // ones is an older page that just loaded — hoisting it to the top was @@ -94,6 +99,10 @@ describe('reconcileOrderIds', () => { it('puts newly-seen ids ahead of the retained saved order', () => { expect(reconcileOrderIds(['fresh', 'a', 'b'], ['b', 'a', 'gone'])).toEqual(['fresh', 'b', 'a']) }) + + it('dedupes a corrupted saved order instead of perpetuating it', () => { + expect(reconcileOrderIds(['a', 'b'], ['a', 'a', 'b'])).toEqual(['a', 'b']) + }) }) describe('sameIds', () => { diff --git a/apps/desktop/src/app/chat/sidebar/order.ts b/apps/desktop/src/app/chat/sidebar/order.ts index c65a53dc0308b..625cb083f3106 100644 --- a/apps/desktop/src/app/chat/sidebar/order.ts +++ b/apps/desktop/src/app/chat/sidebar/order.ts @@ -35,9 +35,13 @@ function mergeFreshByPosition(currentIds: string[], keptIds: string[]): string[] export function reconcileFreshFirst(currentIds: string[], orderIds: string[]): string[] { const current = new Set(currentIds) + // Dedupe both inputs: a corrupted persisted order (same id twice) must not + // self-perpetuate through reconcile, and duplicate live ids (e.g. the same + // repo surfacing under several projects) must not be written back into the + // saved order — either one paints as duplicate headers (#73314). return mergeFreshByPosition( - currentIds, - orderIds.filter(id => current.has(id)) + [...new Set(currentIds)], + [...new Set(orderIds.filter(id => current.has(id)))] ) } @@ -74,7 +78,10 @@ export function orderByIds(items: T[], getId: (item: T) => string, orderIds: for (const id of orderIds) { const item = byId.get(id) - if (item) { + // Guard against duplicates in the persisted order: pushing the same item + // twice renders the row/header twice (e.g. two identical repo headers + // under one project). + if (item && !seen.has(id)) { ordered.push(item) seen.add(id) } @@ -89,10 +96,17 @@ export function orderByIds(items: T[], getId: (item: T) => string, orderIds: const older: T[] = [] items.forEach((item, index) => { - if (seen.has(getId(item))) { + const itemId = getId(item) + + // `seen` doubles as the duplicate guard for live items: two rows carrying + // the same id (e.g. one repo surfacing under several projects) must render + // once, not once per occurrence (#73314). + if (seen.has(itemId)) { return } + seen.add(itemId) + if (firstOrdered >= 0 && index < firstOrdered) { newer.push(item) } else { @@ -119,7 +133,9 @@ export function reconcileOrderIds(currentIds: string[], orderIds: string[]): str } if (!orderIds.length) { - return currentIds + // Still dedupe: persisting duplicate live ids here is what seeded the + // #73314 feedback loop in the first place. + return [...new Set(currentIds)] } return reconcileFreshFirst(currentIds, orderIds) From dac5f86313a59d5d5968c81a7a305a1dd6a99793 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 18:57:59 -0700 Subject: [PATCH 058/376] fix(desktop): keep the session-list merge/dedup/order pipeline invariant-consistent Completes the sidebar order/visibility class on top of the three salvaged contributor commits: - mergeSessionPage (#47203): interleave survivors against the title-preserving merged rows using the backend's effective-recency key (last_active with a started_at fallback), tie-preferring survivors so keep-set rows with no timestamps retain the old prepend contract. - sidebar order helpers (#73314): dedupe live ids as well as persisted ids in reconcileFreshFirst/reconcileOrderIds/orderByIds so the shared-git-root flatMap path can neither render one repo once per project nor write the duplicates back into localStorage (the persisted feedback loop). - Pinned section (#85969): resolvePinnedSessions falls back to the server `pinned` flag when the localStorage pin set is cold or clobbered, so a backend-pinned row is never simultaneously filtered out of every list and absent from the Pinned section (the "session vanishes entirely" state). session-pin-sync then adopts the pin locally on its next reconcile. Regression tests cover survivor interleaving with optimistic bumps and started_at fallback, duplicate live/persisted id dedup, and pin resolution fallback (cold cache, lineage-root pins, undefined flag on old backends). --- apps/desktop/src/app/chat/sidebar/index.tsx | 29 +++++----- apps/desktop/src/app/chat/sidebar/order.ts | 5 +- .../app/chat/sidebar/session-index.test.ts | 51 ++++++++++++++++- .../src/app/chat/sidebar/session-index.ts | 43 ++++++++++++++ apps/desktop/src/store/session.test.ts | 57 ++++++++++++++++--- apps/desktop/src/store/session.ts | 14 ++--- 6 files changed, 161 insertions(+), 38 deletions(-) diff --git a/apps/desktop/src/app/chat/sidebar/index.tsx b/apps/desktop/src/app/chat/sidebar/index.tsx index 7cab99808a50f..6377cd75b8fd7 100644 --- a/apps/desktop/src/app/chat/sidebar/index.tsx +++ b/apps/desktop/src/app/chat/sidebar/index.tsx @@ -161,7 +161,7 @@ import { } from './projects' import { WorktreeDialog } from './projects/worktree-dialog' import { SidebarBlankState, SidebarPinnedEmptyState, SidebarSessionSkeletons } from './section-states' -import { buildSessionByAnyId } from './session-index' +import { buildSessionByAnyId, resolvePinnedSessions } from './session-index' import { SidebarSessionsSection, VIRTUALIZE_THRESHOLD } from './sessions-section' import { CONTEXT_SPLIT_KIT, SplitSubmenu } from './split-submenu' @@ -510,21 +510,18 @@ export function ChatSidebar({ [visibleSessions, cronSessions, messagingSessions] ) - const pinnedSessions = useMemo(() => { - const seen = new Set() - const out: SessionInfo[] = [] - - for (const pinId of pinnedSessionIds) { - const session = sessionByAnyId.get(pinId) - - if (session && !seen.has(session.id)) { - seen.add(session.id) - out.push(session) - } - } - - return out - }, [pinnedSessionIds, sessionByAnyId]) + // Local pin ids first (hand-picked order), then server-flagged pins the + // local set doesn't know about — a backend `pinned=1` row must never be + // invisible just because localStorage is cold or was clobbered (#85969). + const pinnedSessions = useMemo( + () => + resolvePinnedSessions(pinnedSessionIds, sessionByAnyId, [ + ...visibleSessions, + ...cronSessions, + ...messagingSessions + ]), + [pinnedSessionIds, sessionByAnyId, visibleSessions, cronSessions, messagingSessions] + ) // Every id a pin is reachable under: the raw stored ids, plus BOTH identities // of each session we resolved one to. A pin is stored on the durable lineage diff --git a/apps/desktop/src/app/chat/sidebar/order.ts b/apps/desktop/src/app/chat/sidebar/order.ts index 625cb083f3106..0796b55ddff7b 100644 --- a/apps/desktop/src/app/chat/sidebar/order.ts +++ b/apps/desktop/src/app/chat/sidebar/order.ts @@ -39,10 +39,7 @@ export function reconcileFreshFirst(currentIds: string[], orderIds: string[]): s // self-perpetuate through reconcile, and duplicate live ids (e.g. the same // repo surfacing under several projects) must not be written back into the // saved order — either one paints as duplicate headers (#73314). - return mergeFreshByPosition( - [...new Set(currentIds)], - [...new Set(orderIds.filter(id => current.has(id)))] - ) + return mergeFreshByPosition([...new Set(currentIds)], [...new Set(orderIds.filter(id => current.has(id)))]) } export function resolveManualSessionOrderIds(currentIds: string[], orderIds: string[], manual: boolean): string[] { diff --git a/apps/desktop/src/app/chat/sidebar/session-index.test.ts b/apps/desktop/src/app/chat/sidebar/session-index.test.ts index bee60a646037c..a63b484b5776e 100644 --- a/apps/desktop/src/app/chat/sidebar/session-index.test.ts +++ b/apps/desktop/src/app/chat/sidebar/session-index.test.ts @@ -2,7 +2,7 @@ import { describe, expect, it } from 'vitest' import type { SessionInfo } from '@/types/hermes' -import { buildSessionByAnyId } from './session-index' +import { buildSessionByAnyId, resolvePinnedSessions } from './session-index' const row = (id: string, extra: Partial = {}): SessionInfo => ({ id, message_count: 1, source: 'cli', started_at: 0, title: id, ...extra }) as SessionInfo @@ -46,3 +46,52 @@ describe('buildSessionByAnyId', () => { expect(index.get('root')?.id).toBe('root') }) }) + +describe('resolvePinnedSessions', () => { + it('resolves local pin ids in their hand-picked order', () => { + const sessions = [row('a'), row('b'), row('c')] + const index = buildSessionByAnyId(sessions, [], []) + + expect(resolvePinnedSessions(['c', 'a'], index, sessions).map(s => s.id)).toEqual(['c', 'a']) + }) + + it('falls back to the server pinned flag when localStorage is cold (#85969)', () => { + // Backend says pinned=1 but the local pin set is empty (cold localStorage + // after a reload, pin from another client, clobbered persist). Every other + // list filters the row out as pinned, so if the Pinned section can't + // resolve it the session vanishes from the sidebar entirely. + const sessions = [row('a', { pinned: true }), row('b', { pinned: false })] + const index = buildSessionByAnyId(sessions, [], []) + + expect(resolvePinnedSessions([], index, sessions).map(s => s.id)).toEqual(['a']) + }) + + it('does not duplicate a session held both locally and server-side', () => { + const sessions = [row('a', { pinned: true })] + const index = buildSessionByAnyId(sessions, [], []) + + expect(resolvePinnedSessions(['a'], index, sessions).map(s => s.id)).toEqual(['a']) + }) + + it('does not duplicate a server-pinned row whose pin is stored on the lineage root', () => { + const sessions = [row('tip', { _lineage_root_id: 'root', pinned: true })] + const index = buildSessionByAnyId(sessions, [], []) + + expect(resolvePinnedSessions(['root'], index, sessions).map(s => s.id)).toEqual(['tip']) + }) + + it('keeps locally pinned rows ahead of server-only fallback pins', () => { + const sessions = [row('server-pin', { pinned: true }), row('local-pin')] + const index = buildSessionByAnyId(sessions, [], []) + + expect(resolvePinnedSessions(['local-pin'], index, sessions).map(s => s.id)).toEqual(['local-pin', 'server-pin']) + }) + + it('ignores rows from a backend that predates the pinned flag', () => { + // `pinned` undefined means "no opinion", never "pinned". + const sessions = [row('a')] + const index = buildSessionByAnyId(sessions, [], []) + + expect(resolvePinnedSessions([], index, sessions)).toEqual([]) + }) +}) diff --git a/apps/desktop/src/app/chat/sidebar/session-index.ts b/apps/desktop/src/app/chat/sidebar/session-index.ts index 486d977366bf3..be353fe653768 100644 --- a/apps/desktop/src/app/chat/sidebar/session-index.ts +++ b/apps/desktop/src/app/chat/sidebar/session-index.ts @@ -31,3 +31,46 @@ export function buildSessionByAnyId( return map } + +/** + * Resolve the Pinned section's rows: the locally stored pin ids first (in the + * user's hand-picked order), then any row the SERVER flags `pinned` that the + * local set doesn't know about yet. + * + * The local set (`$pinnedSessionIds` in localStorage) is a UI-ordering hint, + * not the source of truth — `sessions.pinned` in the backend's state.db is. + * When the two disagree (cold localStorage after a reload, a pin made from + * another client, a persist that never landed), every other sidebar list + * filters the session out as "pinned" while the Pinned section — resolving + * only local ids — renders empty, so the conversation vanishes from the + * sidebar entirely (#85969). Falling back to the row flag keeps the invariant: + * a session the backend says is pinned is always reachable from the Pinned + * section, whatever the local cache holds. session-pin-sync then adopts the + * pin into the local set on its next reconcile, restoring ordering control. + */ +export function resolvePinnedSessions( + pinnedSessionIds: readonly string[], + sessionByAnyId: Map, + allSessions: readonly SessionInfo[] +): SessionInfo[] { + const seen = new Set() + const out: SessionInfo[] = [] + + for (const pinId of pinnedSessionIds) { + const session = sessionByAnyId.get(pinId) + + if (session && !seen.has(session.id)) { + seen.add(session.id) + out.push(session) + } + } + + for (const session of allSessions) { + if (session.pinned === true && !seen.has(session.id)) { + seen.add(session.id) + out.push(session) + } + } + + return out +} diff --git a/apps/desktop/src/store/session.test.ts b/apps/desktop/src/store/session.test.ts index be1ea4adbccd8..9a80b9231621a 100644 --- a/apps/desktop/src/store/session.test.ts +++ b/apps/desktop/src/store/session.test.ts @@ -276,8 +276,9 @@ describe('mergeSessionPage', () => { const previous = [ session({ id: 'c', last_active: 100 }), session({ id: 'b', last_active: 200 }), - session({ id: 'a', last_active: 300 }), + session({ id: 'a', last_active: 300 }) ] + // Server returns A (fresh page, order=recent), omits B and C (min_messages=1) const incoming = [session({ id: 'a', last_active: 300, message_count: 2 })] @@ -290,19 +291,57 @@ describe('mergeSessionPage', () => { it('places a very recent survivor in correct position among incoming sessions', () => { // A survivor with last_active between two incoming sessions should be // interleaved, not prepended as a block. - const previous = [ - session({ id: 'survivor', last_active: 150 }), - session({ id: 'old', last_active: 50 }), - ] - const incoming = [ - session({ id: 'newest', last_active: 200 }), - session({ id: 'older', last_active: 100 }), - ] + const previous = [session({ id: 'survivor', last_active: 150 }), session({ id: 'old', last_active: 50 })] + + const incoming = [session({ id: 'newest', last_active: 200 }), session({ id: 'older', last_active: 100 })] const merged = mergeSessionPage(previous, incoming, ['survivor']) // survivor (150) should be between newest (200) and older (100) expect(merged.map(s => s.id)).toEqual(['newest', 'survivor', 'older']) + }) + + it('keeps a survivor whose optimistic last_active outranks the whole page on top', () => { + // touchSessionActivity stamps last_active on user-send before the server + // sees the message; that bump must place the survivor by its FRESH time. + const previous = [session({ id: 'typing', last_active: 900 }), session({ id: 'settled', last_active: 100 })] + + const incoming = [session({ id: 'settled', last_active: 100, message_count: 3 })] + + const merged = mergeSessionPage(previous, incoming, ['typing']) + + expect(merged.map(s => s.id)).toEqual(['typing', 'settled']) + }) + + it('falls back to started_at for survivors that have no last_active yet', () => { + // A brand-new session (no persisted message) carries last_active 0; the + // backend's effective-recency key falls back to started_at, so we must + // too, or a fresh draft sinks to the very bottom of the sidebar. + const previous = [ + session({ id: 'draft', last_active: 0, started_at: 500 }), + session({ id: 'other', last_active: 400 }) + ] + + const incoming = [session({ id: 'other', last_active: 400, message_count: 2 })] + + const merged = mergeSessionPage(previous, incoming, ['draft']) + + expect(merged.map(s => s.id)).toEqual(['draft', 'other']) + }) + + it('interleaves against the title-preserving merged rows, not the raw incoming page', () => { + // The optimistic last_active carried onto an incoming row must count for + // its position in the interleave: previous knows 'bumped' was touched at + // 300 even though the server page still reports 100. + const previous = [session({ id: 'survivor', last_active: 200 }), session({ id: 'bumped', last_active: 300 })] + + const incoming = [session({ id: 'bumped', last_active: 100, message_count: 2 })] + + const merged = mergeSessionPage(previous, incoming, ['survivor']) + + expect(merged.map(s => s.id)).toEqual(['bumped', 'survivor']) + expect(merged[0]?.last_active).toBe(300) + }) }) describe('touchSessionActivity', () => { diff --git a/apps/desktop/src/store/session.ts b/apps/desktop/src/store/session.ts index 36c872cc44e5a..8b882b6f66e4d 100644 --- a/apps/desktop/src/store/session.ts +++ b/apps/desktop/src/store/session.ts @@ -443,22 +443,20 @@ export function mergeSessionPage( // Survivors carry their old relative positions from `previous`, which can be // stale — the server page is the fresh `order=recent` truth. Sort survivors - // by the same effective-recency key the backend sorts by (last_active, then - // started_at, then id) and interleave them into the title-preserving merged + // by the same effective-recency key the backend sorts by (last_active with a + // started_at fallback) and interleave them into the title-preserving merged // rows so a retained session lands where recency puts it instead of the // whole set forming a stale block at the top of the sidebar (fixes #47203). - const recency = (session: SessionInfo): number => session.last_active || session.started_at || 0 + // Ties keep the survivor first, matching the old prepend behavior. + const recency = (session: SessionInfo): number => Math.max(session.last_active || 0, session.started_at || 0) - const bySessionRecency = (a: SessionInfo, b: SessionInfo): number => - recency(b) - recency(a) || (b.started_at ?? 0) - (a.started_at ?? 0) || (a.id < b.id ? 1 : a.id > b.id ? -1 : 0) - - const sortedSurvivors = [...survivors].sort(bySessionRecency) + const sortedSurvivors = [...survivors].sort((a, b) => recency(b) - recency(a)) const interleaved: SessionInfo[] = [] let survivorIndex = 0 let mergedIndex = 0 while (survivorIndex < sortedSurvivors.length && mergedIndex < merged.length) { - if (bySessionRecency(sortedSurvivors[survivorIndex], merged[mergedIndex]) <= 0) { + if (recency(sortedSurvivors[survivorIndex]) >= recency(merged[mergedIndex])) { interleaved.push(sortedSurvivors[survivorIndex++]) } else { interleaved.push(merged[mergedIndex++]) From 80645500254a49a18234214908694cbd48b17cc3 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 18:58:18 -0700 Subject: [PATCH 059/376] chore: map contributor email for attribution audit --- contributors/emails/jerry.ytp@gmail.com | 1 + 1 file changed, 1 insertion(+) create mode 100644 contributors/emails/jerry.ytp@gmail.com diff --git a/contributors/emails/jerry.ytp@gmail.com b/contributors/emails/jerry.ytp@gmail.com new file mode 100644 index 0000000000000..4e3c316ed79af --- /dev/null +++ b/contributors/emails/jerry.ytp@gmail.com @@ -0,0 +1 @@ +Jreevo From 5a3b5932305654d168e2cadb4bef26569a62f05f Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 19:29:14 -0700 Subject: [PATCH 060/376] fix(desktop): order mid-turn user messages after the assistant output that predates them MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A message typed while a turn streamed rendered ABOVE assistant output the user had already watched arrive (#73793), and the retired insert-before-the-active-reply fallback could splice the bubble mid-thread — halfway up the chat — when the stream id was missing or stale (#83151). Fix the class at every path that assigns a transcript position to a mid-turn user message: - New shared appendMidTurnUserMessage (rewind.ts): seal the live stream bubble in place (interim), append the correction at the live tail, and clear streamId so post-redirect deltas seed a fresh bubble BELOW the correction. Used by both the primary composer redirect path (use-prompt-actions) and the session-tile steer path (session-tile-actions), replacing the insert-before splice and its last-assistant mid-thread fallback. - appendLiveSessionProjection now projects the resume/reload turn in arrival order (prompt → streamed output → correction → post-redirect output) instead of prompt → corrections → reply, so the projection agrees with the live transcript and messages no longer jump upward on reconnect. With the gateway's new correction_offsets the flat dump is split at each accepted-correction boundary; without offsets the corrections follow the projected reply. - tui_gateway/server.py records correction_offsets (assistant text length at each accepted correction) on the inflight turn and carries them in _inflight_snapshot, only when complete, so resume can rebuild true arrival order. Older gateways/clients degrade cleanly. - preserveLocalPendingTurnMessages and the projection's latest-user-run matcher now treat a live-tail assistant row between the prompt and its correction as part of the same turn's run, so arrival-ordered runs survive refreshes without dropping the prompt. Fixes #73793. Fixes #83151. --- .../src/app/chat/session-tile-actions.ts | 30 ++-- .../hooks/use-prompt-actions/index.test.tsx | 82 +++++++++- .../session/hooks/use-prompt-actions/index.ts | 37 ++--- .../hooks/use-prompt-actions/rewind.test.ts | 74 ++++++++- .../hooks/use-prompt-actions/rewind.ts | 33 ++++ .../hooks/use-session-actions/utils.test.ts | 89 ++++++++++- .../hooks/use-session-actions/utils.ts | 141 ++++++++++++++---- apps/desktop/src/types/hermes.ts | 6 + tests/test_tui_gateway_server.py | 41 +++++ tui_gateway/server.py | 25 +++- 10 files changed, 480 insertions(+), 78 deletions(-) diff --git a/apps/desktop/src/app/chat/session-tile-actions.ts b/apps/desktop/src/app/chat/session-tile-actions.ts index 112a805f95084..b991bf596f403 100644 --- a/apps/desktop/src/app/chat/session-tile-actions.ts +++ b/apps/desktop/src/app/chat/session-tile-actions.ts @@ -34,6 +34,7 @@ import type { SessionInfo } from '@/types/hermes' import { uploadComposerAttachment } from '../session/hooks/use-prompt-actions' import { + appendMidTurnUserMessage, applyBranchVisibility, applyReloadOptimistic, applyRewindOptimistic, @@ -310,27 +311,20 @@ export function useSessionTileActions({ runtimeId, scope, storedSessionId }: Ses const mutate = (updater: (state: ClientSessionState) => ClientSessionState) => sessionTileDelegate()?.updateSession(sessionId, updater) - // Match the primary composer: insert the correction before the active - // reply before awaiting the redirect RPC, whose completion can race us. - mutate(state => { - const message = { + // Match the primary composer: record the correction in arrival order — + // sealed already-streamed output above, correction below, post-redirect + // deltas below that — before awaiting the redirect RPC, whose completion + // can race us. The old insert-before-the-active-reply splice put the + // bubble above output the user had already read (#73793), and its + // last-assistant fallback could land it mid-thread when the stream id + // was missing or stale (#83151). + mutate(state => + appendMidTurnUserMessage(state, { id: messageId, role: 'user' as const, parts: [textPart(text)] - } - - const streamIndex = state.streamId ? state.messages.findIndex(candidate => candidate.id === state.streamId) : -1 - - const lastAssistantIndex = state.messages.map(candidate => candidate.role).lastIndexOf('assistant') - const insertionIndex = streamIndex >= 0 ? streamIndex : lastAssistantIndex - - const messages = - insertionIndex >= 0 - ? [...state.messages.slice(0, insertionIndex), message, ...state.messages.slice(insertionIndex)] - : [...state.messages, message] - - return { ...state, messages } - }) + }) + ) const discardOptimisticMessage = () => mutate(state => ({ diff --git a/apps/desktop/src/app/session/hooks/use-prompt-actions/index.test.tsx b/apps/desktop/src/app/session/hooks/use-prompt-actions/index.test.tsx index f8c41b76253ae..baaffd4787513 100644 --- a/apps/desktop/src/app/session/hooks/use-prompt-actions/index.test.tsx +++ b/apps/desktop/src/app/session/hooks/use-prompt-actions/index.test.tsx @@ -105,6 +105,7 @@ function Harness({ resumeStoredSession, runtimeIdByStoredSessionIdRef: runtimeIdByStoredSessionIdRefProp, seedMessages, + seedStreamId, selectedStoredSessionIdRef: selectedStoredSessionIdRefProp, storedSessionId, activeSessionId, @@ -128,6 +129,7 @@ function Harness({ resumeStoredSession?: (storedSessionId: string) => Promise | void runtimeIdByStoredSessionIdRef?: MutableRefObject> seedMessages?: unknown[] + seedStreamId?: null | string selectedStoredSessionIdRef?: MutableRefObject storedSessionId?: null | string activeSessionId?: null | string @@ -159,7 +161,9 @@ function Harness({ messages: seedMessages ?? [], busy: false, awaitingResponse: false, - interrupted: true + interrupted: true, + streamId: seedStreamId ?? null, + interimBoundaryPending: false } as never) const actions = usePromptActions({ @@ -2189,6 +2193,82 @@ describe('usePromptActions redirectPrompt', () => { expect(requestGateway).not.toHaveBeenCalled() }) + it('records the correction AFTER the assistant output that predates it (#73793, #83151)', async () => { + const requestGateway = vi.fn(async () => ({ status: 'redirected' }) as never) + + let handle: HarnessHandle | null = null + const capturedStates: Record[] = [] + await actRender( + (handle = h)} + onSeedState={state => capturedStates.push(state)} + refreshSessions={async () => undefined} + requestGateway={requestGateway} + seedMessages={[ + { id: 'user-1', role: 'user', parts: [{ type: 'text', text: 'long task' }] }, + { + id: 'assistant-stream-1', + role: 'assistant', + parts: [{ type: 'text', text: 'two screens of already-read output' }], + pending: true + } + ]} + seedStreamId="assistant-stream-1" + /> + ) + + expect(await handle!.redirectPrompt('urgently')).toBe(true) + + const messages = capturedStates.at(-1)?.messages as { id: string; interim?: boolean; pending?: boolean }[] + + // Arrival order: the correction lands BELOW the streamed output the user + // had already read, never spliced above it. + expect(messages.map(message => message.id)).toEqual([ + 'user-1', + 'assistant-stream-1', + expect.stringMatching(/^user-/) + ]) + expect(messages[1]).toMatchObject({ pending: false, interim: true }) + // streamId cleared: the post-redirect deltas seed a fresh bubble below. + expect(capturedStates.at(-1)?.streamId).toBeNull() + }) + + it('appends at the tail — never mid-thread — when the stream id is stale (#83151)', async () => { + const requestGateway = vi.fn(async () => ({ status: 'redirected' }) as never) + + let handle: HarnessHandle | null = null + const capturedStates: Record[] = [] + await actRender( + (handle = h)} + onSeedState={state => capturedStates.push(state)} + refreshSessions={async () => undefined} + requestGateway={requestGateway} + seedMessages={[ + { id: 'user-1', role: 'user', parts: [{ type: 'text', text: 'old prompt' }] }, + { id: 'assistant-1', role: 'assistant', parts: [{ type: 'text', text: 'old committed reply' }] }, + { id: 'user-2', role: 'user', parts: [{ type: 'text', text: 'newer prompt' }] }, + { id: 'assistant-2', role: 'assistant', parts: [{ type: 'text', text: 'newer committed reply' }] } + ]} + seedStreamId="assistant-stream-gone" + /> + ) + + expect(await handle!.redirectPrompt('mid-turn note')).toBe(true) + + const messages = capturedStates.at(-1)?.messages as { id: string }[] + + // The retired fallback spliced this before 'assistant-2' — halfway up the + // chat. It must be the last row. + expect(messages.map(message => message.id)).toEqual([ + 'user-1', + 'assistant-1', + 'user-2', + 'assistant-2', + expect.stringMatching(/^user-/) + ]) + }) + it('accepts a queued redirect during the agent-build window and records the correction', async () => { // running=True but the agent is still building: the gateway queues the // correction instead of rejecting, so the composer must NOT re-queue it. diff --git a/apps/desktop/src/app/session/hooks/use-prompt-actions/index.ts b/apps/desktop/src/app/session/hooks/use-prompt-actions/index.ts index 3fabb514c7bda..20ebc58ccf0a9 100644 --- a/apps/desktop/src/app/session/hooks/use-prompt-actions/index.ts +++ b/apps/desktop/src/app/session/hooks/use-prompt-actions/index.ts @@ -49,6 +49,7 @@ import type { } from '../../../types' import { + appendMidTurnUserMessage, applyBranchVisibility, applyReloadOptimistic, applyRewindOptimistic, @@ -282,7 +283,7 @@ export function usePromptActions({ role: ChatMessage['role'], text: string, storedSessionId?: string | null, - options: { insertBeforeActiveReply?: boolean } = {} + options: { appendAfterActiveReply?: boolean } = {} ) => { // Strip ANSI: slash-command output from the backend worker carries SGR // color codes (e.g. "Unknown command" in red). The ESC byte is invisible @@ -305,23 +306,16 @@ export function usePromptActions({ parts: [textPart(body)] } - const streamIndex = - options.insertBeforeActiveReply && state.streamId - ? state.messages.findIndex(candidate => candidate.id === state.streamId) - : -1 - - const lastAssistantIndex = options.insertBeforeActiveReply - ? state.messages.map(candidate => candidate.role).lastIndexOf('assistant') - : -1 - - const insertionIndex = streamIndex >= 0 ? streamIndex : lastAssistantIndex - - const messages = - insertionIndex >= 0 - ? [...state.messages.slice(0, insertionIndex), message, ...state.messages.slice(insertionIndex)] - : [...state.messages, message] + // Mid-turn correction: arrival order. The bubble lands after the + // assistant output the user had already seen (sealing the live + // stream so post-redirect deltas continue BELOW the correction), + // never spliced above it (#73793) or mid-thread via the old + // last-assistant fallback (#83151). + if (options.appendAfterActiveReply) { + return appendMidTurnUserMessage(state, message) + } - return { ...state, messages } + return { ...state, messages: [...state.messages, message] } }, storedSessionId ?? selectedStoredSessionIdRef.current ) @@ -729,10 +723,11 @@ export function usePromptActions({ // transcript rather than a system note that changes role after reload. const send = async (id: string): Promise => { // Redirect aborts the model request, so the completion event can race - // its RPC response. Insert before the live reply *before* awaiting the - // gateway; appending after the response leaves the correction below a - // reply that the redirect has already replaced. - const messageId = appendSessionTextMessage(id, 'user', text, undefined, { insertBeforeActiveReply: true }) + // its RPC response. Record the correction *before* awaiting the + // gateway, in arrival order: sealed already-streamed output above, + // correction bubble below it, post-redirect deltas below that + // (#73793, #83151). + const messageId = appendSessionTextMessage(id, 'user', text, undefined, { appendAfterActiveReply: true }) const discardOptimisticMessage = () => updateSessionState(id, state => ({ diff --git a/apps/desktop/src/app/session/hooks/use-prompt-actions/rewind.test.ts b/apps/desktop/src/app/session/hooks/use-prompt-actions/rewind.test.ts index 3cbabbf9b617e..ebe6ca7528ef2 100644 --- a/apps/desktop/src/app/session/hooks/use-prompt-actions/rewind.test.ts +++ b/apps/desktop/src/app/session/hooks/use-prompt-actions/rewind.test.ts @@ -2,7 +2,79 @@ import { describe, expect, it } from 'vitest' import { type ChatMessage, textPart } from '@/lib/chat-messages' -import { rebindSurvivorRowIds, survivorRowIdsFrom, truncateSubmitParams } from './rewind' +import { appendMidTurnUserMessage, rebindSurvivorRowIds, survivorRowIdsFrom, truncateSubmitParams } from './rewind' + +const row = (id: string, role: ChatMessage['role'], text: string, extra: Partial = {}): ChatMessage => ({ + id, + role, + parts: [textPart(text)], + ...extra +}) + +type MidTurnState = { interimBoundaryPending: boolean; messages: ChatMessage[]; streamId: null | string } + +describe('appendMidTurnUserMessage', () => { + // #73793: a message typed while a turn streams must land AFTER the assistant + // output the user had already watched arrive — never spliced above it. + it('appends the mid-turn message after the live assistant output, sealed in place', () => { + const state: MidTurnState = { + interimBoundaryPending: false, + streamId: 'assistant-stream-1', + messages: [ + row('user-1', 'user', 'long task'), + row('assistant-stream-1', 'assistant', 'two screens of output', { pending: true }) + ] + } + + const next = appendMidTurnUserMessage(state, row('user-2', 'user', 'urgently')) + + expect(next.messages.map(message => message.id)).toEqual(['user-1', 'assistant-stream-1', 'user-2']) + // The sealed bubble stops streaming; the turn's next delta seeds a fresh + // bubble BELOW the correction instead of mutating the one above it. + expect(next.messages[1]).toMatchObject({ pending: false, interim: true }) + expect(next.streamId).toBeNull() + expect(next.interimBoundaryPending).toBe(true) + }) + + // #83151: the retired insert-before splice fell back to the LAST assistant + // row anywhere in the transcript when the stream id was stale, landing the + // new prompt mid-thread. A stale/missing stream id must append at the tail. + it('appends at the live tail when the stream id is stale or missing', () => { + const state: MidTurnState = { + interimBoundaryPending: false, + streamId: 'assistant-stream-stale', + messages: [ + row('user-1', 'user', 'old prompt'), + row('assistant-1', 'assistant', 'old committed reply'), + row('user-2', 'user', 'newer prompt'), + row('assistant-2', 'assistant', 'newer committed reply') + ] + } + + const next = appendMidTurnUserMessage(state, row('user-3', 'user', 'mid-turn note')) + + expect(next.messages.map(message => message.id)).toEqual(['user-1', 'assistant-1', 'user-2', 'assistant-2', 'user-3']) + expect(next.messages.at(-1)?.id).toBe('user-3') + expect(next.interimBoundaryPending).toBe(false) + }) + + it('drops an empty pending stream placeholder instead of sealing it', () => { + const state: MidTurnState = { + interimBoundaryPending: false, + streamId: 'assistant-stream-1', + messages: [ + row('user-1', 'user', 'task'), + { id: 'assistant-stream-1', role: 'assistant', parts: [], pending: true } + ] + } + + const next = appendMidTurnUserMessage(state, row('user-2', 'user', 'correction')) + + expect(next.messages.map(message => message.id)).toEqual(['user-1', 'user-2']) + expect(next.streamId).toBeNull() + expect(next.interimBoundaryPending).toBe(false) + }) +}) describe('truncateSubmitParams', () => { it('omits truncation fields when no ordinal is set', () => { diff --git a/apps/desktop/src/app/session/hooks/use-prompt-actions/rewind.ts b/apps/desktop/src/app/session/hooks/use-prompt-actions/rewind.ts index 5726a98aa3c6e..fcf83c4ba7102 100644 --- a/apps/desktop/src/app/session/hooks/use-prompt-actions/rewind.ts +++ b/apps/desktop/src/app/session/hooks/use-prompt-actions/rewind.ts @@ -209,6 +209,39 @@ export function finalizeInterruptedMessages(messages: ChatMessage[], streamId?: .map(message => (message.pending || message.id === streamId ? { ...message, pending: false } : message)) } +/** + * Arrival-ordered mid-turn user insert (#73793, #83151). + * + * A message typed while a turn streams must land AFTER every assistant row the + * user had already watched arrive — never spliced above it. Seal the live + * stream bubble in place (marked interim so the terminal completion settles + * onto it or follows it instead of duplicating), append the new user bubble at + * the live tail, and clear `streamId` so the turn's next delta seeds a fresh + * assistant bubble BELOW the correction rather than mutating the sealed one + * above it. Also retires the old insert-before-the-active-reply contract whose + * `lastAssistantIndex` fallback could splice the bubble mid-thread when the + * stream id was missing or stale (#83151). + */ +export function appendMidTurnUserMessage< + State extends { interimBoundaryPending: boolean; messages: ChatMessage[]; streamId: null | string } +>(state: State, message: ChatMessage): State { + const liveId = state.streamId + const sealed = finalizeInterruptedMessages(state.messages, liveId) + const sealedLiveKept = liveId !== null && sealed.some(row => row.id === liveId) + + const messages = [ + ...(sealedLiveKept ? sealed.map(row => (row.id === liveId ? { ...row, interim: true } : row)) : sealed), + message + ] + + return { + ...state, + messages, + streamId: null, + interimBoundaryPending: state.interimBoundaryPending || sealedLiveKept + } +} + // --------------------------------------------------------------------------- // Reload (regenerate) // --------------------------------------------------------------------------- diff --git a/apps/desktop/src/app/session/hooks/use-session-actions/utils.test.ts b/apps/desktop/src/app/session/hooks/use-session-actions/utils.test.ts index 4a18397200e63..27bff4795e07e 100644 --- a/apps/desktop/src/app/session/hooks/use-session-actions/utils.test.ts +++ b/apps/desktop/src/app/session/hooks/use-session-actions/utils.test.ts @@ -714,6 +714,25 @@ describe('preserveLocalPendingTurnMessages', () => { ]) }) + // Arrival-ordered mid-turn corrections (#73793) seal the live output BETWEEN + // the prompt and the correction. The sealed live-tail row must not end the + // optimistic run, or a refresh drops the prompt that started the turn. + it('keeps the whole live run when sealed live output sits between prompt and correction', () => { + const previous = [ + msg('user-1000', 'user', 'remove the session counts'), + msg('assistant-stream-1', 'assistant', 'two screens of output', { interim: true }), + msg('user-2000', 'user', 'hurry up'), + msg('assistant-stream-2', 'assistant', 'post-redirect output', { pending: true }) + ] + + expect(preserveLocalPendingTurnMessages([], previous).map(message => message.id)).toEqual([ + 'user-1000', + 'assistant-stream-1', + 'user-2000', + 'assistant-stream-2' + ]) + }) + it('still drops optimistic rows separated from the live run by an assistant reply', () => { const previous = [ msg('user-stale', 'user', 'compressed-away prompt'), @@ -1148,8 +1167,10 @@ describe('preserveLocalPendingTurnMessages', () => { describe('appendLiveSessionProjection', () => { // Corrections typed while a turn ran are their own user bubbles on the same - // turn. Resume must rebuild the prompt AND every correction, in order. - it('projects mid-turn redirect corrections after the prompt that started the turn', () => { + // turn, ordered by ARRIVAL. Without boundary offsets (older gateway) the + // whole dump precedes them — never the old prompt → corrections → reply + // order that spliced them above output the user had already read (#73793). + it('projects mid-turn redirect corrections after the assistant output that predates them', () => { const restored = appendLiveSessionProjection([], { session_id: 'runtime-1', inflight: { @@ -1162,10 +1183,72 @@ describe('appendLiveSessionProjection', () => { expect(restored.map(message => message.parts.map(part => ('text' in part ? part.text : '')).join(''))).toEqual([ 'remove the session counts', + 'Moving.', + 'hurry up', + 'and the worktree ones' + ]) + }) + + // With correction_offsets the flat dump is split at each accepted-correction + // boundary, so every correction lands after exactly the output it followed + // and before the output it redirected — arrival order end to end (#73793). + it('interleaves corrections into the assistant dump at their arrival offsets', () => { + const restored = appendLiveSessionProjection([], { + session_id: 'runtime-1', + inflight: { + user: 'remove the session counts', + corrections: ['hurry up', 'and the worktree ones'], + correction_offsets: [7, 13], + assistant: 'Moving.Still.Done soon.', + streaming: true + } + }) + + expect(restored.map(message => message.parts.map(part => ('text' in part ? part.text : '')).join(''))).toEqual([ + 'remove the session counts', + 'Moving.', 'hurry up', + 'Still.', 'and the worktree ones', - 'Moving.' + 'Done soon.' + ]) + expect(restored.map(message => message.role)).toEqual([ + 'user', + 'assistant', + 'user', + 'assistant', + 'user', + 'assistant' + ]) + // Only the live tail streams; sealed pre-correction segments are settled. + expect(restored.at(-1)).toMatchObject({ id: 'assistant-stream-runtime-1', pending: true }) + expect(restored[1]).toMatchObject({ pending: false, interim: true }) + expect(restored[3]).toMatchObject({ pending: false, interim: true }) + }) + + it('keeps the live stream row even when every offset points at the dump tail', () => { + const restored = appendLiveSessionProjection([], { + session_id: 'runtime-1', + inflight: { + user: 'prompt', + corrections: ['nudge'], + correction_offsets: [4], + assistant: 'text', + streaming: true + } + }) + + // The whole dump precedes the correction, and the still-streaming turn + // keeps its (empty for now) live row at the tail so future deltas land + // BELOW the correction, not above it. + expect(restored.map(message => message.parts.map(part => ('text' in part ? part.text : '')).join(''))).toEqual([ + 'prompt', + 'text', + 'nudge', + '' ]) + expect(restored.at(-1)).toMatchObject({ id: 'assistant-stream-runtime-1', pending: true }) + expect(restored.at(-1)?.role).toBe('assistant') }) it('does not re-project a correction the transcript already persisted', () => { diff --git a/apps/desktop/src/app/session/hooks/use-session-actions/utils.ts b/apps/desktop/src/app/session/hooks/use-session-actions/utils.ts index 6c031768acc76..1a510c778cd34 100644 --- a/apps/desktop/src/app/session/hooks/use-session-actions/utils.ts +++ b/apps/desktop/src/app/session/hooks/use-session-actions/utils.ts @@ -503,11 +503,21 @@ export function preserveLocalPendingTurnMessages( for (let index = previousMessages.indexOf(newestOptimisticUser); index >= 0; index -= 1) { const candidate = previousMessages[index] - if (candidate.role !== 'user' || !candidate.id.startsWith('user-')) { - break + if (candidate.role === 'user' && candidate.id.startsWith('user-')) { + liveOptimisticUsers.add(candidate) + + continue } - liveOptimisticUsers.add(candidate) + // Arrival-ordered mid-turn corrections sit BELOW the sealed live output + // (#73793): a live-tail assistant row between the prompt and its + // correction is still the same turn's run. Only a committed reply ends + // it — that is the post-compression staleness the rule exists to catch. + if (candidate.role === 'assistant' && isLiveTailRow(candidate)) { + continue + } + + break } } @@ -688,10 +698,22 @@ export function appendLiveSessionProjection( const inflightStreaming = Boolean(projection.inflight?.streaming) // Mid-turn redirect corrections. They are additional user bubbles belonging - // to this same turn, ordered after the prompt that started it. - const inflightCorrections = (projection.inflight?.corrections ?? []) - .map(correction => correction?.trim() ?? '') - .filter(Boolean) + // to this same turn, ordered by arrival: after the output that had already + // streamed when they were typed, before the output they redirected. + // `correction_offsets` (assistant-text length at each accepted correction) + // carries that boundary; older gateways omit it. + const rawCorrections = projection.inflight?.corrections ?? [] + const rawOffsets = projection.inflight?.correction_offsets + + const inflightCorrectionEntries = rawCorrections + .map((correction, index) => ({ text: correction?.trim() ?? '', offset: rawOffsets?.[index] })) + .filter(entry => entry.text) + + const inflightCorrections = inflightCorrectionEntries.map(entry => entry.text) + + const correctionOffsetsUsable = + inflightCorrectionEntries.length > 0 && + inflightCorrectionEntries.every(entry => typeof entry.offset === 'number' && entry.offset >= 0) // A retained failed turn (the gateway keeps error snapshots replayable when // the terminal frame may have been lost to a disconnect) — surface the @@ -718,13 +740,27 @@ export function appendLiveSessionProjection( // Only suppress the projection when the latest authoritative user row is the // same turn — older identical prompts must not hide a newly accepted repeat. // A mid-turn redirect gives that turn a RUN of user rows (prompt + - // corrections), so match the contiguous run ending at the latest user row - // rather than the single last one. + // corrections). Arrival order seals already-streamed output BETWEEN those + // rows (#73793), so collect the run by walking back over the live tail: + // user rows count, live-tail assistant rows are skipped, and a committed + // assistant reply ends the turn. const latestUserIndex = messages.map(message => message.role).lastIndexOf('user') const latestUserRun: ChatMessage[] = [] - for (let index = latestUserIndex; index >= 0 && messages[index].role === 'user'; index -= 1) { - latestUserRun.unshift(messages[index]) + for (let index = latestUserIndex; index >= 0; index -= 1) { + const candidate = messages[index] + + if (candidate.role === 'user') { + latestUserRun.unshift(candidate) + + continue + } + + if (candidate.role === 'assistant' && isLiveTailRow(candidate)) { + continue + } + + break } const persistedInLatestRun = (text: string): boolean => @@ -742,22 +778,6 @@ export function appendLiveSessionProjection( }) } - // Corrections typed while the turn ran. Each is its own bubble, placed after - // the original prompt and before the reply they redirected — the same order - // the live transcript showed. Skip any the transcript already holds so a - // resume doesn't double them. - for (const [index, correction] of inflightCorrections.entries()) { - if (persistedInLatestRun(correction)) { - continue - } - - projected.push({ - id: `user-inflight-correction-${index}-${sessionId}`, - role: 'user', - parts: [textPart(correction)] - }) - } - // Keep a pending assistant boundary even before the first delta when a // queued user turn follows it. This preserves the two distinct turns. // @@ -796,10 +816,65 @@ export function appendLiveSessionProjection( isLiveTailRow(liveAssistantOfCurrentTurn) ) - if (inflightAssistant || inflightStreaming || inflightError || (inflightUser && queuedUser)) { - if (turnAlreadyStructured && !inflightError) { - // Structure is authoritative; skip the text-only dump row. - } else { + const wantsAssistantRow = Boolean( + inflightAssistant || inflightStreaming || inflightError || (inflightUser && queuedUser) + ) + + const projectAssistantDump = wantsAssistantRow && !(turnAlreadyStructured && !inflightError) + + const pushCorrection = (correction: string, index: number): void => { + if (persistedInLatestRun(correction)) { + return + } + + projected.push({ + id: `user-inflight-correction-${index}-${sessionId}`, + role: 'user', + parts: [textPart(correction)] + }) + } + + // Corrections typed while the turn ran are ordered by ARRIVAL: each lands + // after the assistant output that had already streamed when it was typed and + // before the output it redirected (#73793 — the old prompt → corrections → + // reply order spliced them above screens of output the user had already + // read). With usable offsets the flat dump is split at each boundary; without + // them (older gateway, or a structured/error tail that must stay whole) the + // corrections follow the projected reply, matching the live transcript's + // append-at-tail contract. + if (projectAssistantDump && correctionOffsetsUsable && !inflightError && inflightAssistant) { + let cursor = 0 + + for (const [index, entry] of inflightCorrectionEntries.entries()) { + const boundary = Math.min(Math.max(entry.offset as number, cursor), inflightAssistant.length) + const segment = inflightAssistant.slice(cursor, boundary) + + if (segment.trim()) { + // Sealed pre-correction output. The `inflight-assistant-` prefix marks + // it a live-tail row so repeated resumes keep the user run intact. + projected.push({ + id: `inflight-assistant-segment-${index}-${sessionId}`, + role: 'assistant', + parts: [assistantTextPart(segment)], + pending: false, + interim: true + }) + } + + cursor = boundary + pushCorrection(entry.text, index) + } + + const tail = inflightAssistant.slice(cursor) + + projected.push({ + id: liveStreamId, + role: 'assistant', + parts: tail.trim() ? [assistantTextPart(tail)] : [], + pending: inflightStreaming + }) + } else { + if (projectAssistantDump) { projected.push({ id: liveStreamId, role: 'assistant', @@ -808,6 +883,10 @@ export function appendLiveSessionProjection( ...(inflightError ? { error: inflightError } : {}) }) } + + for (const [index, correction] of inflightCorrections.entries()) { + pushCorrection(correction, index) + } } if (queuedUser) { diff --git a/apps/desktop/src/types/hermes.ts b/apps/desktop/src/types/hermes.ts index 8fc2e23ef9016..5d04fdde54fea 100644 --- a/apps/desktop/src/types/hermes.ts +++ b/apps/desktop/src/types/hermes.ts @@ -605,6 +605,12 @@ export interface SessionResumeResponse { /** Mid-turn redirect corrections, oldest first. The turn's original prompt * stays in `user`; these are the follow-ups typed while it ran. */ corrections?: string[] + /** Parallel to `corrections`: the length of `assistant` already streamed + * when each correction was accepted. Lets a resume rebuild arrival order — + * the correction bubble lands after the output the user had already seen + * and before the output it redirected (#73793). Omitted by older + * gateways. */ + correction_offsets?: number[] /** Retained failed turn: the error the terminal frame carried (the frame * itself may have been lost to a disconnect). */ error?: string diff --git a/tests/test_tui_gateway_server.py b/tests/test_tui_gateway_server.py index cfb26d270ae41..e02559a06d687 100644 --- a/tests/test_tui_gateway_server.py +++ b/tests/test_tui_gateway_server.py @@ -9819,6 +9819,47 @@ def test_session_redirect_records_correction_without_erasing_prompt(): assert snapshot["corrections"] == ["hurry up", "and the worktree ones"] +def test_inflight_snapshot_carries_arrival_order_offsets(): + """Each correction records how much assistant text had already streamed. + + Resuming clients rebuild ARRIVAL order from these boundaries: the + correction bubble lands after the output the user had already seen and + before the output it redirected (#73793), instead of above the whole + reply. + """ + session = {} + server._start_inflight_turn(session, "remove the session counts") + server._append_inflight_delta(session, "Moving.") + server._record_inflight_correction(session, "hurry up") + server._append_inflight_delta(session, "Still.") + server._record_inflight_correction(session, "and the worktree ones") + server._append_inflight_delta(session, "Done soon.") + + snapshot = server._inflight_snapshot(session) + assert snapshot is not None + + assert snapshot["corrections"] == ["hurry up", "and the worktree ones"] + assert snapshot["correction_offsets"] == [len("Moving."), len("Moving.Still.")] + + +def test_inflight_snapshot_omits_offsets_when_not_fully_recorded(): + """A pre-upgrade in-memory turn may carry corrections without offsets. + + The parallel list is only sent when every correction has one, so clients + can trust the pairing and older snapshots degrade to the no-offset path. + """ + session = {} + server._start_inflight_turn(session, "prompt") + turn = session["inflight_turn"] + turn["corrections"] = ["legacy correction"] + + snapshot = server._inflight_snapshot(session) + assert snapshot is not None + + assert snapshot["corrections"] == ["legacy correction"] + assert "correction_offsets" not in snapshot + + def test_inflight_snapshot_omits_corrections_when_none_recorded(): session = {} server._start_inflight_turn(session, "just the prompt") diff --git a/tui_gateway/server.py b/tui_gateway/server.py index 00896954bcd05..dead62de1165f 100644 --- a/tui_gateway/server.py +++ b/tui_gateway/server.py @@ -7516,6 +7516,13 @@ def _record_inflight_correction(session: dict, text: Any) -> None: corrections = list(turn.get("corrections") or []) corrections.append(correction) turn["corrections"] = corrections + # Arrival-order boundary: how much assistant text had already streamed + # when this correction was accepted. Resuming clients use it to place the + # correction bubble AFTER the output the user had already seen and BEFORE + # the output it redirected (#73793) instead of above the whole reply. + offsets = list(turn.get("correction_offsets") or []) + offsets.append(len(str(turn.get("assistant") or ""))) + turn["correction_offsets"] = offsets turn["updated_at"] = time.time() session["inflight_turn"] = turn @@ -8081,11 +8088,23 @@ def _inflight_snapshot(session: dict) -> dict | None: "streaming": streaming, "user": user, } - corrections = [c for c in (turn.get("corrections") or []) if str(c).strip()] - if corrections: + raw_corrections = turn.get("corrections") or [] + raw_offsets = turn.get("correction_offsets") or [] + correction_pairs = [ + (str(c), raw_offsets[i] if i < len(raw_offsets) else None) + for i, c in enumerate(raw_corrections) + if str(c).strip() + ] + if correction_pairs: # Mid-turn redirects. Carried alongside the original prompt (not over # it) so resume can rebuild every user bubble the turn produced. - snapshot["corrections"] = [str(c) for c in corrections] + snapshot["corrections"] = [c for c, _ in correction_pairs] + # Assistant-text lengths at each correction boundary (parallel list). + # Only sent when every correction has one, so clients can trust the + # pairing; older in-memory turns without offsets omit the field and + # clients fall back to placing corrections after the assistant dump. + if all(isinstance(offset, int) and offset >= 0 for _, offset in correction_pairs): + snapshot["correction_offsets"] = [int(offset) for _, offset in correction_pairs] # type: ignore[arg-type] if error: # Retained failed turn (see _fail_inflight_turn): carry the error # semantics so a resuming client can rebuild the failed-turn bubble From 01512ca00013222f41f0529e3bfe186fe1efbd15 Mon Sep 17 00:00:00 2001 From: "hermes-seaeye[bot]" <307254004+hermes-seaeye[bot]@users.noreply.github.com> Date: Sat, 15 Aug 2026 03:31:07 +0000 Subject: [PATCH 061/376] fmt(js): `npm run fix` on merge (#86631) Co-authored-by: github-actions[bot] --- apps/desktop/electron/main.ts | 18 ++++++----- .../src/app/chat/session-tile-actions.ts | 16 ++++++++-- .../desktop/src/app/contrib/surfaces.test.tsx | 4 ++- .../hooks/use-prompt-actions/index.test.tsx | 31 +++++++------------ .../hooks/use-prompt-actions/rewind.test.ts | 8 ++++- .../session/hooks/use-prompt-actions/utils.ts | 6 +--- .../hooks/use-session-actions/index.ts | 6 +++- .../hooks/use-session-actions/utils.ts | 10 ++---- .../src/components/chat/status-section.tsx | 4 +-- .../components/pane-shell/pane-lifecycle.ts | 7 ++++- .../src/lib/renderer-loop-pause.test.ts | 5 +-- .../src/store/gateway-shared-remote.test.ts | 4 +-- apps/desktop/src/store/prompts.ts | 3 +- apps/desktop/src/store/subagents.test.ts | 22 +++++++++++-- 14 files changed, 85 insertions(+), 59 deletions(-) diff --git a/apps/desktop/electron/main.ts b/apps/desktop/electron/main.ts index b66faf6b82165..9501ff6a9028d 100644 --- a/apps/desktop/electron/main.ts +++ b/apps/desktop/electron/main.ts @@ -37,11 +37,7 @@ import { dashboardFallbackArgs, sourceDeclaresServe } from './backend-command' import { createBackendConnectionState } from './backend-connection-state' import { buildDesktopBackendEnv, hermesManagedNodePathEntries, normalizeHermesHomeRoot } from './backend-env' import { isReauthRequiredError, waitForHermesReady } from './backend-health' -import { - backendCommandMatches, - createBackendOwnership, - createBackendShutdownCoordinator -} from './backend-ownership' +import { backendCommandMatches, createBackendOwnership, createBackendShutdownCoordinator } from './backend-ownership' import { canImportHermesCli, execProbeSync, @@ -2927,7 +2923,10 @@ function execText(command, args) { async function processStartMarker(pid) { if (process.platform === 'linux') { const stat = await fs.promises.readFile(`/proc/${pid}/stat`, 'utf8') - const fields = stat.slice(stat.lastIndexOf(')') + 1).trim().split(/\s+/) + const fields = stat + .slice(stat.lastIndexOf(')') + 1) + .trim() + .split(/\s+/) if (!/^\d+$/.test(fields[19] || '')) { throw new Error(`Invalid /proc start marker for PID ${pid}`) @@ -2965,7 +2964,12 @@ async function backendCommandForPid(pid) { const command = IS_WINDOWS ? 'powershell.exe' : 'ps' const args = IS_WINDOWS - ? ['-NoProfile', '-NonInteractive', '-Command', `(Get-CimInstance Win32_Process -Filter 'ProcessId = ${pid}').CommandLine`] + ? [ + '-NoProfile', + '-NonInteractive', + '-Command', + `(Get-CimInstance Win32_Process -Filter 'ProcessId = ${pid}').CommandLine` + ] : ['-p', String(pid), '-o', 'command='] return (await execText(command, args)) || null diff --git a/apps/desktop/src/app/chat/session-tile-actions.ts b/apps/desktop/src/app/chat/session-tile-actions.ts index b991bf596f403..c4c6ae028624b 100644 --- a/apps/desktop/src/app/chat/session-tile-actions.ts +++ b/apps/desktop/src/app/chat/session-tile-actions.ts @@ -472,7 +472,13 @@ export function useSessionTileActions({ runtimeId, scope, storedSessionId }: Ses try { applySurvivorRowIds( - await submitRewind(plan.text, plan.truncateOrdinal, interruptFirst, plan.truncateMessageId, plan.truncateRowId) + await submitRewind( + plan.text, + plan.truncateOrdinal, + interruptFirst, + plan.truncateMessageId, + plan.truncateRowId + ) ) } catch (err) { update(state => ({ ...state, busy: false, awaitingResponse: false, messages })) @@ -506,7 +512,13 @@ export function useSessionTileActions({ runtimeId, scope, storedSessionId }: Ses try { applySurvivorRowIds( - await submitRewind(plan.text, plan.truncateOrdinal, interruptFirst, plan.truncateMessageId, plan.truncateRowId) + await submitRewind( + plan.text, + plan.truncateOrdinal, + interruptFirst, + plan.truncateMessageId, + plan.truncateRowId + ) ) } catch (err) { update(state => ({ ...state, busy: false, awaitingResponse: false, messages })) diff --git a/apps/desktop/src/app/contrib/surfaces.test.tsx b/apps/desktop/src/app/contrib/surfaces.test.tsx index 274da802683c3..472bf8cbeaea7 100644 --- a/apps/desktop/src/app/contrib/surfaces.test.tsx +++ b/apps/desktop/src/app/contrib/surfaces.test.tsx @@ -23,7 +23,9 @@ vi.mock('../chat', () => ({ vi.mock('../chat/sidebar', () => ({ ChatSidebar: () => null })) vi.mock('../right-sidebar/terminal/chrome', () => ({ TerminalPaneChrome: () => null })) vi.mock('../shell/hooks/use-status-snapshot', () => ({ useStatusSnapshot: () => ({}) })) -vi.mock('../shell/hooks/use-statusbar-items', () => ({ useStatusbarItems: () => ({ leftStatusbarItems: [], statusbarItems: [] }) })) +vi.mock('../shell/hooks/use-statusbar-items', () => ({ + useStatusbarItems: () => ({ leftStatusbarItems: [], statusbarItems: [] }) +})) vi.mock('../shell/statusbar-controls', () => ({ StatusbarControls: () => null })) vi.mock('../routes', () => ({ contributedRoutes: () => [], diff --git a/apps/desktop/src/app/session/hooks/use-prompt-actions/index.test.tsx b/apps/desktop/src/app/session/hooks/use-prompt-actions/index.test.tsx index baaffd4787513..765a26f05ca27 100644 --- a/apps/desktop/src/app/session/hooks/use-prompt-actions/index.test.tsx +++ b/apps/desktop/src/app/session/hooks/use-prompt-actions/index.test.tsx @@ -147,13 +147,13 @@ function Harness({ const defaultStoredSessionId = storedSessionId === undefined ? RUNTIME_SESSION_ID : storedSessionId const defaultRuntimeSessionId = activeSessionId === undefined ? RUNTIME_SESSION_ID : activeSessionId - const runtimeIdByStoredSessionIdRef: MutableRefObject> = - runtimeIdByStoredSessionIdRefProp ?? { - current: - defaultStoredSessionId && defaultRuntimeSessionId - ? new Map([[defaultStoredSessionId, defaultRuntimeSessionId]]) - : new Map() - } + + const runtimeIdByStoredSessionIdRef: MutableRefObject> = runtimeIdByStoredSessionIdRefProp ?? { + current: + defaultStoredSessionId && defaultRuntimeSessionId + ? new Map([[defaultStoredSessionId, defaultRuntimeSessionId]]) + : new Map() + } const localBusyRef = busyRef ?? { current: false } @@ -4372,9 +4372,7 @@ describe('usePromptActions submit entry-time runtime ownership proof (#64789/#65 session_id: STORED_SESSION_B, source: 'desktop' }) - expect( - calls.find(c => c.method === 'prompt.submit' && c.params?.session_id === RUNTIME_SESSION_A) - ).toBeUndefined() + expect(calls.find(c => c.method === 'prompt.submit' && c.params?.session_id === RUNTIME_SESSION_A)).toBeUndefined() expect( calls.find(c => c.method === 'prompt.submit' && c.params?.session_id === RUNTIME_SESSION_B_RESUMED) ).toBeDefined() @@ -4429,9 +4427,7 @@ describe('usePromptActions submit entry-time runtime ownership proof (#64789/#65 session_id: STORED_SESSION_B, source: 'desktop' }) - expect( - calls.find(c => c.method === 'prompt.submit' && c.params?.session_id === RUNTIME_SESSION_A) - ).toBeUndefined() + expect(calls.find(c => c.method === 'prompt.submit' && c.params?.session_id === RUNTIME_SESSION_A)).toBeUndefined() expect( calls.find(c => c.method === 'prompt.submit' && c.params?.session_id === RUNTIME_SESSION_B_RESUMED) ).toBeDefined() @@ -4444,6 +4440,7 @@ describe('usePromptActions submit entry-time runtime ownership proof (#64789/#65 const selectedStoredSessionIdRef: MutableRefObject = { current: STORED_SESSION_B } const activeSessionIdRef: MutableRefObject = { current: RUNTIME_SESSION_A } + const runtimeIdByStoredSessionIdRef: MutableRefObject> = { current: new Map([[STORED_SESSION_B, RUNTIME_SESSION_A]]) } @@ -4472,9 +4469,7 @@ describe('usePromptActions submit entry-time runtime ownership proof (#64789/#65 await handle!.submitText('first message in a genuinely fresh session') expect(calls.some(c => c.method === 'session.resume')).toBe(false) - expect( - calls.find(c => c.method === 'prompt.submit' && c.params?.session_id === RUNTIME_SESSION_A) - ).toBeDefined() + expect(calls.find(c => c.method === 'prompt.submit' && c.params?.session_id === RUNTIME_SESSION_A)).toBeDefined() }) it('resumes the selected session when its ownership cache entry is missing', async () => { @@ -4514,9 +4509,7 @@ describe('usePromptActions submit entry-time runtime ownership proof (#64789/#65 session_id: STORED_SESSION_B, source: 'desktop' }) - expect( - calls.find(c => c.method === 'prompt.submit' && c.params?.session_id === RUNTIME_SESSION_A) - ).toBeUndefined() + expect(calls.find(c => c.method === 'prompt.submit' && c.params?.session_id === RUNTIME_SESSION_A)).toBeUndefined() expect( calls.find(c => c.method === 'prompt.submit' && c.params?.session_id === RUNTIME_SESSION_B_RESUMED) ).toBeDefined() diff --git a/apps/desktop/src/app/session/hooks/use-prompt-actions/rewind.test.ts b/apps/desktop/src/app/session/hooks/use-prompt-actions/rewind.test.ts index ebe6ca7528ef2..2b8f99ff29359 100644 --- a/apps/desktop/src/app/session/hooks/use-prompt-actions/rewind.test.ts +++ b/apps/desktop/src/app/session/hooks/use-prompt-actions/rewind.test.ts @@ -53,7 +53,13 @@ describe('appendMidTurnUserMessage', () => { const next = appendMidTurnUserMessage(state, row('user-3', 'user', 'mid-turn note')) - expect(next.messages.map(message => message.id)).toEqual(['user-1', 'assistant-1', 'user-2', 'assistant-2', 'user-3']) + expect(next.messages.map(message => message.id)).toEqual([ + 'user-1', + 'assistant-1', + 'user-2', + 'assistant-2', + 'user-3' + ]) expect(next.messages.at(-1)?.id).toBe('user-3') expect(next.interimBoundaryPending).toBe(false) }) diff --git a/apps/desktop/src/app/session/hooks/use-prompt-actions/utils.ts b/apps/desktop/src/app/session/hooks/use-prompt-actions/utils.ts index 77227229e52ca..88880fdc8c840 100644 --- a/apps/desktop/src/app/session/hooks/use-prompt-actions/utils.ts +++ b/apps/desktop/src/app/session/hooks/use-prompt-actions/utils.ts @@ -302,11 +302,7 @@ export function clearSessionRecentlyInterrupted(sessionId?: string): void { } /** Whether a rewind/edit should interrupt before submit — busy OR recent Stop. */ -export function shouldInterruptBeforeRewind(opts: { - busy: boolean - sessionId: string - now?: number -}): boolean { +export function shouldInterruptBeforeRewind(opts: { busy: boolean; sessionId: string; now?: number }): boolean { return opts.busy || isSessionRecentlyInterrupted(opts.sessionId, opts.now) } diff --git a/apps/desktop/src/app/session/hooks/use-session-actions/index.ts b/apps/desktop/src/app/session/hooks/use-session-actions/index.ts index 5b4aeb3ce5b55..e11adefdba9ce 100644 --- a/apps/desktop/src/app/session/hooks/use-session-actions/index.ts +++ b/apps/desktop/src/app/session/hooks/use-session-actions/index.ts @@ -857,7 +857,11 @@ export function useSessionActions({ // wipes the just-restored activate/cache transcript (the same // wipe the `activated.messages.length || ...` guard above // already prevents for the activate payload itself). - if (persisted && persistedMatchesActivatedSession && (persisted.messages.length || !activatedMessages.length)) { + if ( + persisted && + persistedMatchesActivatedSession && + (persisted.messages.length || !activatedMessages.length) + ) { activatedMessages = reconcileAuthoritativeMessages(persisted.messages, activatedMessages) } } diff --git a/apps/desktop/src/app/session/hooks/use-session-actions/utils.ts b/apps/desktop/src/app/session/hooks/use-session-actions/utils.ts index 1a510c778cd34..db6132c556f5f 100644 --- a/apps/desktop/src/app/session/hooks/use-session-actions/utils.ts +++ b/apps/desktop/src/app/session/hooks/use-session-actions/utils.ts @@ -637,10 +637,7 @@ export function preserveLocalPendingTurnMessages( candidate.role === 'assistant' && !isLiveTailRow(candidate) && (textWithoutReferenceLines(chatMessageText(candidate)) === nextText || - isStrictAnswerTextExtension( - textWithoutReferenceLines(chatMessageText(candidate)), - nextText - )) + isStrictAnswerTextExtension(textWithoutReferenceLines(chatMessageText(candidate)), nextText)) ) if (committedMatch) { @@ -651,10 +648,7 @@ export function preserveLocalPendingTurnMessages( candidate => candidate.role === 'assistant' && !isLiveTailRow(candidate) && - isStrictAnswerTextExtension( - nextText, - textWithoutReferenceLines(chatMessageText(candidate)) - ) + isStrictAnswerTextExtension(nextText, textWithoutReferenceLines(chatMessageText(candidate))) ) if (committedPrefix) { diff --git a/apps/desktop/src/components/chat/status-section.tsx b/apps/desktop/src/components/chat/status-section.tsx index ab9c1180503ed..bdbead9b33ad0 100644 --- a/apps/desktop/src/components/chat/status-section.tsx +++ b/apps/desktop/src/components/chat/status-section.tsx @@ -42,9 +42,7 @@ export function StatusSection({ {icon && {icon}} {label} - {collapsed && collapsedIndicator && ( - {collapsedIndicator} - )} + {collapsed && collapsedIndicator && {collapsedIndicator}} {accessory &&
{accessory}
} diff --git a/apps/desktop/src/components/pane-shell/pane-lifecycle.ts b/apps/desktop/src/components/pane-shell/pane-lifecycle.ts index edac97d3dc5f5..6e537854e0022 100644 --- a/apps/desktop/src/components/pane-shell/pane-lifecycle.ts +++ b/apps/desktop/src/components/pane-shell/pane-lifecycle.ts @@ -31,7 +31,12 @@ interface ReconcilePaneLifecycleOptions { */ export function reconcilePaneLifecycle( previous: PaneLifecycleState, - { activeId, hotHiddenCap = DEFAULT_HOT_HIDDEN_PANE_CAP, keepAlive = () => false, paneIds }: ReconcilePaneLifecycleOptions + { + activeId, + hotHiddenCap = DEFAULT_HOT_HIDDEN_PANE_CAP, + keepAlive = () => false, + paneIds + }: ReconcilePaneLifecycleOptions ): PaneLifecycleState { const present = new Set(paneIds) const entries: Record = {} diff --git a/apps/desktop/src/lib/renderer-loop-pause.test.ts b/apps/desktop/src/lib/renderer-loop-pause.test.ts index c6fe0e91de3e0..46563c627a491 100644 --- a/apps/desktop/src/lib/renderer-loop-pause.test.ts +++ b/apps/desktop/src/lib/renderer-loop-pause.test.ts @@ -1,9 +1,6 @@ import { afterEach, describe, expect, it, vi } from 'vitest' -import { - installRendererAnimationPauseState, - RENDERER_ANIMATIONS_PAUSED_ATTRIBUTE -} from './renderer-loop-pause' +import { installRendererAnimationPauseState, RENDERER_ANIMATIONS_PAUSED_ATTRIBUTE } from './renderer-loop-pause' describe('installRendererAnimationPauseState', () => { afterEach(() => { diff --git a/apps/desktop/src/store/gateway-shared-remote.test.ts b/apps/desktop/src/store/gateway-shared-remote.test.ts index c74b312a10127..7d3d33c50160b 100644 --- a/apps/desktop/src/store/gateway-shared-remote.test.ts +++ b/apps/desktop/src/store/gateway-shared-remote.test.ts @@ -127,9 +127,7 @@ describe('ensureGatewayForProfile under a shared global remote', () => { setPrimaryGateway(makePrimary() as never, 'default') installDesktop({ getConnection }) - gatewayMocks.connect - .mockRejectedValueOnce(new Error('temporarily offline')) - .mockResolvedValueOnce(undefined) + gatewayMocks.connect.mockRejectedValueOnce(new Error('temporarily offline')).mockResolvedValueOnce(undefined) await ensureGatewayForProfile('worker') diff --git a/apps/desktop/src/store/prompts.ts b/apps/desktop/src/store/prompts.ts index 35f95455bcb1b..c70a0965cb534 100644 --- a/apps/desktop/src/store/prompts.ts +++ b/apps/desktop/src/store/prompts.ts @@ -135,7 +135,8 @@ export async function replayPendingApproval(gateway: ApprovalGateway | null, ses session_id: sessionId }) - const result = rawResult && typeof rawResult === 'object' ? (rawResult as { approvals?: PendingApprovalPayload[] }) : {} + const result = + rawResult && typeof rawResult === 'object' ? (rawResult as { approvals?: PendingApprovalPayload[] }) : {} const pending = Array.isArray(result?.approvals) ? result.approvals[0] : undefined if (!pending || typeof pending.request_id !== 'string') { diff --git a/apps/desktop/src/store/subagents.test.ts b/apps/desktop/src/store/subagents.test.ts index 4922376ecc0ac..74a604a026c6e 100644 --- a/apps/desktop/src/store/subagents.test.ts +++ b/apps/desktop/src/store/subagents.test.ts @@ -202,8 +202,18 @@ describe('subagent store', () => { upsertSubagent('s1', { goal: 'd', status: 'running', subagent_id: 'd', task_index: 3 }) // Emit terminal events with backend-native status strings - upsertSubagent('s1', { status: 'timeout', subagent_id: 'a', task_index: 0, summary: 'timed out' }, false, 'subagent.complete') - upsertSubagent('s1', { status: 'error', subagent_id: 'b', task_index: 1, summary: 'errored' }, false, 'subagent.complete') + upsertSubagent( + 's1', + { status: 'timeout', subagent_id: 'a', task_index: 0, summary: 'timed out' }, + false, + 'subagent.complete' + ) + upsertSubagent( + 's1', + { status: 'error', subagent_id: 'b', task_index: 1, summary: 'errored' }, + false, + 'subagent.complete' + ) upsertSubagent('s1', { status: 'cancelled', subagent_id: 'c', task_index: 2 }, false, 'subagent.complete') upsertSubagent('s1', { status: 'canceled', subagent_id: 'd', task_index: 3 }, false, 'subagent.complete') @@ -291,7 +301,13 @@ describe('subagent store', () => { status => { upsertSubagent( 's1', - { goal: 'inconsistent completion', status: 'running', subagent_id: 'ic1', task_index: 0, tool_name: 'search_files' }, + { + goal: 'inconsistent completion', + status: 'running', + subagent_id: 'ic1', + task_index: 0, + tool_name: 'search_files' + }, true, 'subagent.start' ) From fceab286023c05092174f649b464fe611f0c0a6d Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 16:18:55 -0700 Subject: [PATCH 062/376] fix(desktop): cold start no longer hides Cloud agents until Portal re-login Fixes #73495. Two cold-start defects made the configured Hermes Cloud agent vanish after a Desktop restart even though the persisted Portal session was still renewable: 1. hasLivePortalSession() trusted the FIRST cookies.get() on the lazy `persist:` partition. It now reuses the warmOauthCookieStore() warm-up + bounded reread that hasLiveOauthSession() gained in PR #67769, so a single hydration false-negative no longer clears the agent list and flips the panel to signed-out. 2. Discovery required the short-lived `privy-token` access cookie but treated its absence as a full interactive re-login, even when the 30-day `privy-session` / `privy-refresh-token` renewal cookies survived the process exit. New cookiesHavePrivyAccessToken() splits "signed in (renewable)" from "discovery can succeed right now"; discoverCloudAgents() and cloudAgentSilentSignIn() now mint a fresh access token via one bounded, hidden, deadline-capped portal load (renewPortalAccessSilently) before or after a 401, and only surface needsCloudLogin when renewal genuinely cannot complete. PRIVY_SESSION_COOKIE_VARIANTS also learns `privy-refresh-token` so a renewal-only jar still counts as signed in rather than demanding an interactive login while usable refresh material sits in the partition. Tests: connection-config.test.ts covers the access/session split, including the exact renewal-only cold-start jar from the issue repro. --- .../electron/connection-config.test.ts | 37 +++ apps/desktop/electron/connection-config.ts | 34 ++- apps/desktop/electron/main.ts | 266 ++++++++++++++++-- 3 files changed, 313 insertions(+), 24 deletions(-) diff --git a/apps/desktop/electron/connection-config.test.ts b/apps/desktop/electron/connection-config.test.ts index a306bec5ca1a9..b6451cad5ddda 100644 --- a/apps/desktop/electron/connection-config.test.ts +++ b/apps/desktop/electron/connection-config.test.ts @@ -21,6 +21,7 @@ import { buildGatewayWsUrlWithTicket, connectionScopeKey, cookiesHaveLiveSession, + cookiesHavePrivyAccessToken, cookiesHavePrivySession, cookiesHaveSession, gatewayTicketFailure, @@ -571,6 +572,42 @@ test('cookiesHavePrivySession is false for unrelated cookies and non-arrays', () assert.equal(cookiesHavePrivySession([]), false) }) +test('cookiesHavePrivySession treats refresh-token material as a (renewable) session', () => { + // #73495: after a restart the ~1h `privy-token` is often gone while the + // 30-day renewal cookies survive. That jar is still SIGNED IN (renewable), + // so the session check must accept it — the access check below is what + // distinguishes "can discovery succeed right now". + assert.equal(cookiesHavePrivySession([{ name: 'privy-refresh-token', value: 'x' }]), true) +}) + +// --- cookiesHavePrivyAccessToken (short-lived access state for /api/agents) --- + +test('cookiesHavePrivyAccessToken detects privy-token and its secured prefixes', () => { + assert.equal(cookiesHavePrivyAccessToken([{ name: 'privy-token', value: 'jwt' }]), true) + assert.equal(cookiesHavePrivyAccessToken([{ name: '__Host-privy-token', value: 'x' }]), true) + assert.equal(cookiesHavePrivyAccessToken([{ name: '__Secure-privy-token', value: 'x' }]), true) +}) + +test('cookiesHavePrivyAccessToken rejects renewal-only jars (the #73495 cold-start state)', () => { + // Session/refresh material present, access token absent: signed in but + // discovery would 401 → the silent-renewal path must trigger, not re-login. + const renewalOnly = [ + { name: 'privy-session', value: 'x' }, + { name: 'privy-refresh-token', value: 'x' } + ] + + assert.equal(cookiesHavePrivySession(renewalOnly), true) + assert.equal(cookiesHavePrivyAccessToken(renewalOnly), false) +}) + +test('cookiesHavePrivyAccessToken is false for empty values, gateway cookies, and non-arrays', () => { + assert.equal(cookiesHavePrivyAccessToken([{ name: 'privy-token', value: '' }]), false) + assert.equal(cookiesHavePrivyAccessToken([{ name: 'hermes_session_at', value: 'x' }]), false) + assert.equal(cookiesHavePrivyAccessToken(null), false) + assert.equal(cookiesHavePrivyAccessToken(undefined), false) + assert.equal(cookiesHavePrivyAccessToken([]), false) +}) + // --- tokenPreview --- test('tokenPreview returns null for empty', () => { diff --git a/apps/desktop/electron/connection-config.ts b/apps/desktop/electron/connection-config.ts index 4644008d48767..c3a676785d32f 100644 --- a/apps/desktop/electron/connection-config.ts +++ b/apps/desktop/electron/connection-config.ts @@ -44,7 +44,21 @@ const RT_COOKIE_VARIANTS = ['__Host-hermes_session_rt', '__Secure-hermes_session // sign-in / discovery liveness must look for the Privy cookie, NOT the gateway // cookies above. `privy-token` is the access token (the required signal); // variants cover the secured-prefix forms and the older `privy-session` name. -const PRIVY_SESSION_COOKIE_VARIANTS = ['__Host-privy-token', '__Secure-privy-token', 'privy-token', 'privy-session'] +const PRIVY_SESSION_COOKIE_VARIANTS = [ + '__Host-privy-token', + '__Secure-privy-token', + 'privy-token', + 'privy-session', + 'privy-refresh-token' +] + +// The short-lived Privy ACCESS token only — the credential `/api/agents` +// actually validates. `privy-session` / `privy-refresh-token` are long-lived +// renewal material: their presence means the session is RENEWABLE (signed in, +// no interactive login needed), but discovery still 401s until a fresh +// `privy-token` is minted. Distinguishing the two is what lets a cold start +// silently renew instead of demanding a re-login (#73495). +const PRIVY_ACCESS_COOKIE_VARIANTS = ['__Host-privy-token', '__Secure-privy-token', 'privy-token'] // Keep this aligned with hermes_cli.profiles.validate_profile_name(). `default` // is the built-in root alias; these names cannot be created as profiles. const RESERVED_REMOTE_PROFILES = new Set(['hermes', 'test', 'tmp', 'root', 'sudo']) @@ -557,6 +571,22 @@ function cookiesHavePrivySession(cookies) { return cookies.some(c => c && c.value && PRIVY_SESSION_COOKIE_VARIANTS.includes(c.name)) } +/** + * True only when the short-lived Privy ACCESS token (`privy-token`) is present + * — the exact cookie `/api/agents` validates. A jar can satisfy + * `cookiesHavePrivySession` (renewable session: `privy-session` / + * `privy-refresh-token`) while failing this check; that gap is the cold-start + * "Signed in" + "No agents found" contradiction, and the signal that a silent + * renewal (not an interactive re-login) is the right recovery (#73495). + */ +function cookiesHavePrivyAccessToken(cookies) { + if (!Array.isArray(cookies)) { + return false + } + + return cookies.some(c => c && c.value && PRIVY_ACCESS_COOKIE_VARIANTS.includes(c.name)) +} + export { AT_COOKIE_VARIANTS, authModeFromStatus, @@ -564,6 +594,7 @@ export { buildGatewayWsUrlWithTicket, connectionScopeKey, cookiesHaveLiveSession, + cookiesHavePrivyAccessToken, cookiesHavePrivySession, cookiesHaveSession, gatewayTicketFailure, @@ -576,6 +607,7 @@ export { normalizeSshConfig, normAuthMode, pathWithGlobalRemoteProfile, + PRIVY_ACCESS_COOKIE_VARIANTS, PRIVY_SESSION_COOKIE_VARIANTS, profileHasRemoteConnection, profileRemoteOverride, diff --git a/apps/desktop/electron/main.ts b/apps/desktop/electron/main.ts index 9501ff6a9028d..d7bd748c477f2 100644 --- a/apps/desktop/electron/main.ts +++ b/apps/desktop/electron/main.ts @@ -62,6 +62,7 @@ import { buildGatewayWsUrlWithTicket, connectionScopeKey, cookiesHaveLiveSession, + cookiesHavePrivyAccessToken, cookiesHavePrivySession, cookiesHaveSession, gatewayTicketFailure, @@ -6745,6 +6746,14 @@ function resolvePortalBaseUrl() { // checks for the `privy-token` cookie on the portal host (NOT // hasLiveOauthSession, which looks for hermes_session_at/rt that the portal // never sets). See connection-config.ts cookiesHavePrivySession. +// +// Mirrors hasLiveOauthSession's cold-start guard (#73495): a `persist:` +// partition's cookie store hydrates lazily, so the FIRST read on a fresh boot +// can come back empty even for a signed-in user. The renderer checks Cloud +// status exactly once on entering cloud mode, so a single false-negative here +// used to clear the discovered agent list and demand a re-login that a plain +// retry would have avoided. Warm the store and re-read with a short backoff +// before trusting a negative. async function hasLivePortalSession() { const sess = getOauthSession() @@ -6755,21 +6764,196 @@ async function hasLivePortalSession() { const portalBaseUrl = resolvePortalBaseUrl() const parsed = new URL(portalBaseUrl) + const readPortal = async () => { + try { + const cookies = await sess.cookies.get({ url: portalBaseUrl }) + + return cookiesHavePrivySession(cookies) + } catch { + try { + const cookies = await sess.cookies.get({ domain: parsed.hostname }) + + return cookiesHavePrivySession(cookies) + } catch { + return false + } + } + } + + if (await readPortal()) { + return true + } + + await warmOauthCookieStore() + + for (const delayMs of [30, 60, 90]) { + if (await readPortal()) { + return true + } + + await new Promise(resolve => setTimeout(resolve, delayMs)) + } + + return readPortal() +} + +// Whether the jar holds the short-lived Privy ACCESS token — the exact cookie +// `/api/agents` validates. hasLivePortalSession() answers "signed in at all?" +// (renewal material counts); this answers "can discovery succeed right now?". +async function hasPortalAccessToken() { + const sess = getOauthSession() + + if (!sess) { + return false + } + + const portalBaseUrl = resolvePortalBaseUrl() + const parsed = new URL(portalBaseUrl) + try { const cookies = await sess.cookies.get({ url: portalBaseUrl }) - return cookiesHavePrivySession(cookies) + return cookiesHavePrivyAccessToken(cookies) } catch { try { const cookies = await sess.cookies.get({ domain: parsed.hostname }) - return cookiesHavePrivySession(cookies) + return cookiesHavePrivyAccessToken(cookies) } catch { return false } } } +// Bounded silent renewal of the short-lived Privy access token (#73495). +// +// After a Desktop restart the long-lived `privy-session` / `privy-refresh-token` +// cookies routinely survive while the ~1h `privy-token` access cookie has +// expired. Discovery then 401s and the only offered recovery used to be a full +// interactive re-login — even though the persisted refresh material can mint a +// fresh access token with no user action: loading any portal page runs the +// Privy client, which rotates a new `privy-token` from the refresh session. +// +// This drives exactly that, headlessly: a hidden window on the portal root in +// the OAuth partition, polled until the access cookie lands, torn down on a +// bounded timeout. Never shown — if renewal can't complete silently the caller +// falls back to the interactive needsCloudLogin path. The in-flight promise is +// shared so concurrent discovery + cascade calls ride one renewal. +let portalAccessRenewal: Promise | null = null + +function renewPortalAccessSilently() { + if (portalAccessRenewal) { + return portalAccessRenewal + } + + portalAccessRenewal = (async () => { + if (!app.isReady()) { + return false + } + + const sess = getOauthSession() + + if (!sess) { + return false + } + + // No renewal material at all → nothing to renew; interactive login is + // genuinely required. + if (!(await hasLivePortalSession())) { + return false + } + + if (await hasPortalAccessToken()) { + return true + } + + const portalBaseUrl = resolvePortalBaseUrl() + + return await new Promise(resolve => { + let settled = false + let win = null + let pollTimer = null + let deadlineTimer = null + + const finish = (ok: boolean) => { + if (settled) { + return + } + + settled = true + + if (pollTimer) { + clearInterval(pollTimer) + } + + if (deadlineTimer) { + clearTimeout(deadlineTimer) + } + + try { + if (win && !win.isDestroyed()) { + win.destroy() + } + } catch { + // window already torn down + } + + rememberLog(`[cloud] silent portal access renewal ${ok ? 'succeeded' : 'did not complete'}`) + resolve(ok) + } + + const checkCookie = async () => { + if (settled) { + return + } + + if (await hasPortalAccessToken()) { + finish(true) + } + } + + try { + win = new BrowserWindow({ + width: 520, + height: 720, + show: false, + title: 'Renewing Hermes Cloud session…', + autoHideMenuBar: true, + webPreferences: { + contextIsolation: true, + nodeIntegration: false, + sandbox: true, + session: sess, + webSecurity: true + } + }) + } catch { + finish(false) + + return + } + + win.webContents.on('did-navigate', () => void checkCookie()) + win.webContents.on('did-redirect-navigation', () => void checkCookie()) + win.webContents.on('did-frame-navigate', () => void checkCookie()) + installWindowRendererLifecycle(win, { kind: 'portal-renew', callbacks: { log: rememberLog } }) + pollTimer = setInterval(() => void checkCookie(), 500) + // Hard deadline: this window is never revealed, so an unrenewable session + // (revoked refresh token, portal down) must resolve false rather than + // hang the discovery call behind an invisible window. + deadlineTimer = setTimeout(() => finish(false), 12_000) + + win.on('closed', () => finish(false)) + + win.loadURL(portalBaseUrl).catch(() => finish(false)) + }) + })().finally(() => { + portalAccessRenewal = null + }) as Promise + + return portalAccessRenewal +} + // Drive a one-time interactive portal sign-in in the OAuth partition. Unlike // openOauthLoginWindow (which targets a gateway's /login), this lands on the // portal itself so the resulting session cookie is portal-scoped — the cookie @@ -6898,37 +7082,65 @@ async function discoverCloudAgents(org?: string) { throw err } + // Renewable session present but the short-lived access token `/api/agents` + // validates is gone (typical after a restart — `privy-token` is ~1h, + // `privy-session`/`privy-refresh-token` last ~30 days). Renew silently up + // front instead of letting the request 401 into a re-login demand (#73495). + if (!(await hasPortalAccessToken())) { + await renewPortalAccessSilently() + } + const orgQuery = org ? `?org=${encodeURIComponent(org)}` : '' let body - try { - body = (await fetchJsonViaOauthSession(`${portalBaseUrl}/api/agents${orgQuery}`, { + const fetchAgents = () => + fetchJsonViaOauthSession(`${portalBaseUrl}/api/agents${orgQuery}`, { method: 'GET', timeoutMs: 15_000 - })) as any - } catch (error) { - // A 401 means the portal session lapsed between the liveness check and the - // call — surface it as a re-login, not a generic failure. - if (error && error.statusCode === 401) { - const err = new Error('Your Hermes Cloud session has expired. Open Settings → Gateway and sign in again.') as any - err.needsCloudLogin = true - err.cause = error - throw err + }) + + try { + body = (await fetchAgents()) as any + } catch (initialError) { + let error = initialError as any + + // A 401 with renewal material still in the jar: attempt ONE bounded silent + // renewal and retry, so a lapsed access token doesn't surface as a full + // interactive re-login while a 30-day refresh session sits unused. Only a + // rejected/failed renewal (or a second 401 on genuinely fresh access) + // falls through to needsCloudLogin. + if (error && error.statusCode === 401 && (await renewPortalAccessSilently())) { + try { + body = (await fetchAgents()) as any + } catch (retryError) { + error = retryError + } } - // A 409 means we're a multi-org user who hasn't picked an org. The body - // carries the user's org list; surface it so the renderer shows a picker - // and re-calls discovery with the chosen org. (fetchJsonViaOauthSession - // throws on >=400 with err.statusCode + err.message "409: ".) - if (error && error.statusCode === 409) { - const orgs = parseOrgSelectionError(error) + if (body === undefined) { + // A 401 means the portal session lapsed (and silent renewal could not + // recover it) — surface it as a re-login, not a generic failure. + if (error && error.statusCode === 401) { + const err = new Error('Your Hermes Cloud session has expired. Open Settings → Gateway and sign in again.') as any + err.needsCloudLogin = true + err.cause = error + throw err + } + + // A 409 means we're a multi-org user who hasn't picked an org. The body + // carries the user's org list; surface it so the renderer shows a picker + // and re-calls discovery with the chosen org. (fetchJsonViaOauthSession + // throws on >=400 with err.statusCode + err.message "409: ".) + if (error && error.statusCode === 409) { + const orgs = parseOrgSelectionError(error) - if (orgs) { - return { needsOrgSelection: true, orgs } + if (orgs) { + return { needsOrgSelection: true, orgs } + } } - } - throw error + throw error + } } return { agents: trimCloudAgents(body), org: trimCloudOrg(body?.org) } @@ -7020,6 +7232,14 @@ async function cloudAgentSilentSignIn(dashboardUrl) { throw err } + // The cascade rides the portal's auto-approve, which needs the short-lived + // access state just like discovery. If only renewal material survived the + // restart, mint a fresh access token first so the hidden cascade window + // auto-SSOs instead of stalling on an interactive chooser (#73495). + if (!(await hasPortalAccessToken())) { + await renewPortalAccessSilently() + } + await openOauthLoginWindow(baseUrl, { silent: true }) return { baseUrl, connected: await hasOauthSessionCookie(baseUrl) } From bb45dc06c35174bcaba2fab79b2f9fcd627aa027 Mon Sep 17 00:00:00 2001 From: konsisumer Date: Wed, 12 Aug 2026 22:18:38 +0200 Subject: [PATCH 063/376] fix(desktop): stabilize virtual session scrolling --- .../sidebar/virtual-session-list.test.tsx | 85 +++++++++++++++++++ .../app/chat/sidebar/virtual-session-list.tsx | 67 ++++++--------- 2 files changed, 109 insertions(+), 43 deletions(-) create mode 100644 apps/desktop/src/app/chat/sidebar/virtual-session-list.test.tsx diff --git a/apps/desktop/src/app/chat/sidebar/virtual-session-list.test.tsx b/apps/desktop/src/app/chat/sidebar/virtual-session-list.test.tsx new file mode 100644 index 0000000000000..3dec464df15d1 --- /dev/null +++ b/apps/desktop/src/app/chat/sidebar/virtual-session-list.test.tsx @@ -0,0 +1,85 @@ +import { cleanup, render } from '@testing-library/react' +import type * as React from 'react' +import { afterEach, describe, expect, it, vi } from 'vitest' + +import type { SidebarListRow } from '@/lib/session-date-groups' + +import { VirtualSessionList } from './virtual-session-list' + +const virtualizer = { + getTotalSize: () => 68, + getVirtualItems: () => [ + { end: 26, index: 0, start: 0 }, + { end: 68, index: 1, start: 26 } + ], + measureElement: vi.fn() +} + +vi.mock('@dnd-kit/sortable', () => ({ useSortable: vi.fn() })) +vi.mock('@dnd-kit/utilities', () => ({ CSS: { Transform: { toString: vi.fn() } } })) +vi.mock('@tanstack/react-virtual', () => ({ useVirtualizer: () => virtualizer })) + +vi.mock('@/i18n', () => ({ + useI18n: () => ({ + t: { + sidebar: { + dateDivider: { + earlierThisMonth: 'Earlier this month', + lastMonth: 'Last month', + lastWeek: 'Last week', + older: 'Older', + today: 'Today', + yesterday: 'Yesterday' + } + } + } + }) +})) + +vi.mock('./chrome', () => ({ + SidebarDateDivider: ({ label, ...props }: { label: string } & React.ComponentProps<'div'>) => ( +
+ ) +})) + +vi.mock('./session-row', () => ({ SidebarSessionRow: () => null })) + +afterEach(cleanup) + +const rows: SidebarListRow[] = [ + { key: 'today', kind: 'divider', label: 'Today' }, + { key: 'older', kind: 'divider', label: 'Older' } +] + +const noop = () => {} + +describe('VirtualSessionList', () => { + it('positions measured rows independently within a total-size spacer', () => { + const { getByTestId } = render( + + ) + + const firstItem = getByTestId('divider-Today').parentElement + const secondItem = getByTestId('divider-Older').parentElement + const spacer = firstItem?.parentElement + + expect(firstItem?.dataset.index).toBe('0') + expect(firstItem?.style.position).toBe('absolute') + expect(firstItem?.style.transform).toBe('translateY(0px)') + expect(secondItem?.dataset.index).toBe('1') + expect(secondItem?.style.transform).toBe('translateY(26px)') + expect(spacer?.className).toBe('relative') + expect(spacer?.style.height).toBe('68px') + expect(spacer?.style.paddingTop).toBe('') + expect(spacer?.style.paddingBottom).toBe('') + }) +}) diff --git a/apps/desktop/src/app/chat/sidebar/virtual-session-list.tsx b/apps/desktop/src/app/chat/sidebar/virtual-session-list.tsx index f99edb5f020da..2a10fd9a3f16a 100644 --- a/apps/desktop/src/app/chat/sidebar/virtual-session-list.tsx +++ b/apps/desktop/src/app/chat/sidebar/virtual-session-list.tsx @@ -2,7 +2,7 @@ import { useSortable } from '@dnd-kit/sortable' import { CSS } from '@dnd-kit/utilities' import { useVirtualizer } from '@tanstack/react-virtual' import type * as React from 'react' -import { type FC, useCallback, useRef } from 'react' +import { type FC, useRef } from 'react' import type { SessionInfo } from '@/hermes' import { useI18n } from '@/i18n' @@ -88,8 +88,6 @@ export const VirtualSessionList: FC = ({ const virtualItems = virtualizer.getVirtualItems() const totalSize = virtualizer.getTotalSize() - const paddingTop = virtualItems[0]?.start ?? 0 - const paddingBottom = Math.max(0, totalSize - (virtualItems[virtualItems.length - 1]?.end ?? 0)) const rows = virtualItems.map(virtualItem => { const row = listRows[virtualItem.index] @@ -98,16 +96,23 @@ export const VirtualSessionList: FC = ({ return null } + const itemStyle: React.CSSProperties = { + left: 0, + position: 'absolute', + top: 0, + transform: `translateY(${virtualItem.start}px)`, + width: '100%' + } + // Dividers are non-sortable, self-measured rows interleaved with sessions. if (row.kind === 'divider') { return ( - +
+ +
) } @@ -129,21 +134,13 @@ export const VirtualSessionList: FC = ({ } return reorderable ? ( - +
+ +
) : ( - +
+ +
) }) @@ -164,41 +161,25 @@ export const VirtualSessionList: FC = ({ )} ref={scrollerRef} > -
- {rows} -
+
{rows}
) } interface VirtualSortableRowProps { - index: number - measureRef: (node: Element | null) => void rowProps: SessionRowCommonProps session: SessionInfo } -function VirtualSortableRow({ index, measureRef, rowProps, session }: VirtualSortableRowProps) { +function VirtualSortableRow({ rowProps, session }: VirtualSortableRowProps) { const { attributes, isDragging, listeners, setNodeRef, transform, transition } = useSortable({ id: session.id }) - // Merge dnd-kit's setNodeRef with the virtualizer's measureElement so - // the row participates in both DnD hit-testing and TanStack height - // measurement. - const refMerged = useCallback( - (node: HTMLDivElement | null) => { - setNodeRef(node) - measureRef(node) - }, - [measureRef, setNodeRef] - ) - return ( Date: Mon, 10 Aug 2026 04:42:34 -0400 Subject: [PATCH 064/376] fix(desktop): reanchor transcript on window focus --- apps/desktop/e2e/mock-server.ts | 63 ++++++++ apps/desktop/e2e/task-panel-clearance.spec.ts | 145 ++++++++++++++++++ .../app/chat/composer/status-stack/index.tsx | 1 + .../assistant-ui/thread/list.test.ts | 75 ++++++++- .../components/assistant-ui/thread/list.tsx | 71 +++++++-- 5 files changed, 338 insertions(+), 17 deletions(-) create mode 100644 apps/desktop/e2e/task-panel-clearance.spec.ts diff --git a/apps/desktop/e2e/mock-server.ts b/apps/desktop/e2e/mock-server.ts index ce4665d1775c8..8de1af8aa4f98 100644 --- a/apps/desktop/e2e/mock-server.ts +++ b/apps/desktop/e2e/mock-server.ts @@ -120,6 +120,9 @@ let _correctionSwitchIndex = 0 /** Per-server counter for the verify-on-stop script. */ let _verificationStopIndex = 0 +/** Per-server counter for the task-panel warm-resume script. */ +let _taskPanelResumeIndex = 0 + /** User messages received by the mock, for E2E assertions on real submits. */ const _receivedUserTexts: string[] = [] @@ -131,6 +134,7 @@ function resetScriptIndex(): void { _queueStopIndex = 0 _correctionSwitchIndex = 0 _verificationStopIndex = 0 + _taskPanelResumeIndex = 0 _receivedUserTexts.length = 0 } @@ -295,6 +299,41 @@ export const VERIFICATION_STOP_TEXT = 'I cannot provide fresh verification evide export const BLOCKING_CLARIFY_TRIGGER = 'E2E_BLOCKING_CLARIFY_TRIGGER' export const BLOCKING_CLARIFY_QUESTION = 'Keep this test turn running?' +/** + * A long live response with a five-row todo card, held open by a foreground tool. + * The transcript is deliberately taller than the viewport so warm-session + * tests can detect when re-opening the session leaves it above the true bottom. + */ +export const TASK_PANEL_RESUME_TRIGGER = 'E2E_TASK_PANEL_RESUME_TRIGGER' +export const TASK_PANEL_RESUME_TEXT = Array.from( + { length: 24 }, + (_, index) => `Task-panel clearance line ${index + 1}: inspect the restored working session geometry.`, +).join('\n\n') + +const TASK_PANEL_RESUME_SCRIPT: ScriptedTurn[] = [ + { + text: TASK_PANEL_RESUME_TEXT, + toolCalls: [ + { + name: 'todo', + args: { + todos: [ + { id: 'design', content: 'Design the restored layout', status: 'completed' }, + { id: 'implement', content: 'Implement the measured clearance', status: 'in_progress' }, + { id: 'verify', content: 'Verify the latest message stays visible', status: 'pending' }, + { id: 'review', content: 'Review the visual regression', status: 'pending' }, + { id: 'ship', content: 'Ship the focused fix', status: 'pending' }, + ], + }, + }, + { + name: 'terminal', + args: { command: 'sleep 60' }, + }, + ], + }, +] + const BLOCKING_CLARIFY_TURN: ScriptedTurn = { text: '', toolCalls: [{ name: 'clarify', args: { question: BLOCKING_CLARIFY_QUESTION, choices: ['Yes', 'No'] } }], @@ -414,6 +453,7 @@ export function startMockServer(options: MockServerOptions = {}): Promise typeof message?.content === 'string' && message.content.includes(VERIFICATION_STOP_TRIGGER), ) @@ -421,6 +461,29 @@ export function startMockServer(options: MockServerOptions = {}): Promise typeof message?.content === 'string' && message.content.includes(CORRECTION_SWITCH_TRIGGER), ) + if (isTaskPanelResumeTrigger) { + const turn = + TASK_PANEL_RESUME_SCRIPT[_taskPanelResumeIndex] ?? + TASK_PANEL_RESUME_SCRIPT[TASK_PANEL_RESUME_SCRIPT.length - 1] + _taskPanelResumeIndex++ + const respond = () => { + if (stream) { + streamScriptedTurn(res, model, turn) + } else { + nonStreamingScriptedTurn(res, model, turn) + } + } + + if (holdThisCompletion) { + heldCompletionCount++ + resolveHeldStreamStarted?.() + void heldStreamReleased.then(respond) + } else { + respond() + } + return + } + if (includesBlockingClarifyTrigger(parsed.messages)) { if (stream) { streamScriptedTurn(res, model, BLOCKING_CLARIFY_TURN) diff --git a/apps/desktop/e2e/task-panel-clearance.spec.ts b/apps/desktop/e2e/task-panel-clearance.spec.ts new file mode 100644 index 0000000000000..19cf8d9541ed3 --- /dev/null +++ b/apps/desktop/e2e/task-panel-clearance.spec.ts @@ -0,0 +1,145 @@ +/** + * Regression coverage for returning to a working session as its task panel + * expands. The transcript must reconcile to the composer's full measured + * height without needing a manual scroll to repair the position. + */ + +import { expect, test, type Page } from './test' + +import { type MockBackendFixture, setupMockBackend, waitForAppReady } from './fixtures' +import { TASK_PANEL_RESUME_TRIGGER } from './mock-server' + +const SURFACE = '[data-composer-target]:visible' +const PROMPT = `${TASK_PANEL_RESUME_TRIGGER}: keep the task panel expanded while this session is reopened.` + +function activeSurface(page: Page) { + return page.locator(SURFACE).last() +} + +async function send(page: Page, text: string): Promise { + const composer = activeSurface(page).locator('[contenteditable="true"]').first() + + await composer.waitFor({ state: 'visible', timeout: 15_000 }) + await composer.click() + await composer.type(text, { delay: 5 }) + await page.keyboard.press('Enter') +} + +async function openFreshDraft(page: Page): Promise { + await page.locator('[data-slot="sidebar"] button[aria-label="New session"]').first().click() + await expect(activeSurface(page).locator('[data-slot="aui_thread-viewport"]')).not.toContainText(PROMPT) + await page.waitForTimeout(1_000) +} + +async function reopenWorkingSession(page: Page): Promise { + const sidebar = page.locator('[data-slot="sidebar"]') + const row = sidebar.getByRole('button', { name: /^(?:Session running|Needs your input|Working)\b/ }).first() + + await row.waitFor({ state: 'visible', timeout: 30_000 }) + await row.click() + await expect(activeSurface(page).locator('[data-slot="aui_thread-viewport"]')).toContainText( + 'Task-panel clearance line 24', + { timeout: 30_000 }, + ) +} + +interface ClearanceMetrics { + composerHeight: number + distanceFromBottom: number + latestMessageBottom: number + statusPanelTop: number + viewportHeight: number +} + +async function clearanceMetrics(page: Page): Promise { + return activeSurface(page).evaluate(surface => { + const chatSurface = surface.closest('[data-chat-surface]')! + const viewport = surface.querySelector('[data-slot="aui_thread-viewport"]')! + const latest = Array.from(surface.querySelectorAll('[data-role="assistant"]')).at(-1)! + const status = surface.querySelector('[data-slot="composer-status-stack"]')! + const styles = getComputedStyle(chatSurface) + + return { + composerHeight: Number.parseFloat(styles.getPropertyValue('--composer-measured-height')), + distanceFromBottom: viewport.scrollHeight - viewport.clientHeight - viewport.scrollTop, + latestMessageBottom: latest.getBoundingClientRect().bottom, + statusPanelTop: status.getBoundingClientRect().top, + viewportHeight: viewport.clientHeight, + } + }) +} + +test.describe('working-session task-panel clearance', () => { + let fixture: MockBackendFixture | null = null + + test.beforeEach(async () => { + fixture = await setupMockBackend({ + mockServer: { holdFirstCompletionContaining: TASK_PANEL_RESUME_TRIGGER }, + }) + await waitForAppReady(fixture, 120_000) + }) + + test.afterEach(async () => { + await fixture?.cleanup() + fixture = null + }) + + test('window focus reanchors a working session above the expanded task panel', async ({}, testInfo) => { + const page = fixture!.page + + await send(page, PROMPT) + await fixture!.mock.waitForHeldCompletion() + await openFreshDraft(page) + + // Re-open while the long response is still streaming. Its todo call lands + // afterward, so the already-visible composer grows only after the initial + // session-load scroll settle has finished. + fixture!.mock.releaseHeldStream() + await page.waitForTimeout(1_000) + await reopenWorkingSession(page) + await expect(activeSurface(page).getByText('Tasks 1/5')).toBeVisible({ timeout: 30_000 }) + + // Reproduce the stale geometry at the foreground boundary. Active turns + // disable Chromium's background throttling, so visibility can stay `visible` + // and window focus is the only foreground edge that can repair it. + await page.waitForTimeout(750) + const staleState = await activeSurface(page) + .locator('[data-slot="aui_thread-viewport"]') + .evaluate(viewport => { + // Grow scrollHeight before the observed thread-content node. This + // shifts the transcript behind the dock without resizing the observed + // node or synthesizing a user scroll (which must escape the lock). + const staleClearance = document.createElement('div') + staleClearance.style.height = '160px' + staleClearance.setAttribute('aria-hidden', 'true') + viewport.prepend(staleClearance) + + const distance = viewport.scrollHeight - viewport.clientHeight - viewport.scrollTop + const surface = viewport.closest('[data-composer-target]')! + const latest = Array.from(surface.querySelectorAll('[data-role="assistant"]')).at(-1)! + const status = surface.querySelector('[data-slot="composer-status-stack"]')! + + window.dispatchEvent(new Event('focus')) + + return { + distance, + following: viewport.dataset.following, + latestMessageBottom: latest.getBoundingClientRect().bottom, + statusPanelTop: status.getBoundingClientRect().top, + visibility: document.visibilityState, + } + }) + + expect(staleState.visibility, JSON.stringify(staleState)).toBe('visible') + expect(staleState.following, JSON.stringify(staleState)).toBe('true') + expect(staleState.distance).toBeGreaterThan(100) + expect(staleState.latestMessageBottom, JSON.stringify(staleState)).toBeGreaterThan(staleState.statusPanelTop) + await page.waitForTimeout(1_000) + const metrics = await clearanceMetrics(page) + await page.screenshot({ path: testInfo.outputPath('task-panel-after-resume.png') }) + + expect(metrics.composerHeight, JSON.stringify(metrics)).toBeGreaterThanOrEqual(190) + expect(metrics.distanceFromBottom, JSON.stringify(metrics)).toBeLessThan(staleState.distance / 2) + expect(metrics.latestMessageBottom, JSON.stringify(metrics)).toBeLessThanOrEqual(metrics.statusPanelTop) + }) +}) diff --git a/apps/desktop/src/app/chat/composer/status-stack/index.tsx b/apps/desktop/src/app/chat/composer/status-stack/index.tsx index c6dd3ff7b7bb8..93f14c8e52d7a 100644 --- a/apps/desktop/src/app/chat/composer/status-stack/index.tsx +++ b/apps/desktop/src/app/chat/composer/status-stack/index.tsx @@ -240,6 +240,7 @@ export function ComposerStatusStack({ queue, sessionId }: ComposerStatusStackPro // bottom-anchored, so this grows upward over the thread without needing // to be positioned — and it shares the dock's left edge for free. className="flex max-h-[40vh] min-h-0 flex-col overflow-y-auto" + data-slot="composer-status-stack" onPointerDownCapture={() => blurComposerInput()} > {/* The card paints the shared --composer-fill (rest / scrolled / focused diff --git a/apps/desktop/src/components/assistant-ui/thread/list.test.ts b/apps/desktop/src/components/assistant-ui/thread/list.test.ts index f6d6a77e02731..8747bcc31ab1e 100644 --- a/apps/desktop/src/components/assistant-ui/thread/list.test.ts +++ b/apps/desktop/src/components/assistant-ui/thread/list.test.ts @@ -1,4 +1,4 @@ -import { describe, expect, it } from 'vitest' +import { afterEach, describe, expect, it, vi } from 'vitest' import { buildGroups, @@ -9,9 +9,82 @@ import { liveTailStart, type MessageGroup, resolveThreadScrollTarget, + subscribeToThreadForeground, transcriptPaneBudget } from './list' +afterEach(() => { + vi.restoreAllMocks() +}) + +describe('subscribeToThreadForeground', () => { + it('reanchors on focus when an active turn keeps document visibility pinned visible', () => { + const reanchor = vi.fn() + + const raf = vi.spyOn(window, 'requestAnimationFrame').mockImplementation(callback => { + callback(0) + + return 1 + }) + + const unsubscribe = subscribeToThreadForeground(() => true, reanchor) + + window.dispatchEvent(new Event('focus')) + + expect(raf).toHaveBeenCalledOnce() + expect(reanchor).toHaveBeenCalledOnce() + unsubscribe() + }) + + it('leaves a scrolled-up reader in place when the window focuses', () => { + const reanchor = vi.fn() + const raf = vi.spyOn(window, 'requestAnimationFrame') + const unsubscribe = subscribeToThreadForeground(() => false, reanchor) + + window.dispatchEvent(new Event('focus')) + + expect(raf).not.toHaveBeenCalled() + expect(reanchor).not.toHaveBeenCalled() + unsubscribe() + }) + + it('drops a queued reanchor when the reader scrolls away before the frame', () => { + const frames: FrameRequestCallback[] = [] + let following = true + const reanchor = vi.fn() + + vi.spyOn(window, 'requestAnimationFrame').mockImplementation(callback => { + frames.push(callback) + + return 7 + }) + + const unsubscribe = subscribeToThreadForeground(() => following, reanchor) + + window.dispatchEvent(new Event('focus')) + following = false + frames[0]?.(0) + + expect(reanchor).not.toHaveBeenCalled() + unsubscribe() + }) + + it('cancels a queued reanchor when its thread unmounts', () => { + const cancel = vi.spyOn(window, 'cancelAnimationFrame') + const reanchor = vi.fn() + + vi.spyOn(window, 'requestAnimationFrame').mockReturnValue(9) + + const unsubscribe = subscribeToThreadForeground(() => true, reanchor) + + window.dispatchEvent(new Event('focus')) + unsubscribe() + + expect(cancel).toHaveBeenCalledWith(9) + expect(reanchor).not.toHaveBeenCalled() + }) +}) + // Signature rows are `${index}:${id}:${role}:${weight}` (see the useAuiState // selector in list.tsx). const signature = (rows: [string, string, number][]) => diff --git a/apps/desktop/src/components/assistant-ui/thread/list.tsx b/apps/desktop/src/components/assistant-ui/thread/list.tsx index b0d2c1cf3a1e6..9db5aedcb16fd 100644 --- a/apps/desktop/src/components/assistant-ui/thread/list.tsx +++ b/apps/desktop/src/components/assistant-ui/thread/list.tsx @@ -22,7 +22,6 @@ import { useI18n } from '@/i18n' import { messagePaintWeight } from '@/lib/render-weight' import { cn } from '@/lib/utils' import { - $threadScrolledUp, onScrollToBottomRequest, onThreadEditClose, onThreadEditOpen, @@ -125,6 +124,49 @@ export const resolveThreadScrollTarget: GetTargetScrollTop = (targetScrollTop, { return remaining >= 0 && remaining <= SCROLL_TARGET_EPSILON_PX ? currentScrollTop : targetScrollTop } +export function subscribeToThreadForeground(shouldReanchor: () => boolean, onReanchor: () => void): () => void { + let frameId: number | null = null + let framePending = false + + const onForeground = () => { + if (framePending || document.visibilityState !== 'visible' || !shouldReanchor()) { + return + } + + framePending = true + + const scheduledId = requestAnimationFrame(() => { + frameId = null + framePending = false + + if (document.visibilityState === 'visible' && shouldReanchor()) { + onReanchor() + } + }) + + // Browser callbacks are asynchronous; the guard also keeps synchronous + // requestAnimationFrame test doubles from leaving a completed frame pending. + if (framePending) { + frameId = scheduledId + } + } + + document.addEventListener('visibilitychange', onForeground) + window.addEventListener('focus', onForeground) + + return () => { + document.removeEventListener('visibilitychange', onForeground) + window.removeEventListener('focus', onForeground) + + if (frameId !== null) { + cancelAnimationFrame(frameId) + } + + frameId = null + framePending = false + } +} + interface ThreadMessageListProps { clampToComposer: boolean components: ThreadMessageComponents @@ -526,21 +568,18 @@ const ThreadMessageListInner: FC = ({ useEffect(() => onScrollToBottomRequest(() => void scrollToBottom()), [scrollToBottom]) // Waking from display: hidden (HUD mode hides the main window; OS hide does - // the same to any window): rAF and ResizeObserver were frozen the whole - // time, so the virtualizer's measurements — and scrollTop itself — are - // stale. If the user was following the bottom, re-anchor once visible; - // leave a scrolled-up reader exactly where they were. - useEffect(() => { - const onVisible = () => { - if (document.visibilityState === 'visible' && !$threadScrolledUp.get()) { - requestAnimationFrame(() => void scrollToBottom()) - } - } - - document.addEventListener('visibilitychange', onVisible) - - return () => document.removeEventListener('visibilitychange', onVisible) - }, [scrollToBottom]) + // the same to any window): rAF and ResizeObserver may have been frozen, so + // the virtualizer's measurements — and scrollTop itself — are stale. Active + // turns disable Chromium's background throttling, which can keep visibility + // pinned at `visible`; window focus is then the only foreground edge. If the + // user was following the bottom, re-anchor on either signal. Consult this + // thread's local state rather than the composer-facing global mirror, which + // can be overwritten by another mounted pane; leave a scrolled-up reader + // exactly where they were. + useEffect( + () => subscribeToThreadForeground(() => isAtBottom, () => void scrollToBottom()), + [isAtBottom, scrollToBottom] + ) const endEditHold = useCallback(() => { scrollRef.current?.removeAttribute('data-editing') From 7d74375b0b75a2e82e931757448dcdec39beb628 Mon Sep 17 00:00:00 2001 From: Hermes Agent Date: Fri, 14 Aug 2026 18:46:33 -0700 Subject: [PATCH 065/376] fix(desktop): keep session-list scrollbar clickable beside pane sash MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The pane-resize sash's 9px grab band was centered on the split boundary, so ~4.5px reached into the leading pane and sat exactly on top of the session list's 4px scrollbar — the pointer always hit the sash (cursor flipped to col-resize) and the scrollbar thumb was unclickable/undraggable. Make the grab band asymmetric: 1px into the leading pane, 7px into the trailing one. The scrollbar regains its full hit area while the sash stays an easy 8px target; the hairline and hover strip are repositioned onto the actual boundary. Fixes #79157 --- .../pane-shell/tree/renderer/tree-split.tsx | 13 +++++++++---- 1 file changed, 9 insertions(+), 4 deletions(-) diff --git a/apps/desktop/src/components/pane-shell/tree/renderer/tree-split.tsx b/apps/desktop/src/components/pane-shell/tree/renderer/tree-split.tsx index 2ed26d311c154..e0a3772b8569c 100644 --- a/apps/desktop/src/components/pane-shell/tree/renderer/tree-split.tsx +++ b/apps/desktop/src/components/pane-shell/tree/renderer/tree-split.tsx @@ -647,7 +647,12 @@ function Sash({
{!disabled && ( @@ -669,8 +674,8 @@ function Sash({ className={cn( 'absolute bg-(--ui-sash-hover-border) opacity-0 transition-opacity duration-100 group-hover:opacity-100', horizontal - ? 'inset-y-0 left-1/2 w-(--vscode-sash-hover-size,0.25rem) -translate-x-1/2' - : 'inset-x-0 top-1/2 h-(--vscode-sash-hover-size,0.25rem) -translate-y-1/2' + ? 'inset-y-0 left-[1px] w-(--vscode-sash-hover-size,0.25rem) -translate-x-1/2' + : 'inset-x-0 top-[1px] h-(--vscode-sash-hover-size,0.25rem) -translate-y-1/2' )} /> )} From 12d99312aa24999ad5e754fbf8ba068b056bfd9c Mon Sep 17 00:00:00 2001 From: Hermes Agent Date: Fri, 14 Aug 2026 18:50:13 -0700 Subject: [PATCH 066/376] chore: map contributor email for nicolasdmolina --- contributors/emails/nicolasdmolina76@gmail.com | 1 + 1 file changed, 1 insertion(+) create mode 100644 contributors/emails/nicolasdmolina76@gmail.com diff --git a/contributors/emails/nicolasdmolina76@gmail.com b/contributors/emails/nicolasdmolina76@gmail.com new file mode 100644 index 0000000000000..db1dbd41eed0c --- /dev/null +++ b/contributors/emails/nicolasdmolina76@gmail.com @@ -0,0 +1 @@ +nicolasdmolina From c0e07837c2af9e4937321d25d68f70dc0540f517 Mon Sep 17 00:00:00 2001 From: KIAgent01 <297567825+KIAgent01@users.noreply.github.com> Date: Thu, 13 Aug 2026 07:24:26 +0200 Subject: [PATCH 067/376] fix(desktop): clear stale turn state after lost stream events Three safeguards keep finished chats from looking busy: tool rows seal on turn settle, vanished runtimes clear awaiting state and open tool parts, and late stream events no longer land in a freshly opened session. --- .../contrib/hooks/live-status-reap.test.ts | 63 ++++++++++++++++++- .../app/contrib/hooks/use-background-sync.ts | 11 +++- .../hooks/use-message-stream/gateway-event.ts | 28 ++++++++- .../session/hooks/use-message-stream/index.ts | 7 +++ apps/desktop/src/lib/chat-messages.test.ts | 63 +++++++++++++++++++ apps/desktop/src/lib/chat-messages.ts | 41 ++++++++++++ apps/desktop/src/lib/gateway-events.test.ts | 25 ++++++++ apps/desktop/src/lib/gateway-events.ts | 15 ++++- 8 files changed, 246 insertions(+), 7 deletions(-) diff --git a/apps/desktop/src/app/contrib/hooks/live-status-reap.test.ts b/apps/desktop/src/app/contrib/hooks/live-status-reap.test.ts index 8b4688331c0e2..b5d92aa13ae4e 100644 --- a/apps/desktop/src/app/contrib/hooks/live-status-reap.test.ts +++ b/apps/desktop/src/app/contrib/hooks/live-status-reap.test.ts @@ -1,7 +1,14 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' -import { $selectedStoredSessionId, $unreadFinishedSessionIds } from '@/store/session' -import { $attentionSessionIds, $workingSessionIds, clearAllSessionStates } from '@/store/session-states' +import { createClientSessionState } from '@/lib/chat-runtime' +import { $activeSessionId, $selectedStoredSessionId, $unreadFinishedSessionIds } from '@/store/session' +import { + $attentionSessionIds, + $sessionStates, + $workingSessionIds, + clearAllSessionStates, + publishSessionState +} from '@/store/session-states' import { rehydrateLiveSessionStatuses } from './use-background-sync' @@ -25,6 +32,7 @@ describe('rehydrateLiveSessionStatuses — reaping vanished runtimes', () => { vi.useRealTimers() clearAllSessionStates() $unreadFinishedSessionIds.set([]) + $activeSessionId.set(null) }) it('clears a working session that disappears from the live snapshot', () => { @@ -76,4 +84,55 @@ describe('rehydrateLiveSessionStatuses — reaping vanished runtimes', () => { expect($workingSessionIds.get()).toEqual(['stored-other']) }) + + it('seals open tool parts and clears awaitingResponse when a session vanishes', () => { + const openTool = { + type: 'tool-call', + toolCallId: 'call-1', + toolName: 'patch', + args: {}, + argsText: '{}' + } as never + + publishSessionState('runtime-tools', { + ...createClientSessionState('stored-tools'), + busy: true, + awaitingResponse: true, + messages: [ + { id: 'a1', role: 'assistant', parts: [openTool], pending: false } as never + ] + }) + + // Keep the runtime referenced so the settled state stays in the store + // instead of being evicted as no-longer-needed. + $activeSessionId.set('runtime-tools') + + rehydrateLiveSessionStatuses({ + sessions: [{ id: 'runtime-tools', session_key: 'stored-tools', status: 'working' }] + }) + rehydrateLiveSessionStatuses({ sessions: [] }) + + const state = $sessionStates.get()['runtime-tools'] + + expect(state.busy).toBe(false) + expect(state.awaitingResponse).toBe(false) + expect((state.messages[0].parts[0] as { result?: unknown }).result).toBeDefined() + }) + + it('clears a session stuck awaiting a response without the busy flag', () => { + publishSessionState('runtime-await', { + ...createClientSessionState('stored-await'), + awaitingResponse: true, + busy: false + }) + + $activeSessionId.set('runtime-await') + + rehydrateLiveSessionStatuses({ + sessions: [{ id: 'runtime-await', session_key: 'stored-await', status: 'working' }] + }) + rehydrateLiveSessionStatuses({ sessions: [] }) + + expect($sessionStates.get()['runtime-await'].awaitingResponse).toBe(false) + }) }) diff --git a/apps/desktop/src/app/contrib/hooks/use-background-sync.ts b/apps/desktop/src/app/contrib/hooks/use-background-sync.ts index e42084999c17d..1776d647101a6 100644 --- a/apps/desktop/src/app/contrib/hooks/use-background-sync.ts +++ b/apps/desktop/src/app/contrib/hooks/use-background-sync.ts @@ -2,7 +2,7 @@ import { useStore } from '@nanostores/react' import { type MutableRefObject, useCallback, useEffect, useRef } from 'react' import { getLatestSessionMessages } from '@/hermes' -import { preserveLocalAssistantErrors, toChatMessages } from '@/lib/chat-messages' +import { preserveLocalAssistantErrors, sealOpenToolParts, toChatMessages } from '@/lib/chat-messages' import { createClientSessionState } from '@/lib/chat-runtime' import { sessionMessagesSignature } from '@/lib/session-signatures' import { $changeEventsAvailable, $cronChangeTick, $sessionsChangeTick } from '@/store/live-sync' @@ -247,14 +247,19 @@ export function rehydrateLiveSessionStatuses( const existing = $sessionStates.get()[runtimeSessionId] - if (existing?.busy || existing?.needsInput) { + if (existing?.busy || existing?.needsInput || existing?.awaitingResponse) { publishSessionState(runtimeSessionId, { ...existing, awaitingResponse: false, busy: false, needsInput: false, streamId: null, - turnStartedAt: null + turnStartedAt: null, + // The turn ended without its completion events reaching us — a lost + // `tool.complete` would otherwise leave a spinning tool row in an + // idle session. Seal open tool parts the same way the settle path + // does, so the transcript matches the state. + messages: sealOpenToolParts(existing.messages) }) } } diff --git a/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event.ts b/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event.ts index 4592653758bfc..238f3c9547be3 100644 --- a/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event.ts +++ b/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event.ts @@ -12,7 +12,7 @@ import { translateNow } from '@/i18n' import { type GatewayEventPayload, textPart } from '@/lib/chat-messages' import { coerceGatewayText, coerceThinkingText, normalizePersonalityValue } from '@/lib/chat-runtime' import { playCompletionSound } from '@/lib/completion-sound' -import { approvalReplaySessionId, resolveGatewayEventSessionId } from '@/lib/gateway-events' +import { approvalReplaySessionId, resolveGatewayEventSessionId, UNSCOPED_STREAM_EVENT_TYPES } from '@/lib/gateway-events' import { triggerHaptic } from '@/lib/haptics' import { modelOptionsQueryKey } from '@/lib/model-options' import { isProviderSetupErrorMessage } from '@/lib/provider-setup-errors' @@ -321,6 +321,32 @@ export function useGatewayEventHandler(deps: GatewayEventDeps) { } const sessionId = route.sessionId + + // Late stragglers: an unscoped stream event attributed via the + // active-session fallback (no pin) to a session that has no live turn + // belongs to a turn that already ended elsewhere. Dropping it keeps the + // previous session's tail events (a delayed `thinking.delta` or + // `status.update`) from landing in a freshly opened chat (#43142 family: + // busy/streaming UI inherited when switching sessions). + if ( + sessionId && + !explicitSid && + !route.pinned && + event.type && + event.type !== 'message.start' && + UNSCOPED_STREAM_EVENT_TYPES.has(event.type) + ) { + const state = sessionStateByRuntimeIdRef.current.get(sessionId) + + const hasLiveTurn = Boolean( + state && (state.awaitingResponse || state.busy || state.streamId || state.sawAssistantPayload) + ) + + if (!hasLiveTurn) { + return + } + } + const isActiveEvent = !!sessionId && sessionId === activeSessionIdRef.current const replaySessionId = approvalReplaySessionId(event.type, activeSessionIdRef.current, sessionId) diff --git a/apps/desktop/src/app/session/hooks/use-message-stream/index.ts b/apps/desktop/src/app/session/hooks/use-message-stream/index.ts index 01cf8df499a6f..745c95db59d20 100644 --- a/apps/desktop/src/app/session/hooks/use-message-stream/index.ts +++ b/apps/desktop/src/app/session/hooks/use-message-stream/index.ts @@ -13,6 +13,7 @@ import { mergeFinalAssistantText, reasoningPart, renderMediaTags, + sealOpenToolParts, upsertToolPart } from '@/lib/chat-messages' import { @@ -666,6 +667,12 @@ export function useMessageStream({ } } + // Turn-settle reconciliation: a `tool.complete` event lost to a + // degraded websocket leaves its tool row spinning forever. The turn is + // provably done here — nothing can still be running — so seal any + // tool-call parts that never saw their completion event. + nextMessages = sealOpenToolParts(nextMessages) + const hasInlineError = nextMessages.some(m => m.role === 'assistant' && m.error && !m.hidden) const lastVisible = [...nextMessages].reverse().find(m => !m.hidden) const unresolvedUserTail = lastVisible?.role === 'user' diff --git a/apps/desktop/src/lib/chat-messages.test.ts b/apps/desktop/src/lib/chat-messages.test.ts index fa1a6744cd95d..af178036a46f8 100644 --- a/apps/desktop/src/lib/chat-messages.test.ts +++ b/apps/desktop/src/lib/chat-messages.test.ts @@ -12,6 +12,7 @@ import { preserveLocalAssistantErrors, reasoningPart, renderMediaTags, + sealOpenToolParts, toChatMessages, upsertToolPart } from './chat-messages' @@ -1166,3 +1167,65 @@ describe('collectUnspokenTurnSpeech', () => { expect(collectUnspokenTurnSpeech([user('u1', 'hello'), assistant('a1', '')], null)).toBeNull() }) }) + +describe('sealOpenToolParts', () => { + const toolPart = (over: Partial = {}): ChatMessagePart => + ({ + type: 'tool-call', + toolCallId: 'call-1', + toolName: 'terminal', + args: {}, + argsText: '{}', + ...over + }) as ChatMessagePart + + const assistantWithParts = (parts: ChatMessagePart[], over: Partial = {}): ChatMessage => + ({ + id: 'a1', + role: 'assistant', + parts, + ...over + }) as ChatMessage + + it('seals open tool-call parts in settled assistant messages', () => { + const messages = [assistantWithParts([toolPart()])] + + const next = sealOpenToolParts(messages) + + expect(next[0].parts[0]).toHaveProperty('result') + }) + + it('leaves already-completed tool parts untouched', () => { + const done = toolPart({ result: { code: 0 } }) + const messages = [assistantWithParts([done])] + + const next = sealOpenToolParts(messages) + + expect(next[0].parts[0]).toBe(done) + }) + + it('leaves pending messages alone', () => { + const messages = [assistantWithParts([toolPart()], { pending: true })] + + const next = sealOpenToolParts(messages) + + expect(next[0].parts[0]).not.toHaveProperty('result') + }) + + it('leaves non-tool parts untouched', () => { + const text = { type: 'text', text: 'hello' } as ChatMessagePart + const messages = [assistantWithParts([text, toolPart()])] + + const next = sealOpenToolParts(messages) + + expect(next[0].parts[0]).toBe(text) + expect(next[0].parts[1]).toHaveProperty('result') + }) + + it('returns the same array reference when nothing needs sealing', () => { + const done = toolPart({ result: { code: 0 } }) + const messages = [assistantWithParts([done])] + + expect(sealOpenToolParts(messages)).toBe(messages) + }) +}) diff --git a/apps/desktop/src/lib/chat-messages.ts b/apps/desktop/src/lib/chat-messages.ts index fa98550a006d2..3328672e7ddb7 100644 --- a/apps/desktop/src/lib/chat-messages.ts +++ b/apps/desktop/src/lib/chat-messages.ts @@ -711,6 +711,47 @@ export function upsertToolPart( return next } +/** + * Turn-settle reconciliation: close every tool-call part that never received + * its completion event. A `tool.complete` lost to a degraded websocket + * (reconnect, profile swap, hidden window) leaves the part without a `result`, + * which renders as a permanently spinning tool row even though the turn itself + * completed. A settled session cannot have tools still running, so an open + * part at settle time is a lost event, not live work. Pending messages are + * left alone, and no-op calls return the input array unchanged. + */ +export function sealOpenToolParts(messages: ChatMessage[]): ChatMessage[] { + let changed = false + + const next = messages.map(message => { + if (message.role !== 'assistant' || message.pending) { + return message + } + + let partChanged = false + + const parts = message.parts.map(part => { + if (part.type !== 'tool-call' || Object.hasOwn(part, 'result')) { + return part + } + + partChanged = true + + return { ...part, result: {} } + }) + + if (!partChanged) { + return message + } + + changed = true + + return { ...message, parts } + }) + + return changed ? next : messages +} + function recordFromUnknown(value: unknown): Record | null { return value && typeof value === 'object' ? (value as Record) : null } diff --git a/apps/desktop/src/lib/gateway-events.test.ts b/apps/desktop/src/lib/gateway-events.test.ts index 907acbdd80ad0..5ec527d0bcfb5 100644 --- a/apps/desktop/src/lib/gateway-events.test.ts +++ b/apps/desktop/src/lib/gateway-events.test.ts @@ -43,6 +43,7 @@ describe('gateway event routing', () => { expect(started).toEqual({ drop: false, nextUnscopedStreamSessionId: 'session-a', + pinned: false, sessionId: 'session-a' }) @@ -56,6 +57,7 @@ describe('gateway event routing', () => { expect(delta).toEqual({ drop: false, nextUnscopedStreamSessionId: 'session-a', + pinned: true, sessionId: 'session-a' }) @@ -69,6 +71,7 @@ describe('gateway event routing', () => { expect(completed).toEqual({ drop: false, nextUnscopedStreamSessionId: null, + pinned: true, sessionId: 'session-a' }) }) @@ -84,6 +87,27 @@ describe('gateway event routing', () => { expect(routed).toEqual({ drop: false, nextUnscopedStreamSessionId: 'session-b', + pinned: false, + sessionId: 'session-b' + }) + }) + + it('attributes an unpinned stream event to the active session without the pin flag', () => { + // A late straggler (no pin left after the previous turn completed) falls + // back to the active session. The handler drops this case when the target + // session has no live turn — the straggler belongs to a turn that already + // ended elsewhere (#43142 family). + const routed = resolveGatewayEventSessionId({ + activeSessionId: 'session-b', + eventType: 'thinking.delta', + explicitSessionId: '', + unscopedStreamSessionId: null + }) + + expect(routed).toEqual({ + drop: false, + nextUnscopedStreamSessionId: null, + pinned: false, sessionId: 'session-b' }) }) @@ -99,6 +123,7 @@ describe('gateway event routing', () => { expect(routed).toEqual({ drop: false, nextUnscopedStreamSessionId: null, + pinned: true, sessionId: 'session-a' }) }) diff --git a/apps/desktop/src/lib/gateway-events.ts b/apps/desktop/src/lib/gateway-events.ts index c937acf50c255..6375955510de4 100644 --- a/apps/desktop/src/lib/gateway-events.ts +++ b/apps/desktop/src/lib/gateway-events.ts @@ -17,7 +17,12 @@ function asRecord(payload: unknown): Record { * Without this, ``explicitSid || activeSessionId`` reattributes live deltas to * the newly focused chat. */ -const UNSCOPED_STREAM_EVENT_TYPES = new Set([ +/** Unscoped stream events that must stay pinned to the session that received + * ``message.start`` after the user switches chats mid-turn (#47709 / #48281). + * Without this, ``explicitSid || activeSessionId`` reattributes live deltas to + * the newly focused chat. Exported so the event handler can tell which events + * are pin-eligible when deciding whether an unpinned straggler is legitimate. */ +export const UNSCOPED_STREAM_EVENT_TYPES = new Set([ 'approval.request', 'browser.progress', 'clarify.request', @@ -67,6 +72,11 @@ export interface GatewayEventSessionRouteInput { export interface GatewayEventSessionRoute { drop: boolean nextUnscopedStreamSessionId: null | string + /** True when the event was attributed via the pinned stream session rather + * than the active-session fallback. The caller uses this to drop late + * stragglers: an unpinned stream event landing on a session that has no + * live turn belongs to a turn that already ended elsewhere. */ + pinned: boolean sessionId: null | string } @@ -108,6 +118,7 @@ export function resolveGatewayEventSessionId({ return { drop: false, nextUnscopedStreamSessionId, + pinned: true, sessionId: explicitSessionId } } @@ -116,6 +127,7 @@ export function resolveGatewayEventSessionId({ return { drop: true, nextUnscopedStreamSessionId: unscopedStreamSessionId, + pinned: false, sessionId: null } } @@ -140,6 +152,7 @@ export function resolveGatewayEventSessionId({ return { drop: false, nextUnscopedStreamSessionId, + pinned: streamEvent && eventType !== 'message.start' && Boolean(unscopedStreamSessionId), sessionId } } From 6202fca05915be1801024c94f17e757ff6a71d0f Mon Sep 17 00:00:00 2001 From: razultull Date: Fri, 14 Aug 2026 19:02:39 -0700 Subject: [PATCH 068/376] fix(desktop): keep session marked in-progress during async delegation MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Salvaged from #51358 (razultull), rebuilt against the rewritten status architecture on main. The original PR patched setSessionWorking / noteSessionActivity / the statusbar counter, all of which have since been replaced (busy now lives in session-states.ts, the watchdog only marks stalled, and the statusbar Agents item already shows a pure subagent count — the conflated counter the PR split no longer exists). What still applied is the core bug: a parent that delegates via delegate_task(background=true) ends its own turn the moment the handle returns, so the sidebar row dropped to a plain idle dot while the spawned subagents kept working for minutes — the session read as "done" mid-task. Add $delegatingSessionIds — sessions whose subagents are still queued or running — as an input to the session dot projection, claiming the same 'background' treatment as running background processes (and yielding to 'working' while the parent turn itself is live). Uses the same runtime→stored bridge, lineage aliasing, and fresh-chat runtime-id fallback as $backgroundRunningSessionIds, and clears by construction the moment the last subagent reaches a terminal status. --- .../src/store/session-dot-state.test.ts | 52 ++++++++++++++++++- apps/desktop/src/store/session-dot-state.ts | 47 +++++++++++++++-- 2 files changed, 94 insertions(+), 5 deletions(-) diff --git a/apps/desktop/src/store/session-dot-state.test.ts b/apps/desktop/src/store/session-dot-state.test.ts index 5e81c3201b14d..5f9120a391515 100644 --- a/apps/desktop/src/store/session-dot-state.test.ts +++ b/apps/desktop/src/store/session-dot-state.test.ts @@ -1,6 +1,11 @@ -import { describe, expect, it } from 'vitest' +import { afterEach, describe, expect, it } from 'vitest' -import { hasLiveTurn, showsRunningArc } from './session-dot-state' +import { createClientSessionState } from '@/lib/chat-runtime' + +import { $sessions } from './session' +import { $delegatingSessionIds, hasLiveTurn, showsRunningArc } from './session-dot-state' +import { clearAllSessionStates, publishSessionState } from './session-states' +import { $subagentsBySession, type SubagentProgress } from './subagents' describe('showsRunningArc', () => { it('keeps the arc when an authoritative turn goes quiet', () => { @@ -35,3 +40,46 @@ describe('hasLiveTurn', () => { expect(hasLiveTurn('unread')).toBe(false) }) }) + +describe('$delegatingSessionIds', () => { + const subagent = (status: SubagentProgress['status']): SubagentProgress => ({ + id: 'sub-1', + parentId: null, + goal: 'do a thing', + status, + taskCount: 1, + taskIndex: 0, + startedAt: 0, + updatedAt: 0, + filesRead: [], + filesWritten: [], + stream: [] + }) + + afterEach(() => { + clearAllSessionStates() + $subagentsBySession.set({}) + $sessions.set([]) + }) + + it('claims the stored id while a subagent is running after the parent turn ended', () => { + publishSessionState('runtime-1', { ...createClientSessionState('stored-1'), busy: false }) + $subagentsBySession.set({ 'runtime-1': [subagent('running')] }) + + expect($delegatingSessionIds.get()).toContain('stored-1') + }) + + it('drops the session once every subagent reaches a terminal status', () => { + publishSessionState('runtime-1', { ...createClientSessionState('stored-1'), busy: false }) + $subagentsBySession.set({ 'runtime-1': [subagent('running')] }) + $subagentsBySession.set({ 'runtime-1': [subagent('completed')] }) + + expect($delegatingSessionIds.get()).not.toContain('stored-1') + }) + + it('falls back to the runtime id for a not-yet-persisted conversation', () => { + $subagentsBySession.set({ 'runtime-fresh': [subagent('queued')] }) + + expect($delegatingSessionIds.get()).toContain('runtime-fresh') + }) +}) diff --git a/apps/desktop/src/store/session-dot-state.ts b/apps/desktop/src/store/session-dot-state.ts index bf94a1658eddb..7a83e7faa33db 100644 --- a/apps/desktop/src/store/session-dot-state.ts +++ b/apps/desktop/src/store/session-dot-state.ts @@ -19,11 +19,46 @@ import { computed } from 'nanostores' -import { stableRecord } from '@/lib/stable-array' +import { stableArray, stableRecord } from '@/lib/stable-array' import { $backgroundRunningSessionIds } from './composer-status' import { $sessions, $unreadFinishedSessionIds, lineageAliases } from './session' -import { $attentionSessionIds, $draftSessionIds, $stalledSessionIds, $workingSessionIds } from './session-states' +import { + $attentionSessionIds, + $draftSessionIds, + $sessionStates, + $stalledSessionIds, + $workingSessionIds +} from './session-states' +import { $subagentsBySession, activeSubagentCount } from './subagents' + +// Sessions parked in async delegation: the parent turn has ended (busy=false — +// delegate_task(background=true) returns its handle the moment the children +// are spawned) while those subagents keep working for minutes. Without this +// input the sidebar row dropped to a plain idle dot mid-delegation, reading as +// "done" while work was still running in child sessions. Same runtime→stored +// bridge and fresh-chat fallback as $backgroundRunningSessionIds: +// $subagentsBySession is keyed by runtime id, surfaces key on stored ids, and +// lineageAliases covers whichever tip of the conversation a surface holds. +let delegatingIds: readonly string[] = [] +export const $delegatingSessionIds = computed( + [$subagentsBySession, $sessionStates, $sessions], + (bySession, states, sessions) => { + const ids = new Set() + + for (const [runtimeId, items] of Object.entries(bySession)) { + if (activeSubagentCount(items) === 0) { + continue + } + + for (const alias of lineageAliases(states[runtimeId]?.storedSessionId ?? runtimeId, sessions)) { + ids.add(alias) + } + } + + return (delegatingIds = stableArray(delegatingIds, [...ids])) + } +) export type SessionDotState = 'background' | 'draft' | 'idle' | 'needs-input' | 'stalled' | 'unread' | 'working' @@ -63,11 +98,12 @@ export const $sessionDotStateById = computed( $workingSessionIds, $stalledSessionIds, $backgroundRunningSessionIds, + $delegatingSessionIds, $unreadFinishedSessionIds, $draftSessionIds, $sessions ], - (attention, working, stalled, background, unread, draft, sessions) => { + (attention, working, stalled, background, delegating, unread, draft, sessions) => { const next: Record = {} const claim = (ids: readonly string[], state: SessionDotState) => { @@ -87,6 +123,11 @@ export const $sessionDotStateById = computed( claim(draft, 'draft') claim(unread, 'unread') claim(background, 'background') + // Async delegation: the parent turn has ended but its subagents are still + // running, so the session's work continues in child sessions. Same visual + // claim as background processes — and it yields to `working` below the + // moment the parent turn itself is live (synchronous orchestrator children). + claim(delegating, 'background') claim(working, 'working') // Stalled REFINES working rather than rivalling it — the turn is still From cb3ca0af0edda6c3212807e0a03ccc43a74f6fc4 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 19:06:59 -0700 Subject: [PATCH 069/376] chore: add contributor email mapping for razultull --- contributors/emails/razultull@gmail.com | 1 + 1 file changed, 1 insertion(+) create mode 100644 contributors/emails/razultull@gmail.com diff --git a/contributors/emails/razultull@gmail.com b/contributors/emails/razultull@gmail.com new file mode 100644 index 0000000000000..734d2ec892ebe --- /dev/null +++ b/contributors/emails/razultull@gmail.com @@ -0,0 +1 @@ +razultull From 662a77f31e7787aca62b0e868625c3fc014b6b4d Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 19:07:23 -0700 Subject: [PATCH 070/376] fix(desktop): don't let a stale resume snapshot mark an open chat idle MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Fixes the #70449 symptom that remained after the two salvaged commits: opening or viewing a chat whose turn is still running cleared its working indicator. `session.activate` / `session.resume` report `running` as a snapshot taken when the RPC was issued; a turn that started or kept streaming while the RPC was in flight has already marked the runtime busy in the live cache, and both resume paths overwrote that newer truth with the stale `running: false`, dropping the session out of the working set and painting it done mid-turn. Add `resolveResumedBusy`: a snapshot saying running always wins (adopting a live turn is never stale), but a snapshot saying idle can no longer rewind a live busy — the turn's own terminal signal (running:false via session.info / the settle path) stays the only authority that ends it, and the background-sync reaper still clears truly lost turns. Wired into both the warm `session.activate` path and the cold `session.resume` path, reading the freshest cache entry rather than the pre-await state. Includes an eslint --fix formatting pass over the touched files. --- .../contrib/hooks/live-status-reap.test.ts | 4 +--- .../hooks/use-session-actions/index.ts | 18 +++++++++++++++-- .../hooks/use-session-actions/utils.test.ts | 19 ++++++++++++++++++ .../hooks/use-session-actions/utils.ts | 20 +++++++++++++++++++ 4 files changed, 56 insertions(+), 5 deletions(-) diff --git a/apps/desktop/src/app/contrib/hooks/live-status-reap.test.ts b/apps/desktop/src/app/contrib/hooks/live-status-reap.test.ts index b5d92aa13ae4e..cca8adf154511 100644 --- a/apps/desktop/src/app/contrib/hooks/live-status-reap.test.ts +++ b/apps/desktop/src/app/contrib/hooks/live-status-reap.test.ts @@ -98,9 +98,7 @@ describe('rehydrateLiveSessionStatuses — reaping vanished runtimes', () => { ...createClientSessionState('stored-tools'), busy: true, awaitingResponse: true, - messages: [ - { id: 'a1', role: 'assistant', parts: [openTool], pending: false } as never - ] + messages: [{ id: 'a1', role: 'assistant', parts: [openTool], pending: false } as never] }) // Keep the runtime referenced so the settled state stays in the store diff --git a/apps/desktop/src/app/session/hooks/use-session-actions/index.ts b/apps/desktop/src/app/session/hooks/use-session-actions/index.ts index e11adefdba9ce..88deda8dcb3e1 100644 --- a/apps/desktop/src/app/session/hooks/use-session-actions/index.ts +++ b/apps/desktop/src/app/session/hooks/use-session-actions/index.ts @@ -86,6 +86,7 @@ import { patchSessionWorkspace, preserveLocalPendingTurnMessages, reconcileResumeMessages, + resolveResumedBusy, resolveSessionProfile, resolveStoredSession, sessionMatchesStoredId, @@ -830,7 +831,13 @@ export function useSessionActions({ ? reconcileAuthoritativeMessages(activated.messages, cachedViewState.messages, activated) : cachedViewState.messages - const running = Boolean(activated.running ?? cachedViewState.busy) + // #70449: never let the activate snapshot's stale running:false + // rewind a turn that started while the RPC was in flight — read + // the freshest cache entry, not the pre-await cachedViewState. + const running = resolveResumedBusy( + activated.running ?? cachedViewState.busy, + Boolean(sessionStateByRuntimeIdRef.current.get(cachedRuntimeId)?.busy) + ) // While idle, the persisted REST transcript is the display // authority: session.activate returns the runtime's compressed @@ -1051,7 +1058,14 @@ export function useSessionActions({ return chatMessageArraysEquivalent(currentMessages, resumedMessages) ? currentMessages : resumedMessages })() - resumedRunning = Boolean((resumed as { running?: boolean }).running) + // #70449: same stale-snapshot guard as the warm path — a turn that + // started while the resume RPC was in flight has already marked the + // rebound runtime busy via gateway events; the snapshot must not + // rewind it to idle just because the user opened the chat. + resumedRunning = resolveResumedBusy( + (resumed as { running?: boolean }).running, + Boolean(sessionStateByRuntimeIdRef.current.get(resumed.session_id)?.busy) + ) // Crash-survivable turn progress: fold a journaled in-flight tail // (persisted by use-session-state-cache while the turn streamed; diff --git a/apps/desktop/src/app/session/hooks/use-session-actions/utils.test.ts b/apps/desktop/src/app/session/hooks/use-session-actions/utils.test.ts index 27bff4795e07e..1e2667e9372ac 100644 --- a/apps/desktop/src/app/session/hooks/use-session-actions/utils.test.ts +++ b/apps/desktop/src/app/session/hooks/use-session-actions/utils.test.ts @@ -25,6 +25,7 @@ import { isSessionGoneError, preserveLocalPendingTurnMessages, reconcileResumeMessages, + resolveResumedBusy, sessionMatchesStoredId, sessionShouldHaveTranscript, toBranchMessages @@ -1428,3 +1429,21 @@ describe('appendLiveSessionProjection', () => { }) }) }) + +describe('resolveResumedBusy', () => { + it('keeps a live busy turn when the resume snapshot stalely reports idle (#70449)', () => { + expect(resolveResumedBusy(false, true)).toBe(true) + expect(resolveResumedBusy(undefined, true)).toBe(true) + expect(resolveResumedBusy(null, true)).toBe(true) + }) + + it('clears busy when both the snapshot and the live cache agree the turn ended', () => { + expect(resolveResumedBusy(false, false)).toBe(false) + expect(resolveResumedBusy(undefined, false)).toBe(false) + }) + + it('adopts a running turn reported by the snapshot even without live state', () => { + expect(resolveResumedBusy(true, false)).toBe(true) + expect(resolveResumedBusy(true, true)).toBe(true) + }) +}) diff --git a/apps/desktop/src/app/session/hooks/use-session-actions/utils.ts b/apps/desktop/src/app/session/hooks/use-session-actions/utils.ts index db6132c556f5f..cfa9477f9ba1a 100644 --- a/apps/desktop/src/app/session/hooks/use-session-actions/utils.ts +++ b/apps/desktop/src/app/session/hooks/use-session-actions/utils.ts @@ -1255,3 +1255,23 @@ export function isSessionGoneError(err: unknown): boolean { return message.includes('404') || /session not found/i.test(message) } + +/** + * The busy value a resume/activate response should land with (#70449). + * + * `running` in a `session.activate` / `session.resume` payload is a snapshot + * taken when the RPC was issued. A turn that started — or streamed — after + * that snapshot has already marked the runtime busy in the live cache, so a + * stale `running: false` must never rewind it: that is exactly how opening an + * in-progress chat cleared its working indicator while the agent was still + * going. Preserving the newer live busy is safe, because the turn's own + * terminal signal (running:false via session.info / the settle path) remains + * the only authority that ends it, and the background-sync reaper clears + * truly lost turns. + * + * A snapshot that says `running: true` always wins — adopting a live turn is + * never stale. + */ +export function resolveResumedBusy(snapshotRunning: boolean | null | undefined, liveBusy: boolean): boolean { + return Boolean(snapshotRunning) || liveBusy +} From efc3cd8697f320b57d9a1a22b977c534786baa9d Mon Sep 17 00:00:00 2001 From: "hermes-seaeye[bot]" <307254004+hermes-seaeye[bot]@users.noreply.github.com> Date: Sat, 15 Aug 2026 03:42:49 +0000 Subject: [PATCH 071/376] fmt(js): `npm run fix` on merge (#86634) Co-authored-by: github-actions[bot] --- apps/desktop/electron/main.ts | 5 ++++- apps/desktop/src/store/prompts.ts | 1 + 2 files changed, 5 insertions(+), 1 deletion(-) diff --git a/apps/desktop/electron/main.ts b/apps/desktop/electron/main.ts index d7bd748c477f2..9f2b3c9f26f03 100644 --- a/apps/desktop/electron/main.ts +++ b/apps/desktop/electron/main.ts @@ -2924,6 +2924,7 @@ function execText(command, args) { async function processStartMarker(pid) { if (process.platform === 'linux') { const stat = await fs.promises.readFile(`/proc/${pid}/stat`, 'utf8') + const fields = stat .slice(stat.lastIndexOf(')') + 1) .trim() @@ -7121,7 +7122,9 @@ async function discoverCloudAgents(org?: string) { // A 401 means the portal session lapsed (and silent renewal could not // recover it) — surface it as a re-login, not a generic failure. if (error && error.statusCode === 401) { - const err = new Error('Your Hermes Cloud session has expired. Open Settings → Gateway and sign in again.') as any + const err = new Error( + 'Your Hermes Cloud session has expired. Open Settings → Gateway and sign in again.' + ) as any err.needsCloudLogin = true err.cause = error throw err diff --git a/apps/desktop/src/store/prompts.ts b/apps/desktop/src/store/prompts.ts index c70a0965cb534..da9c7dedb5411 100644 --- a/apps/desktop/src/store/prompts.ts +++ b/apps/desktop/src/store/prompts.ts @@ -137,6 +137,7 @@ export async function replayPendingApproval(gateway: ApprovalGateway | null, ses const result = rawResult && typeof rawResult === 'object' ? (rawResult as { approvals?: PendingApprovalPayload[] }) : {} + const pending = Array.isArray(result?.approvals) ? result.approvals[0] : undefined if (!pending || typeof pending.request_id !== 'string') { From 48cca664c6cfd9929382f90a54289f44606b5c55 Mon Sep 17 00:00:00 2001 From: AlexFucuson9 Date: Thu, 16 Jul 2026 19:37:13 +0700 Subject: [PATCH 072/376] fix: detect in-stream error chunks in SSE streaming path Some OpenAI-compatible providers (DeepInfra, etc.) return validation errors as in-stream SSE chunks: HTTP 200 with choices=None and error_type/error_message in model_extra. The streaming loop silently dropped these chunks, causing EmptyStreamError ("empty stream") and pointless retries on the same bad request. Fix: check for error_type/error_message on chunks with no choices before skipping. When found, raise RuntimeError with the provider's error message so the error classifier can properly handle it. Fixes #65631 --- agent/chat_completion_helpers.py | 14 ++++++++++++++ 1 file changed, 14 insertions(+) diff --git a/agent/chat_completion_helpers.py b/agent/chat_completion_helpers.py index 3c5a90370ab66..cc6299d39c8ca 100644 --- a/agent/chat_completion_helpers.py +++ b/agent/chat_completion_helpers.py @@ -3916,6 +3916,20 @@ def _flush_pending_stream_text(): # Usage comes in the final chunk with empty choices if hasattr(chunk, "usage") and chunk.usage: usage_obj = chunk.usage + # Some OpenAI-compatible providers (DeepInfra, etc.) + # return validation errors as in-stream error chunks: + # choices=None with error_type/error_message in + # model_extra. Without this check the error is + # silently dropped and the stream ends empty → + # EmptyStreamError → misleading "empty stream" message + # and pointless retries on the same bad request. (#65631) + _err_type = getattr(chunk, "error_type", None) + _err_msg = getattr(chunk, "error_message", None) + if _err_type or _err_msg: + raise RuntimeError( + f"Provider in-stream error" + f" ({_err_type or 'unknown'}): {_err_msg or chunk}" + ) continue delta = chunk.choices[0].delta From 5d9e4aaaf2cc2edab6868d146ff5311b3eae4820 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 20:36:08 -0700 Subject: [PATCH 073/376] fix: raise ProviderStreamError for choiceless error chunks + regression tests --- agent/chat_completion_helpers.py | 16 +++++++++--- tests/run_agent/test_run_agent.py | 43 +++++++++++++++++++++++++++++++ 2 files changed, 56 insertions(+), 3 deletions(-) diff --git a/agent/chat_completion_helpers.py b/agent/chat_completion_helpers.py index cc6299d39c8ca..c11899c63824b 100644 --- a/agent/chat_completion_helpers.py +++ b/agent/chat_completion_helpers.py @@ -3926,9 +3926,19 @@ def _flush_pending_stream_text(): _err_type = getattr(chunk, "error_type", None) _err_msg = getattr(chunk, "error_message", None) if _err_type or _err_msg: - raise RuntimeError( - f"Provider in-stream error" - f" ({_err_type or 'unknown'}): {_err_msg or chunk}" + _status = _status_code_from_payload( + {"code": _err_type, "message": _err_msg} + ) or _status_code_from_value(_err_type) + raise ProviderStreamError( + status_code=_status, + body=_provider_error_body( + { + "code": _err_type or "provider_in_stream_error", + "message": str(_err_msg or chunk), + }, + _status, + ), + raw_text=f"{_err_type}: {_err_msg}", ) continue diff --git a/tests/run_agent/test_run_agent.py b/tests/run_agent/test_run_agent.py index 7ad5c391b01a3..4fb43f77d70d8 100644 --- a/tests/run_agent/test_run_agent.py +++ b/tests/run_agent/test_run_agent.py @@ -5652,6 +5652,49 @@ def test_error_finish_bare_sse_error_payload_raises_provider_error(self, agent): assert "Rate limit exceeded" in str(exc) agent.stream_delta_callback.assert_not_called() + def test_choiceless_error_chunk_raises_provider_stream_error(self, agent): + """DeepInfra-style in-stream error: choices=None + error_type/error_message. + + Regression for #65631: the choiceless-chunk skip silently dropped + error-bearing chunks, the stream ended empty, and the caller got a + misleading EmptyStreamError plus pointless retries of the same bad + request. The chunk must instead surface as ProviderStreamError so + the classifier sees the real provider error. + """ + err_chunk = SimpleNamespace( + model="test/model", + choices=None, + error_type="400 BadRequestError", + error_message="context length exceeded", + ) + agent.client.chat.completions.create.return_value = iter([err_chunk]) + agent.stream_delta_callback = MagicMock() + + with pytest.raises(Exception) as exc_info: + agent._interruptible_streaming_api_call({"messages": []}) + + exc = exc_info.value + assert type(exc).__name__ == "ProviderStreamError" + assert getattr(exc, "status_code", None) == 400 + assert "context length exceeded" in str(exc) + agent.stream_delta_callback.assert_not_called() + + def test_choiceless_usage_only_chunk_still_skipped(self, agent): + """Usage-only final chunks (choices empty, no error fields) keep flowing.""" + usage = SimpleNamespace(prompt_tokens=1, completion_tokens=2, total_tokens=3) + chunks = [ + _make_chunk(content="Hi"), + _make_chunk(finish_reason="stop"), + SimpleNamespace(model="test/model", choices=[], usage=usage), + ] + agent.client.chat.completions.create.return_value = iter(chunks) + agent.stream_delta_callback = MagicMock() + + resp = agent._interruptible_streaming_api_call({"messages": []}) + + assert resp.choices[0].message.content == "Hi" + assert resp.choices[0].finish_reason == "stop" + def test_named_non_json_sse_error_preserves_provider_message(self, agent): """SDK-level plain-text SSE errors retain their actionable message.""" import httpx From acaafcc6bba3f6fa1af0702ce986e0c4f79ed0c6 Mon Sep 17 00:00:00 2001 From: Evgenii <413011+smwbev@users.noreply.github.com> Date: Sun, 9 Aug 2026 06:30:50 +0000 Subject: [PATCH 074/376] fix(cron): make immediate execution race-safe - claim_job_for_fire returns the atomically claimed snapshot with a unique fire owner; heartbeat_fire_claim renews the lease; mark_job_run fences terminal writes by expected_fire_owner so a stale worker cannot record over a replacement claim. - run_one_job heartbeats the fire claim and forwards a combined cancel event (ownership loss OR external cancel) into run_job; the agent path is interrupted cooperatively and script-based jobs (no_agent + pre-run scripts) are hard-stopped with a process-tree kill (POSIX killpg SIGTERM then SIGKILL for surviving group members; Windows taskkill /T /F), with a bounded pipe drain so a SIGTERM-ignoring descendant cannot wedge the worker on communicate() EOF. - Shutdown interruption is scoped to the exact execution token instead of the bare job ID, so a replacement run of the same job never consumes a stale interrupted flag. - fire_claim_fence serializes save/deliver side effects per profile+job with a cross-process flock; remove_job prunes the fence-lock entry. - Preserves upstream BaseException terminal recording (#73973), completed one-shot retention (#80624), blocked_config preflight (T1-26), and the advance_next_runs batch on top of current main. --- cron/executions.py | 10 +- cron/jobs.py | 218 +++++- cron/scheduler.py | 673 ++++++++++++++++-- cron/scheduler_provider.py | 177 ++++- gateway/run.py | 14 +- tests/cron/test_claim_job_for_fire.py | 157 ++++ tests/cron/test_cron_no_agent.py | 33 +- tests/cron/test_cron_script.py | 54 +- tests/cron/test_cron_workdir.py | 47 ++ tests/cron/test_execution_ledger.py | 20 +- tests/cron/test_parallel_pool.py | 136 +++- .../cron/test_recurring_eagain_redispatch.py | 26 +- tests/cron/test_run_one_job.py | 12 +- tests/cron/test_scheduler.py | 453 +++++++++++- tests/cron/test_scheduler_provider.py | 238 ++++++- tests/cron/test_script_claim_heartbeat.py | 415 +++++++++++ tests/cron/test_sessiondb_init_hang.py | 2 +- tests/cron/test_shutdown_interrupt.py | 386 +++++++++- 18 files changed, 2937 insertions(+), 134 deletions(-) diff --git a/cron/executions.py b/cron/executions.py index 01437ab9847b0..40a9780700a8d 100644 --- a/cron/executions.py +++ b/cron/executions.py @@ -17,7 +17,10 @@ from hermes_constants import get_hermes_home from hermes_time import now as _hermes_now -EXECUTIONS_FILE = get_hermes_home().resolve() / "cron" / "executions.db" +# Optional test override. Production resolves the path at transaction time so +# dashboard operations that temporarily enter another profile cannot leak that +# profile's execution records into the import-time home. +EXECUTIONS_FILE: Optional[Path] = None MAX_TERMINAL_EXECUTIONS = 1000 _TERMINAL_STATES = ("completed", "failed", "unknown") _lock = threading.RLock() @@ -25,8 +28,9 @@ def _connect() -> sqlite3.Connection: - EXECUTIONS_FILE.parent.mkdir(parents=True, exist_ok=True) - return sqlite3.connect(EXECUTIONS_FILE, timeout=5) + path = EXECUTIONS_FILE or (get_hermes_home().resolve() / "cron" / "executions.db") + path.parent.mkdir(parents=True, exist_ok=True) + return sqlite3.connect(path, timeout=5) def _initialize_schema(conn: sqlite3.Connection) -> None: diff --git a/cron/jobs.py b/cron/jobs.py index d225fe4ec7c1e..766209148d4c1 100644 --- a/cron/jobs.py +++ b/cron/jobs.py @@ -103,6 +103,8 @@ def _ensure_croniter() -> bool: # concurrent mark_job_run / advance_next_run calls can clobber each other. _jobs_file_lock = threading.RLock() _jobs_lock_state = threading.local() +_fire_fence_locks: Dict[str, threading.RLock] = {} +_fire_fence_locks_guard = threading.Lock() # Upper bound on waiting for the cross-process .jobs.lock flock (#60703). # Every cron function in the process funnels through _jobs_lock(), and the @@ -371,6 +373,86 @@ def _jobs_lock(): _jobs_lock_state.depth = 0 _jobs_lock_state.load_stamp = None + +@contextlib.contextmanager +def _fire_job_lock(job_id: str): + """Serialize one job's owner mutations and external side effects. + + Unlike the global jobs lock, this lock may be held across network delivery. + It is scoped to one profile + job, so unrelated cron jobs keep progressing. + Fencing fails closed when cross-process locking is unavailable. + """ + cron_dir = _current_cron_store().cron_dir + lock_key = f"{cron_dir.resolve()}::{job_id}" + with _fire_fence_locks_guard: + local_lock = _fire_fence_locks.setdefault(lock_key, threading.RLock()) + + with local_lock: + ensure_dirs() + lock_name = uuid.uuid5(uuid.NAMESPACE_URL, lock_key).hex + lock_path = cron_dir / f".fire-{lock_name}.lock" + lock_fd = None + acquired = False + try: + lock_fd = open(lock_path, "a+", encoding="utf-8") + lock_fd.seek(0) + if fcntl is not None: + deadline = time.monotonic() + _JOBS_LOCK_TIMEOUT_SECONDS + while True: + try: + fcntl.flock(lock_fd, fcntl.LOCK_EX | fcntl.LOCK_NB) + acquired = True + break + except (OSError, IOError): + if time.monotonic() >= deadline: + logger.error( + "Timed out waiting for fire fence %s; failing closed", + lock_path, + ) + break + time.sleep(0.1) + elif msvcrt is not None: + getattr(msvcrt, "locking")( + lock_fd.fileno(), getattr(msvcrt, "LK_LOCK"), 1 + ) + acquired = True + else: # pragma: no cover - supported platforms provide one backend + logger.error("No cross-process lock backend for cron fire fence") + except (OSError, IOError) as exc: + logger.error("Cron fire fence unavailable for %s: %s", job_id, exc) + + try: + yield acquired + finally: + if lock_fd is not None: + try: + if acquired and fcntl is not None: + fcntl.flock(lock_fd, fcntl.LOCK_UN) + elif acquired and msvcrt is not None: + getattr(msvcrt, "locking")( + lock_fd.fileno(), getattr(msvcrt, "LK_UNLCK"), 1 + ) + except (OSError, IOError): + pass + finally: + lock_fd.close() + + +@contextlib.contextmanager +def fire_claim_fence(job_id: str, *, expected_owner: str): + """Hold a per-job fence while an owner performs an external side effect.""" + with _fire_job_lock(job_id) as acquired: + if not acquired: + yield False + return + with _jobs_lock(): + job = next((item for item in load_jobs() if item.get("id") == job_id), None) + claim = job.get("fire_claim") if isinstance(job, dict) else None + owns_claim = ( + isinstance(claim, dict) and claim.get("by") == expected_owner + ) + yield owns_claim + # Fields on a cron job that must never change after creation. ``id`` is used # as a filesystem path component under ``OUTPUT_DIR``; allowing it to be # updated lets an unsafe value (``../escape``, absolute path, nested) leak @@ -2069,10 +2151,37 @@ def remove_job(job_id: str) -> bool: "Failed to clear notepad for removed job %s", canonical_id, exc_info=True, ) + # Prune the per-job fire-fence lock entry so the registry does + # not grow monotonically across create/remove cycles. + _fence_key = f"{_current_cron_store().cron_dir.resolve()}::{canonical_id}" + with _fire_fence_locks_guard: + _fire_fence_locks.pop(_fence_key, None) return True return False +def mark_job_run( + job_id: str, + success: bool, + error: Optional[str] = None, + delivery_error: Optional[str] = None, + status: Optional[str] = None, + *, + expected_fire_owner: Optional[str] = None, +) -> bool: + with _fire_job_lock(job_id) as acquired: + if not acquired: + return False + return _mark_job_run_locked( + job_id, + success, + error, + delivery_error, + status=status, + expected_fire_owner=expected_fire_owner, + ) + + def _set_alert_flag(job_id: str, field: str, value: bool) -> bool: """Set/clear a persisted alert-dedup marker; return the PRIOR value. @@ -2124,9 +2233,15 @@ def clear_drift_alerted(job_id: str) -> None: _set_alert_flag(job_id, "drift_alerted", False) -def mark_job_run(job_id: str, success: bool, error: Optional[str] = None, - delivery_error: Optional[str] = None, - status: Optional[str] = None): +def _mark_job_run_locked( + job_id: str, + success: bool, + error: Optional[str] = None, + delivery_error: Optional[str] = None, + *, + status: Optional[str] = None, + expected_fire_owner: Optional[str] = None, +) -> bool: """ Mark a job as having been run. @@ -2146,6 +2261,15 @@ def mark_job_run(job_id: str, success: bool, error: Optional[str] = None, jobs = load_jobs() for i, job in enumerate(jobs): if job["id"] == job_id: + if expected_fire_owner is not None: + claim = job.get("fire_claim") + if not isinstance(claim, dict) or claim.get("by") != expected_fire_owner: + logger.warning( + "mark_job_run: job_id %s fire claim owner changed; " + "discarding stale completion", + job_id, + ) + return False now = _hermes_now().isoformat() job["last_run_at"] = now job["last_status"] = status or ("ok" if success else "error") @@ -2205,7 +2329,7 @@ def mark_job_run(job_id: str, success: bool, error: Optional[str] = None, job["state"] = "completed" job["next_run_at"] = None save_jobs(jobs) - return + return True # Compute next run job["next_run_at"] = compute_next_run(job["schedule"], now) @@ -2240,9 +2364,10 @@ def mark_job_run(job_id: str, success: bool, error: Optional[str] = None, job["state"] = "scheduled" save_jobs(jobs) - return + return True logger.warning("mark_job_run: job_id %s not found, skipping save", job_id) + return False def _write_wedged_oneshot_diagnostic(job: Dict[str, Any]) -> None: @@ -2474,7 +2599,31 @@ def _machine_id() -> str: return f"{host}:{os.getpid()}" -def claim_job_for_fire(job_id: str, *, claim_ttl_seconds: int = 300) -> bool: +def claim_job_for_fire( + job_id: str, + *, + claim_ttl_seconds: int = 300, + force: bool = False, + return_job: bool = False, +) -> Union[bool, Dict[str, Any]]: + with _fire_job_lock(job_id) as acquired: + if not acquired: + return False + return _claim_job_for_fire_locked( + job_id, + claim_ttl_seconds=claim_ttl_seconds, + force=force, + return_job=return_job, + ) + + +def _claim_job_for_fire_locked( + job_id: str, + *, + claim_ttl_seconds: int = 300, + force: bool = False, + return_job: bool = False, +) -> Union[bool, Dict[str, Any]]: """Atomically claim a job for a single external 'fire' (multi-machine at-most-once). Returns True iff THIS caller won the claim. @@ -2482,7 +2631,10 @@ def claim_job_for_fire(job_id: str, *, claim_ttl_seconds: int = 300) -> bool: external scheduler (Chronos) signals a job is due across N gateway replicas: exactly one wins. Single-machine deployments always win. - Under the file lock: reject if the job is missing/disabled/paused. If a + Under the file lock: reject if the job is missing/disabled/paused. An + explicit manual fire may pass ``force=True`` to atomically enable and + resume the job as part of the claim; external scheduler callbacks must + leave it false so a stale callback cannot resurrect a paused job. If a fresh claim (younger than ``claim_ttl_seconds``) already exists, lose. Otherwise stamp a ``fire_claim`` and, for recurring jobs, advance ``next_run_at`` (mirrors ``advance_next_run``'s at-most-once bump so a stale @@ -2501,8 +2653,10 @@ def claim_job_for_fire(job_id: str, *, claim_ttl_seconds: int = 300) -> bool: if job["id"] != job_id: continue # enabled + pause markers must both clear — a half-paused record - # (enabled=true, state=paused/paused_at set) must not claim. - if not is_job_runnable(job): + # (enabled=true, state=paused/paused_at set) must not claim. An + # explicit ``force`` (Trigger-now on a paused job) bypasses the + # gate and atomically resumes the job below. + if not force and not is_job_runnable(job): return False now = _hermes_now() existing = job.get("fire_claim") @@ -2520,14 +2674,23 @@ def claim_job_for_fire(job_id: str, *, claim_ttl_seconds: int = 300) -> bool: return False # someone holds a fresh claim except Exception: pass # malformed claim → overwrite - job["fire_claim"] = {"at": now.isoformat(), "by": _machine_id()} + if force: + job["enabled"] = True + job["state"] = "scheduled" + job["paused_at"] = None + job["paused_reason"] = None + # Per-acquisition token: a process may legitimately reclaim its own + # stale lease, and the previous runner must not heartbeat the new + # claim merely because hostname + PID are unchanged. + owner = f"{_machine_id()}:{uuid.uuid4().hex}" + job["fire_claim"] = {"at": now.isoformat(), "by": owner} kind = job.get("schedule", {}).get("kind") if kind in {"cron", "interval"}: nxt = compute_next_run(job["schedule"], now.isoformat()) if nxt: job["next_run_at"] = nxt save_jobs(jobs) - return True + return copy.deepcopy(job) if return_job else True return False @@ -2617,6 +2780,39 @@ def _sweep_completed_oneshots( return removed +def heartbeat_fire_claim(job_id: str, *, expected_owner: str) -> bool: + with _fire_job_lock(job_id) as acquired: + if not acquired: + return False + return _heartbeat_fire_claim_locked( + job_id, + expected_owner=expected_owner, + ) + + +def _heartbeat_fire_claim_locked(job_id: str, *, expected_owner: str) -> bool: + """Refresh an active ``fire_claim`` without extending another owner's lease. + + A cron execution can legitimately outlive the fire-claim TTL. The shared + run wrapper calls this periodically so another scheduler process cannot + treat a live execution as abandoned and dispatch it again. Comparing the + owner copied at dispatch prevents a stale runner from refreshing a claim + that has since been recovered by another process. + """ + with _jobs_lock(): + jobs = load_jobs() + for job in jobs: + if job.get("id") != job_id: + continue + claim = job.get("fire_claim") + if not isinstance(claim, dict) or claim.get("by") != expected_owner: + return False + claim["at"] = _hermes_now().isoformat() + save_jobs(jobs) + return True + return False + + def get_due_jobs() -> List[Dict[str, Any]]: """Get all jobs that are due to run now. diff --git a/cron/scheduler.py b/cron/scheduler.py index 345a098b96d8a..bc3f5e7182814 100644 --- a/cron/scheduler.py +++ b/cron/scheduler.py @@ -11,12 +11,14 @@ import asyncio import atexit import concurrent.futures +import contextlib import contextvars import json import logging import os import re import shutil +import signal import subprocess import sys import threading @@ -34,7 +36,7 @@ except ImportError: msvcrt = None from pathlib import Path -from typing import Any, List, Optional +from typing import Any, List, Optional, Protocol # Add parent directory to path for imports BEFORE repo-level imports. # Without this, standalone invocations (e.g. after `hermes update` reloads @@ -427,7 +429,18 @@ def _resolve_cron_enabled_toolsets(job: dict, cfg: dict) -> list[str] | None: "QQBOT_HOME_CHANNEL": "QQ_HOME_CHANNEL", } -from cron.jobs import get_due_jobs, mark_job_run, save_job_output, advance_next_runs, claim_dispatch, heartbeat_run_claim +from cron.jobs import ( + advance_next_runs, + claim_dispatch, + claim_job_for_fire, + fire_claim_fence, + get_due_jobs, + heartbeat_fire_claim, + heartbeat_run_claim, + mark_job_run, + save_job_output, + use_cron_store, +) from cron.executions import create_execution, finish_execution, mark_execution_running # Sentinel: when a cron agent has nothing new to report, it can start its @@ -471,6 +484,7 @@ def _is_cron_silence_response(text: str) -> bool: _parallel_pool: Optional[concurrent.futures.ThreadPoolExecutor] = None _parallel_pool_max_workers: Optional[int] = None _running_job_ids: set = set() +_running_fire_owners: dict[str, dict[object, tuple[Optional[str], Path]]] = {} _running_lock = threading.Lock() # Wall-clock (time.time()) instant each in-flight job id was claimed by @@ -507,16 +521,48 @@ def _is_cron_silence_response(text: str) -> bool: _INFLIGHT_MIN_ALLOWANCE_MINUTES = 30.0 -# Job IDs the gateway shutdown path force-killed the tool subprocess of -# while still in ``_running_job_ids`` (see ``mark_running_jobs_interrupted`` -# below). ``run_one_job``'s own completion path checks this set before -# writing its own ``last_status`` so a cron agent thread that keeps running -# in-process after its tool was killed out from under it — and produces a -# plausible-looking final response from truncated output — can never -# overwrite the interrupted status with a false "ok" (#60432). +# Execution tokens (``object()`` identity keys from ``_running_fire_owners``) +# of runs the shutdown path force-interrupted — see +# ``mark_running_jobs_interrupted`` below. ``run_one_job``'s own completion +# path checks its OWN token before writing ``last_status`` so a cron agent +# thread that keeps running in-process after its tool was killed out from +# under it — and produces a plausible-looking final response from truncated +# output — can never overwrite the interrupted status with a false "ok" +# (#60432). Token keying keeps an interruption scoped to that exact +# execution: a later run of the same job ID (recurring jobs reuse the ID +# every fire) must not inherit the stale flag. Legacy dispatch paths without +# a registered fire owner fall back to storing the bare job ID. _interrupted_job_ids: set = set() +class _CancelEventLike(Protocol): + """Structural type for cancellation sources (``threading.Event`` and + ``_CombinedCancelEvent`` both satisfy it).""" + + def is_set(self) -> bool: ... + def set(self) -> None: ... + + +class _CombinedCancelEvent: + """Duck-typed ``threading.Event`` that ORs several cancellation sources. + + ``run_one_job`` already derives a ``lost_ownership`` event from the + fire-claim heartbeat; transports (dashboard webhook drain, API server + shutdown) contribute their own per-task event. The worker only ever + calls ``is_set()`` / ``set()``, so a tiny wrapper beats a pump thread. + """ + + def __init__(self, *events: Optional["_CancelEventLike"]) -> None: + self._events = [event for event in events if event is not None] + + def is_set(self) -> bool: + return any(event.is_set() for event in self._events) + + def set(self) -> None: + for event in self._events: + event.set() + + def get_running_job_ids() -> "frozenset[str]": """Thread-safe snapshot of cron job IDs currently executing. @@ -533,7 +579,7 @@ def get_running_job_ids() -> "frozenset[str]": blind to them (#60432). """ with _running_lock: - return frozenset(_running_job_ids) + return frozenset(_running_job_ids | _running_fire_owners.keys()) def try_register_running_job(job_id: str) -> bool: @@ -829,7 +875,11 @@ def sweep_stale_inflight(due_jobs: Optional[list] = None) -> list: return [s[0] for s in stale] -def mark_running_jobs_interrupted(reason: str) -> list: +def mark_running_jobs_interrupted( + reason: str, + *, + only_owners: Optional[set] = None, +) -> list: """Best-effort: mark every currently in-flight cron job interrupted. Called by the gateway shutdown path immediately after it force-kills @@ -851,24 +901,62 @@ def mark_running_jobs_interrupted(reason: str) -> list: every entry in ``_running_agents`` on a drain timeout without per-agent correlation either. + ``only_owners``: optional set of ``(job_id, fire_owner)`` pairs. When + given (dashboard webhook drain), ONLY those exact executions are + marked — unrelated runs sharing the process (e.g. the desktop ticker's + own jobs) are left untouched. Interruption flags are recorded per + execution token, so a later run of the same job ID never consumes a + stale flag that targeted its dead predecessor. + Returns the list of job IDs marked, for the caller to log. """ with _running_lock: - job_ids = list(_running_job_ids) - _interrupted_job_ids.update(job_ids) + active_fires = [ + (token, job_id, owner, profile_home) + for job_id, executions in _running_fire_owners.items() + for token, (owner, profile_home) in executions.items() + ] + if only_owners is not None: + active_fires = [ + fire for fire in active_fires + if (fire[1], fire[2]) in only_owners + ] + registered_ids = {job_id for _t, job_id, _o, _p in active_fires} + if only_owners is None: + active_fires.extend( + (None, job_id, None, _get_hermes_home()) + for job_id in _running_job_ids - registered_ids + ) + _interrupted_job_ids.update( + token if token is not None else job_id + for token, job_id, _owner, _profile_home in active_fires + ) marked = [] - for job_id in job_ids: + for _token, job_id, fire_owner, profile_home in active_fires: + if not fire_owner: + logger.warning( + "Job '%s' interrupted before its durable fire owner was registered; " + "leaving persisted state untouched", + job_id, + ) + continue try: - mark_job_run(job_id, False, reason) - marked.append(job_id) + with use_cron_store(profile_home): + if mark_job_run( + job_id, + False, + reason, + expected_fire_owner=fire_owner, + ): + marked.append(job_id) except Exception as e: logger.warning("Failed to mark job %s interrupted: %s", job_id, e) return marked -def _is_interrupted(job_id: str) -> bool: - """Non-destructive peek at whether the shutdown path has marked - ``job_id`` interrupted (see ``mark_running_jobs_interrupted``). +def _is_interrupted(job_id: str, token: Optional[object] = None) -> bool: + """Non-destructive peek at whether the shutdown path has marked THIS + execution interrupted (see ``mark_running_jobs_interrupted``). Called by ``run_one_job`` BEFORE it decides what to deliver — a job whose tool subprocess was killed mid-flight may still produce a @@ -876,24 +964,35 @@ def _is_interrupted(job_id: str) -> bool: that must not go out to the user as if it were a normal result. Unlike ``_consume_interrupted_flag`` below, this does not clear the flag: the later, authoritative check (right before ``last_status`` is - written) still needs to see it.""" + written) still needs to see it. ``token`` scopes the check to one + exact execution: owner-registered runs are matched by token, so a + fresh run reusing the same job ID is not poisoned by a flag that + targeted its dead predecessor. The bare job ID is only ever stored + for legacy dispatch paths with no registered fire owner. + """ with _running_lock: + if token is not None and token in _interrupted_job_ids: + return True return job_id in _interrupted_job_ids -def _consume_interrupted_flag(job_id: str) -> bool: +def _consume_interrupted_flag(job_id: str, token: Optional[object] = None) -> bool: """Return True and clear the flag if the shutdown path already marked - ``job_id`` interrupted (see ``mark_running_jobs_interrupted``). + THIS execution interrupted (see ``mark_running_jobs_interrupted``). Called by ``run_one_job`` right before it would otherwise write its own ``last_status``. Consuming (discarding) rather than just checking keeps the flag from leaking across a later, unrelated run of the same job ID (recurring jobs reuse their ID every fire).""" with _running_lock: + hit = False + if token is not None and token in _interrupted_job_ids: + _interrupted_job_ids.discard(token) + hit = True if job_id in _interrupted_job_ids: _interrupted_job_ids.discard(job_id) - return True - return False + hit = True + return hit # Sequential (env-mutating) cron jobs — workdir jobs that touch @@ -2808,6 +2907,7 @@ def _deliver_result(job: dict, content: str, adapters=None, loop=None) -> Option # Backward-compatible module override used by tests and emergency monkeypatches. _SCRIPT_TIMEOUT = _DEFAULT_SCRIPT_TIMEOUT _RUN_CLAIM_HEARTBEAT_SECONDS = 60.0 +_FIRE_CLAIM_HEARTBEAT_GRACE_SECONDS = _RUN_CLAIM_HEARTBEAT_SECONDS * 3 def _get_script_timeout() -> int: @@ -2901,9 +3001,91 @@ def _windows_cron_python_invocation(python_exe: str) -> tuple[str, dict[str, str return str(interpreter), env_overlay +def _terminate_cron_script_process(proc: subprocess.Popen) -> None: + """Best-effort hard stop of a cron script and every child it spawned.""" + if proc.poll() is not None: + return + if sys.platform == "win32": + try: + subprocess.run( + ["taskkill", "/PID", str(proc.pid), "/T", "/F"], + capture_output=True, + timeout=10, + creationflags=windows_hide_flags(), + check=False, + ) + except (OSError, subprocess.TimeoutExpired): + proc.kill() + else: + try: + process_group: Optional[int] = os.getpgid(proc.pid) + except (ProcessLookupError, OSError): + process_group = None + if process_group is not None: + try: + os.killpg(process_group, signal.SIGTERM) # windows-footgun: ok — POSIX-only branch (win32 handled above) + except (ProcessLookupError, PermissionError, OSError): + process_group = None + if process_group is not None: + try: + proc.wait(timeout=1.0) + except subprocess.TimeoutExpired: + pass + # Escalate whenever ANY group member survived the TERM: a + # TERM-ignoring descendant keeps the stdio pipe write ends + # open, and the caller's communicate() would then block on + # EOF forever. killpg(pgid, 0) probes group liveness. + try: + os.killpg(process_group, 0) # windows-footgun: ok — POSIX-only branch + except (ProcessLookupError, OSError): + process_group = None + if process_group is not None: + try: + os.killpg(process_group, getattr(signal, "SIGKILL", signal.SIGTERM)) + except (ProcessLookupError, PermissionError, OSError): + pass + try: + proc.wait(timeout=1.0) + except subprocess.TimeoutExpired: + proc.kill() + proc.wait(timeout=1.0) + + +def _drain_script_pipes(proc: subprocess.Popen) -> None: + """Reap a terminated script process without ever blocking indefinitely. + + A descendant that survived the tree kill can hold the pipe write ends + open, so a bare ``communicate()`` would wait for EOF forever. Bound the + drain, then abandon the pipes — the caller only needs the process reaped + and the worker thread unblocked, not the output. + """ + try: + proc.communicate(timeout=5.0) + return + except subprocess.TimeoutExpired: + pass + try: + proc.kill() + except OSError: + pass + for stream in (proc.stdout, proc.stderr): + try: + if stream is not None: + stream.close() + except OSError: + pass + try: + proc.wait(timeout=5.0) + except subprocess.TimeoutExpired: + # Truly wedged — leave the zombie to the OS reaper rather than + # blocking the cron worker thread forever. + pass + + def _run_job_script( script_path: str, workdir: Optional[str] = None, + cancel_event: Optional[_CancelEventLike] = None, ) -> tuple[bool, str]: """Execute a cron job's data-collection script and capture its output. @@ -3006,10 +3188,11 @@ def _run_job_script( try: from tools.environments.local import build_subprocess_env - popen_kwargs = {} + popen_kwargs: dict[str, Any] = {"start_new_session": True} if sys.platform == "win32": popen_kwargs = { - "creationflags": windows_hide_flags(), + "creationflags": windows_hide_flags() + | getattr(subprocess, "CREATE_NEW_PROCESS_GROUP", 0), "encoding": "utf-8", "errors": "replace", } @@ -3020,17 +3203,34 @@ def _run_job_script( # NEVER mutate the Python process cwd — that would leak into # concurrent gateway sessions (#69396). _script_cwd = workdir or str(path.parent) - result = subprocess.run( + proc = subprocess.Popen( argv, - capture_output=True, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, text=True, - timeout=script_timeout, cwd=_script_cwd, env=env, **popen_kwargs, ) - stdout = (result.stdout or "").strip() - stderr = (result.stderr or "").strip() + deadline = time.monotonic() + script_timeout + while True: + if cancel_event is not None and cancel_event.is_set(): + _terminate_cron_script_process(proc) + _drain_script_pipes(proc) + return False, "Script cancelled because cron fire ownership was lost" + remaining = deadline - time.monotonic() + if remaining <= 0: + _terminate_cron_script_process(proc) + _drain_script_pipes(proc) + return False, f"Script timed out after {script_timeout}s: {path}" + try: + stdout_raw, stderr_raw = proc.communicate(timeout=min(0.1, remaining)) + break + except subprocess.TimeoutExpired: + continue + + stdout = (stdout_raw or "").strip() + stderr = (stderr_raw or "").strip() # Redact secrets from both stdout and stderr before any return path. try: @@ -3042,8 +3242,8 @@ def _run_job_script( stdout = "[REDACTED - redaction failed]" stderr = "[REDACTED - redaction failed]" - if result.returncode != 0: - parts = [f"Script exited with code {result.returncode}"] + if proc.returncode != 0: + parts = [f"Script exited with code {proc.returncode}"] if stderr: parts.append(f"stderr:\n{stderr}") if stdout: @@ -3052,14 +3252,15 @@ def _run_job_script( return True, stdout - except subprocess.TimeoutExpired: - return False, f"Script timed out after {script_timeout}s: {path}" except Exception as exc: return False, f"Script execution failed: {exc}" def _run_job_script_with_claim_heartbeat( - job: dict, script_path: str, workdir: Optional[str] = None, + job: dict, + script_path: str, + workdir: Optional[str] = None, + cancel_event: Optional[_CancelEventLike] = None, ) -> tuple[bool, str]: """Run a cron script while keeping its owned one-shot claim fresh. @@ -3081,7 +3282,7 @@ def _run_job_script_with_claim_heartbeat( and schedule.get("kind") == "once" and owner ): - return _run_job_script(script_path, workdir=workdir) + return _run_job_script(script_path, workdir=workdir, cancel_event=cancel_event) job_id = str(job.get("id") or "") stop = threading.Event() @@ -3112,10 +3313,10 @@ def _heartbeat_loop() -> None: job_id, exc_info=True, ) - return _run_job_script(script_path, workdir=workdir) + return _run_job_script(script_path, workdir=workdir, cancel_event=cancel_event) try: - return _run_job_script(script_path, workdir=workdir) + return _run_job_script(script_path, workdir=workdir, cancel_event=cancel_event) finally: stop.set() # Event.wait() wakes immediately. Keep completion bounded if the @@ -3726,8 +3927,11 @@ def _preflight_job_config(job: dict, cfg: dict) -> Optional[str]: def run_job( - job: dict, *, defer_agent_teardown: Optional[list] = None, + job: dict, + *, + defer_agent_teardown: Optional[list] = None, extra_prompt: Optional[str] = None, + cancel_event: Optional[_CancelEventLike] = None, ) -> tuple[bool, str, str, Optional[str]]: """ Execute a single cron job. @@ -3807,7 +4011,7 @@ def run_job( try: ok, output = _run_job_script_with_claim_heartbeat( - job, script_path, workdir=_job_workdir, + job, script_path, workdir=_job_workdir, cancel_event=cancel_event, ) except Exception as exc: logger.exception( @@ -4009,7 +4213,9 @@ def run_job( prerun_script = None script_path = job.get("script") if script_path: - prerun_script = _run_job_script_with_claim_heartbeat(job, script_path) + prerun_script = _run_job_script_with_claim_heartbeat( + job, script_path, cancel_event=cancel_event, + ) _ran_ok, _script_output = prerun_script if _ran_ok and not _parse_wake_gate(_script_output): logger.info( @@ -4725,6 +4931,15 @@ def run_job( ) _last_claim_heartbeat = time.monotonic() + def _abort_if_fire_claim_lost() -> None: + if cancel_event is None or not cancel_event.is_set(): + return + if agent is not None and hasattr(agent, "interrupt"): + agent.interrupt("Cron fire claim ownership was lost") + raise RuntimeError( + f"Cron job '{job_name}' lost its durable fire claim ownership" + ) + def _heartbeat_run_claim_if_due(): nonlocal _last_claim_heartbeat if not _is_oneshot or not _run_claim_owner: @@ -4754,15 +4969,17 @@ def _heartbeat_run_claim_if_due(): if _cron_inactivity_limit is None: # Unlimited — no inactivity watchdog, but a one-shot still # needs its run_claim heartbeat, so poll instead of blocking. - if _is_oneshot: + if _is_oneshot or cancel_event is not None: result = None while True: done, _ = concurrent.futures.wait( {_cron_future}, timeout=_POLL_INTERVAL, ) if done: + _abort_if_fire_claim_lost() result = _cron_future.result() break + _abort_if_fire_claim_lost() _heartbeat_run_claim_if_due() else: result = _cron_future.result() @@ -4773,8 +4990,10 @@ def _heartbeat_run_claim_if_due(): {_cron_future}, timeout=_POLL_INTERVAL, ) if done: + _abort_if_fire_claim_lost() result = _cron_future.result() break + _abort_if_fire_claim_lost() _heartbeat_run_claim_if_due() # Agent still running — check inactivity. _idle_secs = 0.0 @@ -5120,9 +5339,117 @@ def _teardown_cron_agent(agent, job_id: str) -> None: logger.debug("Job '%s': failed to reap stale auxiliary clients: %s", job_id, e) +def _run_with_fire_claim_heartbeat(job: dict, run) -> bool: + """Run ``run`` while keeping this job's owned durable fire claim fresh.""" + claim = job.get("fire_claim") + owner = str(claim.get("by") or "") if isinstance(claim, dict) else "" + if not owner: + return run(None) + + job_id = str(job.get("id") or "") + stop = threading.Event() + lost_ownership = threading.Event() + heartbeat_context = contextvars.copy_context() + + def _finish_unstarted(error: str) -> None: + execution_id = job.get("execution_id") + if not execution_id: + return + try: + finish_execution(execution_id, success=False, error=error) + except Exception: + logger.warning( + "Job '%s': failed to close unstarted execution ledger row", + job_id, + exc_info=True, + ) + + try: + owns_fire_claim = heartbeat_fire_claim(job_id, expected_owner=owner) + except Exception: + logger.warning( + "Job '%s': initial fire_claim validation failed", + job_id, + exc_info=True, + ) + _finish_unstarted( + "Fire claim ownership could not be validated before execution started." + ) + return True + + if owns_fire_claim is False: + logger.warning( + "Job '%s': fire claim ownership was already lost before execution", + job_id, + ) + _finish_unstarted("Fire claim ownership lost before execution started.") + return True + + def _heartbeat_loop() -> None: + last_confirmed = time.monotonic() + while not stop.wait(_RUN_CLAIM_HEARTBEAT_SECONDS): + try: + if not heartbeat_fire_claim(job_id, expected_owner=owner): + lost_ownership.set() + logger.warning( + "Job '%s': fire claim ownership lost; interrupting stale run", + job_id, + ) + return + last_confirmed = time.monotonic() + except Exception: + logger.debug( + "Job '%s': fire_claim heartbeat failed", + job_id, + exc_info=True, + ) + if ( + time.monotonic() - last_confirmed + >= _FIRE_CLAIM_HEARTBEAT_GRACE_SECONDS + ): + lost_ownership.set() + logger.warning( + "Job '%s': fire_claim could not be renewed within %.1fs; " + "interrupting uncertain run", + job_id, + _FIRE_CLAIM_HEARTBEAT_GRACE_SECONDS, + ) + return + + heartbeat_thread = threading.Thread( + target=heartbeat_context.run, + args=(_heartbeat_loop,), + name="cron-fire-claim-heartbeat", + daemon=True, + ) + try: + heartbeat_thread.start() + except Exception: + logger.warning( + "Job '%s': could not start fire_claim heartbeat", + job_id, + exc_info=True, + ) + _finish_unstarted( + "Fire claim heartbeat could not be started; execution was not run." + ) + return True + + try: + return run(lost_ownership) + finally: + stop.set() + heartbeat_thread.join(timeout=1.0) + + def run_one_job( - job: dict, *, adapters=None, loop=None, verbose: bool = False, + job: dict, + *, + adapters=None, + loop=None, + verbose: bool = False, extra_prompt: Optional[str] = None, + cancel_event: Optional[_CancelEventLike] = None, ) -> bool: """Run ONE due job end-to-end: execute → save output → deliver → mark. @@ -5130,14 +5457,94 @@ def run_one_job( that BOTH the built-in ticker and an external provider's ``fire_due`` (e.g. Chronos) run the identical sequence — no duplicated correctness. - It does NOT decide whether the job is due, claim it, or compute the next - run — those are the caller's concern (``tick`` advances ``next_run_at`` - under the file lock before dispatch; an external provider claims via the - store CAS). This function only fires the given job once. + It does NOT decide whether the job is due or acquire the initial claim — + both the ticker and external providers use the same store CAS before + calling it. It does keep an acquired claim alive for the full execution. Returns True if the job was processed (even if the job itself failed — failure is recorded via ``mark_job_run``), False only if processing raised. + + ``cancel_event``: optional transport-level cancellation source (dashboard + webhook drain, API server shutdown). It is OR-combined with the internal + fire-claim heartbeat's lost-ownership event, so either trigger stops the + run cooperatively — agent interruption AND script process-tree kill — + through the single fenced completion path. """ + claim = job.get("fire_claim") + fire_owner = str(claim.get("by") or "") if isinstance(claim, dict) else "" + execution_token = object() + profile_home = _get_hermes_home().resolve() + with _running_lock: + _running_fire_owners.setdefault(job["id"], {})[execution_token] = ( + fire_owner or None, + profile_home, + ) + try: + return _run_with_fire_claim_heartbeat( + job, + lambda lost_ownership: _run_one_job_body( + job, + adapters=adapters, + loop=loop, + verbose=verbose, + extra_prompt=extra_prompt, + fire_claim_lost=( + _CombinedCancelEvent(lost_ownership, cancel_event) + if cancel_event is not None + else lost_ownership + ), + execution_token=execution_token, + ), + ) + finally: + with _running_lock: + executions = _running_fire_owners.get(job["id"]) + if executions is not None: + executions.pop(execution_token, None) + if not executions: + _running_fire_owners.pop(job["id"], None) + + +def _run_one_job_body( + job: dict, + *, + adapters=None, + loop=None, + verbose: bool = False, + extra_prompt: Optional[str] = None, + fire_claim_lost: Optional[_CancelEventLike] = None, + execution_token: Optional[object] = None, +) -> bool: + claim = job.get("fire_claim") + fire_owner = str(claim.get("by") or "") if isinstance(claim, dict) else None + + class _FireClaimLostDuringSideEffect(Exception): + pass + + def _side_effect_fence(): + if fire_owner is None: + return contextlib.nullcontext(True) + return fire_claim_fence(job["id"], expected_owner=fire_owner) + + def _fire_claim_ownership_lost() -> bool: + if fire_claim_lost is not None and fire_claim_lost.is_set(): + return True + if fire_owner is None: + return False + try: + if heartbeat_fire_claim(job["id"], expected_owner=fire_owner): + return False + except Exception: + logger.debug( + "Job '%s': fire_claim ownership validation failed", + job["id"], + exc_info=True, + ) + return False + if fire_claim_lost is not None: + fire_claim_lost.set() + return True + execution_id = job.get("execution_id") if not execution_id: execution_id = create_execution(job["id"], source="direct")["id"] @@ -5190,10 +5597,19 @@ def run_one_job( # interpreter-shutdown guard in _deliver_result. _deferred_agents: list = [] try: - success, output, final_response, error = run_job( - job, defer_agent_teardown=_deferred_agents, - extra_prompt=extra_prompt, - ) + if fire_claim_lost is None: + success, output, final_response, error = run_job( + job, + defer_agent_teardown=_deferred_agents, + extra_prompt=extra_prompt, + ) + else: + success, output, final_response, error = run_job( + job, + defer_agent_teardown=_deferred_agents, + extra_prompt=extra_prompt, + cancel_event=fire_claim_lost, + ) except BaseException: # run_job's finally still hands back the agent when it raises; tear # it down here so a failed run never leaks its async resources @@ -5206,6 +5622,37 @@ def run_one_job( finally: reset_secret_scope(_scope_token) + if _fire_claim_ownership_lost(): + for _deferred_agent in _deferred_agents: + _teardown_cron_agent(_deferred_agent, job["id"]) + # Distinguish a real ownership loss (TTL expiry / replacement + # claim) from a transport-level cancel (dashboard drain): in the + # latter case WE still own the claim, and silently discarding + # would leave fire_claim lingering until TTL and last_status + # stale. Probe ownership once; if still ours, record the + # interruption through the owner-fenced terminal write. + if fire_owner is not None and heartbeat_fire_claim( + job["id"], expected_owner=fire_owner, + ): + mark_job_run( + job["id"], + False, + "Interrupted by shutdown before terminal completion.", + expected_fire_owner=fire_owner, + ) + finish_execution( + execution_id, + success=False, + error="Interrupted by shutdown before terminal completion.", + ) + else: + finish_execution( + execution_id, + success=False, + error="Fire claim ownership lost; stale result was discarded.", + ) + return True + # Everything from here through delivery runs with the agent still live # (deferred teardown). Wrap it ALL in a try/finally so that if any step # between run_job returning and delivery — save_job_output, the [SILENT] @@ -5214,8 +5661,12 @@ def run_one_job( # swallow the error and leak the agent's subprocesses/clients (#10200). delivery_error = None blocked_config = False + side_effect_ownership_lost = False try: - output_file = save_job_output(job["id"], output) + with _side_effect_fence() as owns_output: + if not owns_output: + raise _FireClaimLostDuringSideEffect + output_file = save_job_output(job["id"], output) if verbose: logger.info("Output saved to: %s", output_file) @@ -5226,7 +5677,7 @@ def run_one_job( # "this run was interrupted" summary instead of that response. # Peek-only: the flag stays set for the authoritative check # right before mark_job_run below. - if success and _is_interrupted(job["id"]): + if success and _is_interrupted(job["id"], execution_token): success = False error = ( "Interrupted by gateway shutdown before the run finished " @@ -5303,16 +5754,35 @@ def run_one_job( logger.info("Job '%s': agent returned %s — skipping delivery", job["id"], SILENT_MARKER) should_deliver = False + if should_deliver and _fire_claim_ownership_lost(): + should_deliver = False + logger.warning( + "Job '%s': skipping delivery after fire claim ownership loss", + job["id"], + ) + if should_deliver: unresolved_origin = ( _normalize_deliver_value(job.get("deliver", "local")) == "origin" and not _resolve_delivery_targets(job) ) try: - delivery_error = _deliver_result(job, deliver_content, adapters=adapters, loop=loop) + with _side_effect_fence() as owns_delivery: + if not owns_delivery: + raise _FireClaimLostDuringSideEffect + delivery_error = _deliver_result( + job, + deliver_content, + adapters=adapters, + loop=loop, + ) except Exception as de: + if isinstance(de, _FireClaimLostDuringSideEffect): + raise delivery_error = str(de) logger.error("Delivery failed for job %s: %s", job["id"], de) + except _FireClaimLostDuringSideEffect: + side_effect_ownership_lost = True finally: # Tear down the deferred agent(s) now that save + delivery have run # (or raised). Must happen on every path so cron agents never leak @@ -5320,6 +5790,32 @@ def run_one_job( for _deferred_agent in _deferred_agents: _teardown_cron_agent(_deferred_agent, job["id"]) + if side_effect_ownership_lost or _fire_claim_ownership_lost(): + # Same transport-cancel distinction as the pre-side-effect path: + # if WE still own the claim, record the interruption instead of + # discarding silently (lingering claim + stale last_status). + if fire_owner is not None and heartbeat_fire_claim( + job["id"], expected_owner=fire_owner, + ): + mark_job_run( + job["id"], + False, + "Interrupted by shutdown before terminal completion.", + expected_fire_owner=fire_owner, + ) + finish_execution( + execution_id, + success=False, + error="Interrupted by shutdown before terminal completion.", + ) + else: + finish_execution( + execution_id, + success=False, + error="Fire claim ownership lost; stale result was discarded.", + ) + return True + # Treat empty final_response as a soft failure so last_status # is not "ok" — the agent ran but produced nothing useful. # (issue #8585) @@ -5327,14 +5823,28 @@ def run_one_job( success = False error = "Agent completed but produced empty response (model error, timeout, or misconfiguration)" - if not _consume_interrupted_flag(job["id"]): - if blocked_config: - mark_job_run( - job["id"], success, error, delivery_error=delivery_error, - status="blocked_config", - ) - else: - mark_job_run(job["id"], success, error, delivery_error=delivery_error) + interrupted = _consume_interrupted_flag(job["id"], execution_token) + if interrupted: + finish_execution( + execution_id, + success=False, + error="Interrupted by gateway shutdown before terminal completion.", + ) + return True + + mark_kwargs = {"delivery_error": delivery_error} + if fire_owner is not None: + mark_kwargs["expected_fire_owner"] = fire_owner + if blocked_config: + mark_kwargs["status"] = "blocked_config" + marked = mark_job_run(job["id"], success, error, **mark_kwargs) + if fire_owner is not None and not marked: + finish_execution( + execution_id, + success=False, + error="Fire claim ownership lost before terminal completion.", + ) + return True normalized_deliver = _normalize_deliver_value(job.get("deliver", "local")) if delivery_error: delivery_outcome = "failed" @@ -5361,12 +5871,16 @@ def run_one_job( # is never written, so the job sits in state "scheduled" until the # run-claim TTL expires and the dispatch-limit guard removes it with # no output and no error. Record the failure first, then re-raise - # anything that isn't a plain Exception. + # anything that isn't a plain Exception. Owner fencing still applies: + # a stale worker must not record over a replacement claim owner. _err_text = str(e) or type(e).__name__ logger.error("Error processing job %s: %s", job['id'], _err_text) try: - if not _consume_interrupted_flag(job["id"]): - mark_job_run(job["id"], False, _err_text) + if not _consume_interrupted_flag(job["id"], execution_token): + mark_kwargs = {} + if fire_owner is not None: + mark_kwargs["expected_fire_owner"] = fire_owner + mark_job_run(job["id"], False, _err_text, **mark_kwargs) except Exception as record_err: # Never let bookkeeping mask the original interruption. logger.error( @@ -5555,6 +6069,10 @@ def tick( # bumping next_run_at forward so the grace window never expires. # mark_job_run() overwrites next_run_at on completion. # Batched: one load + one save for the whole due set, not one per job. + # Composes with the claim-time advance in claim_job_for_fire: for + # cron-kind jobs both compute the same next occurrence; interval jobs + # re-anchor from their own "now" at claim time (harmless for + # at-most-once — mark_job_run re-anchors at completion regardless). advance_next_runs([job["id"] for job in due_jobs]) # Resolve max parallel workers: env var > config.yaml > unbounded. @@ -5589,7 +6107,28 @@ def _process_job(job: dict) -> bool: module-level ``run_one_job`` so ``tick`` and external providers (Chronos ``fire_due``) use the identical execute→save→deliver→mark body.""" - return run_one_job(job, adapters=adapters, loop=loop, verbose=verbose) + # Acquire the durable claim only when this worker actually starts, + # not while it may wait behind other work in an executor queue. + # This prevents a queued lease from expiring before execution. + claimed = claim_job_for_fire(job["id"], return_job=True) + if not claimed: + finish_execution( + job["execution_id"], + success=False, + error="Fire claim lost; execution was not started.", + ) + return True + # Production CAS returns the exact persisted record with its unique + # owner. Bool fallback keeps older test doubles/API overrides + # compatible; real callers using return_job=True never take it. + claimed_job = dict(claimed) if isinstance(claimed, dict) else dict(job) + claimed_job["execution_id"] = job["execution_id"] + return run_one_job( + claimed_job, + adapters=adapters, + loop=loop, + verbose=verbose, + ) # Partition due jobs: those with a per-job workdir mutate # os.environ["TERMINAL_CWD"] inside run_job, which is process-global, so diff --git a/cron/scheduler_provider.py b/cron/scheduler_provider.py index db3641a8c9633..e5923d5ed0387 100644 --- a/cron/scheduler_provider.py +++ b/cron/scheduler_provider.py @@ -19,6 +19,7 @@ """ from __future__ import annotations +import inspect import threading from abc import ABC, abstractmethod from typing import Any @@ -98,7 +99,23 @@ def recover_interrupted(self) -> int: return recover_interrupted_executions() - def fire_due(self, job_id: str, *, adapters: Any = None, loop: Any = None) -> bool: + @property + def supports_force_fire(self) -> bool: + """Whether ``fire_due`` accepts the additive ``force`` keyword. + + Signature detection keeps providers written before ``force`` was added + source-compatible. Providers accepting ``**kwargs`` are compatible. + """ + return provider_supports_force_fire(self) + + def fire_due( + self, + job_id: str, + *, + adapters: Any = None, + loop: Any = None, + force: bool = False, + ) -> bool: """Run a single job NOW via the shared orchestrator. Called by the inbound fire webhook when an external scheduler signals a job is due. @@ -107,20 +124,72 @@ def fire_due(self, job_id: str, *, adapters: Any = None, loop: Any = None) -> bo ``run_one_job`` body. Built-in never calls this (it has its own tick loop); an external provider routes its inbound fire here. - Returns True if THIS caller claimed and ran the job, False if the claim - was lost (another machine/retry won it) or the job no longer exists. + Returns True if THIS caller claimed and processed the attempt, even if + the job itself failed. Returns False only if the claim was lost + (another machine/retry won it) or the job no longer exists. + """ + claimed_job = self.claim_fire(job_id, force=force) + if claimed_job is None: + return False + return self.fire_claimed(claimed_job, adapters=adapters, loop=loop) + + def claim_fire(self, job_id: str, *, force: bool = False) -> dict | None: + """Durably claim one fire and create its audit attempt before dispatch. + + Webhook transports call this synchronously before acknowledging the + external scheduler, then pass the exact owner-bearing snapshot to + ``fire_claimed`` in tracked background work. + """ + from cron.executions import create_execution, finish_execution + from cron.jobs import claim_job_for_fire + + execution = create_execution(job_id, source=self.name) + claim_kwargs = {"return_job": True} + if force: + claim_kwargs["force"] = True + try: + claimed_job = claim_job_for_fire(job_id, **claim_kwargs) + except BaseException as exc: + finish_execution( + execution["id"], + success=False, + error=f"Fire claim failed before dispatch: {type(exc).__name__}: {exc}", + ) + raise + if not isinstance(claimed_job, dict): + finish_execution( + execution["id"], + success=False, + error="Fire claim was not acquired", + ) + return None + claimed_job["execution_id"] = execution["id"] + return claimed_job + + def fire_claimed( + self, + claimed_job: dict, + *, + adapters: Any = None, + loop: Any = None, + cancel_event: Any = None, + ) -> bool: + """Run an exact snapshot returned by ``claim_fire``. + + ``cancel_event``: optional transport-owned ``threading.Event`` (or + compatible) that lets the caller stop this execution cooperatively + — e.g. the dashboard lifespan drain signalling pending webhook + fires before the event loop shuts down. """ - from cron.jobs import claim_job_for_fire, get_job - from cron.executions import create_execution from cron.scheduler import run_one_job - if not claim_job_for_fire(job_id): - return False # another machine already claimed this fire - job = get_job(job_id) - if job is None: - return False # job removed (e.g. repeat-N exhausted) between arm and fire - job["execution_id"] = create_execution(job_id, source=self.name)["id"] - return run_one_job(job, adapters=adapters, loop=loop) + run_one_job( + claimed_job, + adapters=adapters, + loop=loop, + cancel_event=cancel_event, + ) + return True def reconcile(self) -> None: """Converge the external registry toward jobs.json (the desired state): @@ -129,6 +198,68 @@ def reconcile(self) -> None: return None +def provider_supports_force_fire(provider: Any) -> bool: + """Return whether a provider can safely receive ``fire_due(force=...)``.""" + try: + parameters = inspect.signature(provider.fire_due).parameters.values() + except (TypeError, ValueError): + return False + return any( + parameter.kind is inspect.Parameter.VAR_KEYWORD + or ( + parameter.name == "force" + and parameter.kind + in (inspect.Parameter.POSITIONAL_OR_KEYWORD, inspect.Parameter.KEYWORD_ONLY) + ) + for parameter in parameters + ) + + +def provider_supports_split_fire(provider: Any) -> bool: + """Return whether a provider implements the two-phase fire contract. + + The webhook admission path uses ``claim_fire`` + ``fire_claimed`` so the + 202 response is backed by a durable, owner-fenced claim. A legacy + third-party provider that overrides the documented single-phase + ``fire_due`` hook (custom claim/re-arm/telemetry behavior) but inherits + the base ``claim_fire`` must keep being driven through its own + ``fire_due`` — silently routing around its override would drop that + behavior. Providers that customize ``claim_fire`` itself are already + split-aware and keep the two-phase path. + """ + cls = type(provider) + fire_due_impl = getattr(cls, "fire_due", None) + claim_fire_impl = getattr(cls, "claim_fire", None) + fire_claimed_impl = getattr(cls, "fire_claimed", None) + if claim_fire_impl is not None and claim_fire_impl is not CronScheduler.claim_fire: + return True + # Overriding the second phase is also proof of split-awareness (the + # provider composes with the inherited claim path) — e.g. Chronos keeps + # its re-arm logic in ``fire_claimed`` only. + if fire_claimed_impl is not None and fire_claimed_impl is not CronScheduler.fire_claimed: + return True + if fire_due_impl is None or fire_due_impl is CronScheduler.fire_due: + return True + return False + + +def provider_supports_fire_cancel(provider: Any) -> bool: + """Return whether ``fire_claimed`` accepts a ``cancel_event`` kwarg.""" + try: + parameters = inspect.signature(provider.fire_claimed).parameters.values() + except (TypeError, ValueError): + return False + return any( + parameter.kind is inspect.Parameter.VAR_KEYWORD + or ( + parameter.name == "cancel_event" + and parameter.kind + in (inspect.Parameter.POSITIONAL_OR_KEYWORD, inspect.Parameter.KEYWORD_ONLY) + ) + for parameter in parameters + ) + + def resolve_cron_scheduler() -> "CronScheduler": """Return the active cron scheduler provider. @@ -169,6 +300,28 @@ def resolve_cron_scheduler() -> "CronScheduler": return InProcessCronScheduler() +def scheduler_for_profile_mode( + provider: "CronScheduler", *, multiplex_profiles: bool +) -> "CronScheduler": + """Return a scheduler that can safely serve the gateway's profile mode. + + External providers currently own one unscoped remote registry/client and + therefore cannot safely reconcile several profile stores from one process. + Fail closed to the built-in multiplex ticker until the provider API carries + explicit profile identity through lifecycle and webhook calls. + """ + if not multiplex_profiles or isinstance(provider, InProcessCronScheduler): + return provider + + import logging + + logging.getLogger("cron.scheduler_provider").warning( + "cron.provider '%s' does not support multiplex_profiles; using built-in ticker", + provider.name, + ) + return InProcessCronScheduler() + + class InProcessCronScheduler(CronScheduler): """Default provider: the historical in-process 60s ticker. diff --git a/gateway/run.py b/gateway/run.py index 1dd0e9e7c3465..081af9f49b3a7 100644 --- a/gateway/run.py +++ b/gateway/run.py @@ -29630,9 +29630,17 @@ def restart_signal_handler(): # historical in-process 60s ticker; an external provider (e.g. chronos) # may arm a schedule and return. Pass the event loop so cron delivery can # use live adapters (E2EE support). - from cron.scheduler_provider import InProcessCronScheduler, resolve_cron_scheduler + from cron.scheduler_provider import ( + InProcessCronScheduler, + resolve_cron_scheduler, + scheduler_for_profile_mode, + ) cron_stop = threading.Event() - cron_provider = resolve_cron_scheduler() + multiplex_cron = bool(getattr(runner.config, "multiplex_profiles", False)) + cron_provider = scheduler_for_profile_mode( + resolve_cron_scheduler(), + multiplex_profiles=multiplex_cron, + ) cron_start_kwargs: Dict[str, Any] = {"adapters": runner.adapters, "loop": asyncio.get_running_loop()} # Multiplex profiles: tell the built-in ticker which profile homes to @@ -29643,7 +29651,7 @@ def restart_signal_handler(): # never execute because no ticker owns that store. if ( isinstance(cron_provider, InProcessCronScheduler) - and getattr(runner.config, "multiplex_profiles", False) + and multiplex_cron ): try: profile_homes = _multiplex_profile_homes(runner.config) diff --git a/tests/cron/test_claim_job_for_fire.py b/tests/cron/test_claim_job_for_fire.py index 16c827972e4a6..fa0f7b39b3495 100644 --- a/tests/cron/test_claim_job_for_fire.py +++ b/tests/cron/test_claim_job_for_fire.py @@ -7,6 +7,9 @@ These exercise the real store against a temp HERMES_HOME (no mocks) per the E2E-over-mocks discipline for file-touching code. """ +import threading +import time + import pytest @@ -32,6 +35,47 @@ def test_claim_succeeds_once_then_blocks(temp_home): assert get_job(jid)["next_run_at"] != before +def test_claim_oneshot_cannot_be_double_claimed(temp_home): + """A one-shot can't be double-claimed (the fresh claim blocks the retry).""" + from cron.jobs import create_job, claim_job_for_fire + + job = create_job(prompt="x", schedule="30m", name="o") + assert claim_job_for_fire(job["id"]) is True + assert claim_job_for_fire(job["id"]) is False + + +def test_claim_unknown_job_returns_false(temp_home): + from cron.jobs import claim_job_for_fire + + assert claim_job_for_fire("nope-does-not-exist") is False + + +def test_claim_paused_job_returns_false(temp_home): + """A paused job can't be claimed.""" + from cron.jobs import create_job, claim_job_for_fire, pause_job + + job = create_job(prompt="x", schedule="every 5m", name="p") + pause_job(job["id"]) + assert claim_job_for_fire(job["id"]) is False + + +def test_forced_claim_atomically_resumes_paused_job(temp_home): + """Explicit manual fire may resume a paused job without exposing a due + intermediate state to the ticker.""" + from cron.jobs import create_job, claim_job_for_fire, get_job, pause_job + + job = create_job(prompt="x", schedule="every 5m", name="manual") + pause_job(job["id"]) + + assert claim_job_for_fire(job["id"], force=True) is True + claimed = get_job(job["id"]) + assert claimed["enabled"] is True + assert claimed["state"] == "scheduled" + assert claimed["paused_at"] is None + assert claimed["paused_reason"] is None + assert claimed["fire_claim"] is not None + + def test_stale_claim_is_reclaimable(temp_home, monkeypatch): """A claim older than the TTL is overwritten — the fire isn't stuck forever if the winning machine crashed before mark_job_run cleared the claim.""" @@ -58,3 +102,116 @@ def test_mark_job_run_clears_claim(temp_home): assert get_job(jid).get("fire_claim") is None # …and the re-armed recurring job is claimable again. assert claim_job_for_fire(jid) is True + + +def test_fire_claim_heartbeat_refreshes_only_expected_owner(temp_home, monkeypatch): + from datetime import datetime, timedelta + + import cron.jobs as jobs + + job = jobs.create_job(prompt="x", schedule="every 5m", name="heartbeat") + assert jobs.claim_job_for_fire(job["id"]) is True + claimed = jobs.get_job(job["id"])["fire_claim"] + claimed_at = datetime.fromisoformat(claimed["at"]) + monkeypatch.setattr( + jobs, + "_hermes_now", + lambda: claimed_at + timedelta(seconds=30), + ) + + assert jobs.heartbeat_fire_claim( + job["id"], + expected_owner=claimed["by"], + ) is True + refreshed = jobs.get_job(job["id"])["fire_claim"] + assert refreshed["at"] != claimed["at"] + assert refreshed["by"] == claimed["by"] + assert jobs.heartbeat_fire_claim( + job["id"], + expected_owner="replacement-owner", + ) is False + + +def test_reclaimed_fire_uses_new_owner_token(temp_home, monkeypatch): + from datetime import datetime, timedelta + + import cron.jobs as jobs + + job = jobs.create_job(prompt="x", schedule="every 5m", name="reclaim") + assert jobs.claim_job_for_fire(job["id"]) is True + original = dict(jobs.get_job(job["id"])["fire_claim"]) + original_at = datetime.fromisoformat(original["at"]) + monkeypatch.setattr( + jobs, + "_hermes_now", + lambda: original_at + timedelta(seconds=301), + ) + + assert jobs.claim_job_for_fire(job["id"]) is True + replacement = dict(jobs.get_job(job["id"])["fire_claim"]) + assert replacement["by"] != original["by"] + assert jobs.heartbeat_fire_claim( + job["id"], + expected_owner=original["by"], + ) is False + assert jobs.get_job(job["id"])["fire_claim"] == replacement + + +def test_stale_fire_owner_cannot_mark_replacement_run(temp_home): + import cron.jobs as jobs + + job = jobs.create_job(prompt="x", schedule="every 5m", name="fenced") + assert jobs.claim_job_for_fire(job["id"]) is True + original = dict(jobs.get_job(job["id"])["fire_claim"]) + records = jobs.load_jobs() + records[0]["fire_claim"] = {"at": original["at"], "by": "replacement"} + jobs.save_jobs(records) + + assert jobs.mark_job_run( + job["id"], + success=True, + expected_fire_owner=original["by"], + ) is False + persisted = jobs.get_job(job["id"]) + assert persisted["fire_claim"]["by"] == "replacement" + assert persisted.get("last_run_at") is None + + +def test_fire_claim_fence_serializes_terminal_revocation(temp_home): + """A side effect authorized by owner linearizes before terminal revocation.""" + from cron.jobs import ( + claim_job_for_fire, + create_job, + fire_claim_fence, + mark_job_run, + ) + + job = create_job(prompt="x", schedule="every 5m", name="fenced-side-effect") + claimed = claim_job_for_fire(job["id"], return_job=True) + assert isinstance(claimed, dict) + owner = claimed["fire_claim"]["by"] + terminal_done = threading.Event() + + def finish_run(): + mark_job_run(job["id"], True, expected_fire_owner=owner) + terminal_done.set() + + with fire_claim_fence(job["id"], expected_owner=owner) as owns_claim: + assert owns_claim is True + thread = threading.Thread(target=finish_run) + thread.start() + time.sleep(0.05) + assert terminal_done.is_set() is False + + thread.join(timeout=1) + assert terminal_done.is_set() is True + + +def test_fire_claim_fence_rejects_stale_owner(temp_home): + from cron.jobs import claim_job_for_fire, create_job, fire_claim_fence + + job = create_job(prompt="x", schedule="every 5m", name="stale-fence") + claim_job_for_fire(job["id"]) + + with fire_claim_fence(job["id"], expected_owner="stale") as owns_claim: + assert owns_claim is False diff --git a/tests/cron/test_cron_no_agent.py b/tests/cron/test_cron_no_agent.py index 6378e4bde6252..52f51ba303855 100644 --- a/tests/cron/test_cron_no_agent.py +++ b/tests/cron/test_cron_no_agent.py @@ -158,10 +158,37 @@ def test_timed_out_no_agent_script_delivery_is_not_mislabeled_as_provider_failur ) delivered = [] - def _timeout(*_args, **kwargs): - raise subprocess.TimeoutExpired(cmd="slow.py", timeout=kwargs["timeout"]) + # The script runner uses Popen + a polling loop (cancel/timeout aware), + # so simulate a process that never finishes: communicate() always times + # out and the script deadline is shrunk to keep the test fast. + class _NeverFinishes: + returncode = None + pid = 0 + stdout = None + stderr = None - monkeypatch.setattr(scheduler.subprocess, "run", _timeout) + def __init__(self, *_args, **_kwargs): + pass + + def poll(self): + return None + + def communicate(self, timeout=None): + raise subprocess.TimeoutExpired(cmd="slow.py", timeout=timeout) + + def wait(self, timeout=None): + raise subprocess.TimeoutExpired(cmd="slow.py", timeout=timeout) + + def kill(self): + self.returncode = -9 + + monkeypatch.setattr(scheduler.subprocess, "Popen", _NeverFinishes) + monkeypatch.setattr(scheduler, "_get_script_timeout", lambda: 1) + monkeypatch.setattr( + scheduler, + "_terminate_cron_script_process", + lambda proc: setattr(proc, "returncode", -15), + ) monkeypatch.setattr( scheduler, "_deliver_result", diff --git a/tests/cron/test_cron_script.py b/tests/cron/test_cron_script.py index 32642e0341b60..311140c3e9541 100644 --- a/tests/cron/test_cron_script.py +++ b/tests/cron/test_cron_script.py @@ -156,20 +156,39 @@ def test_windows_uv_venv_python_script_bypasses_launcher(self, cron_env, tmp_pat captured = {} - def fake_run(argv, **kwargs): - captured["argv"] = argv - captured["kwargs"] = kwargs - return SimpleNamespace(returncode=0, stdout="ok\n", stderr="") + class FakeProc: + def __init__(self, argv, **kwargs): + captured["argv"] = argv + captured["kwargs"] = kwargs + self.returncode = 0 + + def poll(self): + return self.returncode + + def communicate(self, timeout=None): + return ("ok\n", "") + + def wait(self, timeout=None): + return self.returncode + + fake_run = FakeProc monkeypatch.setattr(sched_mod.sys, "executable", str(venv_python)) - monkeypatch.setattr(sched_mod.subprocess, "run", fake_run) + monkeypatch.setattr(sched_mod, "windows_hide_flags", lambda: 0x08000000) + monkeypatch.setattr(sched_mod.subprocess, "Popen", fake_run) success, output = _run_job_script("probe.py") assert success is True assert output == "ok" assert captured["argv"] == [str(base_python), str(script.resolve())] - assert captured["kwargs"]["creationflags"] == sched_mod.windows_hide_flags() + # The script runner always adds CREATE_NEW_PROCESS_GROUP on win32 so a + # cancel can taskkill the whole tree; on POSIX the getattr default is + # 0 and the flag set is exactly windows_hide_flags(). + expected_flags = sched_mod.windows_hide_flags() | getattr( + sched_mod.subprocess, "CREATE_NEW_PROCESS_GROUP", 0 + ) + assert captured["kwargs"]["creationflags"] == expected_flags env = captured["kwargs"]["env"] assert env["VIRTUAL_ENV"] == str(venv) assert str(site_packages) in env["PYTHONPATH"] @@ -185,12 +204,25 @@ def test_non_windows_script_preserves_default_text_decoding(self, cron_env, monk captured = {} - def fake_run(argv, **kwargs): - captured["argv"] = argv - captured["kwargs"] = kwargs - return SimpleNamespace(returncode=0, stdout="ok\n", stderr="") + class FakeProc: + def __init__(self, argv, **kwargs): + captured["argv"] = argv + captured["kwargs"] = kwargs + self.returncode = 0 + + def poll(self): + return self.returncode + + def communicate(self, timeout=None): + return ("ok\n", "") + + def wait(self, timeout=None): + return self.returncode + + fake_run = FakeProc - monkeypatch.setattr(sched_mod.subprocess, "run", fake_run) + monkeypatch.setattr(sched_mod.sys, "platform", "linux") + monkeypatch.setattr(sched_mod.subprocess, "Popen", fake_run) success, output = _run_job_script("probe.py") diff --git a/tests/cron/test_cron_workdir.py b/tests/cron/test_cron_workdir.py index be2c2aced20aa..284675de89212 100644 --- a/tests/cron/test_cron_workdir.py +++ b/tests/cron/test_cron_workdir.py @@ -145,6 +145,53 @@ class TestTickWorkdirPartition: pieces tick() calls. """ + def test_workdir_jobs_run_sequentially(self, tmp_path, monkeypatch): + import cron.scheduler as sched + + # Two workdir jobs (both sequential) + one parallel job. + workdir_a = {"id": "a", "name": "A", "workdir": str(tmp_path)} + workdir_b = {"id": "b", "name": "B", "workdir": str(tmp_path)} + parallel_job = {"id": "c", "name": "C", "workdir": None} + + monkeypatch.setattr(sched, "get_due_jobs", lambda: [workdir_a, workdir_b, parallel_job]) + monkeypatch.setattr(sched, "claim_job_for_fire", lambda *_a, **_kw: True) + + # Record call order / thread context. + import threading + calls: list[tuple[str, str]] = [] + order_lock = threading.Lock() + + def fake_run_job(job, *, defer_agent_teardown=None, **_kw): + # Return a minimal tuple matching run_job's signature. + with order_lock: + calls.append((job["id"], threading.current_thread().name)) + return True, "output", "response", None + + monkeypatch.setattr(sched, "run_job", fake_run_job) + monkeypatch.setattr(sched, "save_job_output", lambda _jid, _o: None) + monkeypatch.setattr(sched, "mark_job_run", lambda *_a, **_kw: None) + monkeypatch.setattr( + sched, "_deliver_result", lambda *_a, **_kw: None + ) + + n = sched.tick(verbose=False) + assert n == 3 + + ids = [c[0] for c in calls] + # Sequential workdir jobs preserve submission order relative to each + # other (single-thread pool). + assert ids.index("a") < ids.index("b") + + # Workdir jobs run on the persistent single-thread cron-seq pool — + # NOT the main thread — so a long workdir job never blocks the ticker. + main_thread_name = threading.current_thread().name + for jid in ("a", "b"): + workdir_thread_name = next(t for j, t in calls if j == jid) + assert workdir_thread_name != main_thread_name + assert workdir_thread_name.startswith("cron-seq"), workdir_thread_name + par_thread_name = next(t for j, t in calls if j == "c") + assert par_thread_name.startswith("cron-parallel"), par_thread_name + # --------------------------------------------------------------------------- # scheduler.run_job: TERMINAL_CWD + skip_context_files wiring diff --git a/tests/cron/test_execution_ledger.py b/tests/cron/test_execution_ledger.py index 5f268c64192a2..5c02b6eea8eab 100644 --- a/tests/cron/test_execution_ledger.py +++ b/tests/cron/test_execution_ledger.py @@ -39,6 +39,24 @@ def test_execution_transitions_are_durable(monkeypatch, tmp_path): assert persisted == [completed] +def test_execution_ledger_follows_the_current_profile_home(monkeypatch, tmp_path): + import cron.executions as executions + + current_home = {"path": tmp_path / "default"} + monkeypatch.setattr(executions, "EXECUTIONS_FILE", None) + monkeypatch.setattr(executions, "get_hermes_home", lambda: current_home["path"]) + + default_row = executions.create_execution("default-job", source="builtin") + current_home["path"] = tmp_path / "worker" + worker_row = executions.create_execution("worker-job", source="builtin") + + assert executions.list_executions() == [worker_row] + current_home["path"] = tmp_path / "default" + assert executions.list_executions() == [default_row] + assert (tmp_path / "default" / "cron" / "executions.db").is_file() + assert (tmp_path / "worker" / "cron" / "executions.db").is_file() + + def test_terminal_execution_cannot_be_rewritten(monkeypatch, tmp_path): executions = _point_ledger(monkeypatch, tmp_path) record = executions.create_execution("immutable", source="builtin") @@ -182,7 +200,7 @@ def submit(self, _callable): lambda execution_id, **kwargs: finished.append((execution_id, kwargs)), ) monkeypatch.setattr(scheduler, "get_due_jobs", lambda: [{"id": "submit-fail"}]) - monkeypatch.setattr(scheduler, "advance_next_runs", lambda _ids: 0) + monkeypatch.setattr(scheduler, "claim_job_for_fire", lambda _job_id: True) monkeypatch.setattr(scheduler, "_get_parallel_pool", lambda _workers: BrokenPool()) assert scheduler.tick(verbose=False, sync=False) == 0 diff --git a/tests/cron/test_parallel_pool.py b/tests/cron/test_parallel_pool.py index 4eca57463eab2..41d3cff547e94 100644 --- a/tests/cron/test_parallel_pool.py +++ b/tests/cron/test_parallel_pool.py @@ -71,7 +71,7 @@ def test_running_set_prevents_double_dispatch(self, tmp_path, monkeypatch): dispatched = [] monkeypatch.setattr(sched, "get_due_jobs", lambda: [job]) - monkeypatch.setattr(sched, "advance_next_runs", lambda *_a, **_kw: 0) + monkeypatch.setattr(sched, "claim_job_for_fire", lambda *_a, **_kw: True) monkeypatch.setattr(sched, "run_job", lambda j, **_kw: dispatched.append(j["id"]) or (True, "out", "resp", None)) monkeypatch.setattr(sched, "save_job_output", lambda *_a, **_kw: None) monkeypatch.setattr(sched, "mark_job_run", lambda *_a, **_kw: None) @@ -85,6 +85,56 @@ def test_running_set_prevents_double_dispatch(self, tmp_path, monkeypatch): sched._shutdown_parallel_pool() + def test_fire_claim_is_acquired_only_when_executor_worker_starts(self, monkeypatch): + """Queue wait must not consume the durable claim TTL.""" + import cron.scheduler as sched + + sched._running_job_ids.clear() + job = { + "id": "queued-job", + "name": "queued", + "prompt": "test", + "schedule": "every 5m", + "enabled": True, + "next_run_at": "2020-01-01T00:00:00", + "deliver": "local", + } + submitted = [] + claim_calls = [] + + class DeferredPool: + def submit(self, callback): + future = concurrent.futures.Future() + submitted.append((callback, future)) + return future + + monkeypatch.setattr(sched, "get_due_jobs", lambda: [job]) + monkeypatch.setattr(sched, "_get_parallel_pool", lambda _workers: DeferredPool()) + monkeypatch.setattr( + sched, + "create_execution", + lambda *_a, **_kw: {"id": "execution-1"}, + ) + monkeypatch.setattr( + sched, + "claim_job_for_fire", + lambda job_id, **kwargs: claim_calls.append((job_id, kwargs)) + or {**job, "fire_claim": {"by": "worker-owner", "at": "now"}}, + ) + monkeypatch.setattr(sched, "run_one_job", lambda *_a, **_kw: True) + + assert sched.tick(verbose=False, sync=False) == 1 + assert claim_calls == [] + assert len(submitted) == 1 + + callback, future = submitted[0] + result = callback() + future.set_result(result) + + assert claim_calls == [("queued-job", {"return_job": True})] + assert "queued-job" not in sched._running_job_ids + + class TestSyncMode: """tick() blocks by default (sync=True); tick(sync=False) returns immediately.""" @@ -104,7 +154,7 @@ def test_sync_true_blocks_and_returns_correct_count(self, tmp_path, monkeypatch) ] monkeypatch.setattr(sched, "get_due_jobs", lambda: jobs) - monkeypatch.setattr(sched, "advance_next_runs", lambda *_a, **_kw: 0) + monkeypatch.setattr(sched, "claim_job_for_fire", lambda *_a, **_kw: True) monkeypatch.setattr(sched, "run_job", lambda j, **_kw: (True, "out", "resp", None)) monkeypatch.setattr(sched, "save_job_output", lambda *_a, **_kw: "/tmp/out") monkeypatch.setattr(sched, "mark_job_run", lambda *_a, **_kw: None) @@ -115,6 +165,49 @@ def test_sync_true_blocks_and_returns_correct_count(self, tmp_path, monkeypatch) sched._shutdown_parallel_pool() + def test_sync_false_returns_immediately(self, tmp_path, monkeypatch): + """sync=False returns before parallel jobs finish (optimistic count).""" + import cron.scheduler as sched + + sched._parallel_pool = None + sched._parallel_pool_max_workers = None + sched._running_job_ids.clear() + + job = { + "id": "slow-job", + "name": "slow", + "prompt": "test", + "schedule": "every 5m", + "enabled": True, + "next_run_at": "2020-01-01T00:00:00", + "deliver": "local", + } + + barrier = threading.Barrier(2, timeout=5) + + def slow_run(j, *, defer_agent_teardown=None, **_kw): + barrier.wait() # blocks until test thread also waits + return True, "out", "resp", None + + monkeypatch.setattr(sched, "get_due_jobs", lambda: [job]) + monkeypatch.setattr(sched, "claim_job_for_fire", lambda *_a, **_kw: True) + monkeypatch.setattr(sched, "run_job", slow_run) + monkeypatch.setattr(sched, "save_job_output", lambda *_a, **_kw: "/tmp/out") + monkeypatch.setattr(sched, "mark_job_run", lambda *_a, **_kw: None) + monkeypatch.setattr(sched, "_deliver_result", lambda *_a, **_kw: None) + + start = time.monotonic() + n = sched.tick(verbose=False, sync=False) # opt-in: non-blocking + elapsed = time.monotonic() - start + + assert n == 1 # optimistic count + assert elapsed < 1.0 # returned immediately, didn't wait for slow_run + + # Let the job finish so cleanup works. + barrier.wait() + time.sleep(0.1) + sched._shutdown_parallel_pool() + class TestSequentialPool: """Sequential (workdir) jobs use the persistent cron-seq pool. @@ -151,7 +244,7 @@ def slow_run(j, *, defer_agent_teardown=None, **_kw): return True, "out", "resp", None monkeypatch.setattr(sched, "get_due_jobs", lambda: [job]) - monkeypatch.setattr(sched, "advance_next_runs", lambda *_a, **_kw: 0) + monkeypatch.setattr(sched, "claim_job_for_fire", lambda *_a, **_kw: True) monkeypatch.setattr(sched, "run_job", slow_run) monkeypatch.setattr(sched, "save_job_output", lambda *_a, **_kw: "/tmp/out") monkeypatch.setattr(sched, "mark_job_run", lambda *_a, **_kw: None) @@ -168,6 +261,43 @@ def slow_run(j, *, defer_agent_teardown=None, **_kw): time.sleep(0.1) sched._shutdown_parallel_pool() + def test_sequential_running_guard_prevents_double_dispatch(self, tmp_path, monkeypatch): + """A workdir job already in _running_job_ids is skipped on next tick.""" + import cron.scheduler as sched + + sched._parallel_pool = None + sched._parallel_pool_max_workers = None + sched._sequential_pool = None + sched._running_job_ids.clear() + + job = { + "id": "guard-seq", + "name": "guard-seq", + "prompt": "test", + "schedule": "every 5m", + "enabled": True, + "next_run_at": "2020-01-01T00:00:00", + "deliver": "local", + "workdir": str(tmp_path), + } + + # Simulate the job already running. + sched._running_job_ids.add("guard-seq") + + dispatched = [] + monkeypatch.setattr(sched, "get_due_jobs", lambda: [job]) + monkeypatch.setattr(sched, "claim_job_for_fire", lambda *_a, **_kw: True) + monkeypatch.setattr(sched, "run_job", lambda j, **_kw: dispatched.append(j["id"]) or (True, "out", "resp", None)) + monkeypatch.setattr(sched, "save_job_output", lambda *_a, **_kw: None) + monkeypatch.setattr(sched, "mark_job_run", lambda *_a, **_kw: None) + monkeypatch.setattr(sched, "_deliver_result", lambda *_a, **_kw: None) + + n = sched.tick(verbose=False) + assert n == 0 # skipped, not dispatched + assert dispatched == [] + + sched._running_job_ids.discard("guard-seq") + sched._shutdown_parallel_pool() def test_get_sequential_pool_is_persistent(self): """_get_sequential_pool returns the same single-thread pool.""" diff --git a/tests/cron/test_recurring_eagain_redispatch.py b/tests/cron/test_recurring_eagain_redispatch.py index 74a28eb32ffa0..5e9d499541f67 100644 --- a/tests/cron/test_recurring_eagain_redispatch.py +++ b/tests/cron/test_recurring_eagain_redispatch.py @@ -68,18 +68,34 @@ def wedge_env(tmp_path, monkeypatch): class TestEAGAINRecurringRedispatches: def _make_script_eagain(self, env, monkeypatch): - """Make the next subprocess.run raise EAGAIN once, then pass.""" + """Make the next subprocess.Popen raise EAGAIN once, then pass. + + The script runner spawns via Popen (polling loop for cancel/timeout), + so the substrate-failure injection point is the Popen constructor. + """ import cron.scheduler as sched_mod - real_run = sched_mod.subprocess.run state = {"n": 0} - def fake_run(argv, **kwargs): + class _OkProc: + def __init__(self, argv, **kwargs): + self.returncode = 0 + + def poll(self): + return self.returncode + + def communicate(self, timeout=None): + return ("ok\n", "") + + def wait(self, timeout=None): + return 0 + + def fake_popen(argv, **kwargs): state["n"] += 1 if state["n"] == 1: raise OSError(11, "Resource temporarily unavailable") - return subprocess.CompletedProcess(argv, 0, stdout="ok\n", stderr="") + return _OkProc(argv, **kwargs) - monkeypatch.setattr(sched_mod.subprocess, "run", fake_run) + monkeypatch.setattr(sched_mod.subprocess, "Popen", fake_popen) return state def test_eagain_then_redispatched_on_next_tick(self, wedge_env, monkeypatch, tmp_path): diff --git a/tests/cron/test_run_one_job.py b/tests/cron/test_run_one_job.py index a61866c54413d..462416fe1b385 100644 --- a/tests/cron/test_run_one_job.py +++ b/tests/cron/test_run_one_job.py @@ -46,7 +46,7 @@ def test_tick_process_job_sequence(monkeypatch): sequence run_job → save → deliver → mark, in that order.""" calls = _patch_pipeline(monkeypatch) monkeypatch.setattr(s, "get_due_jobs", lambda: [{"id": "j1", "name": "t"}]) - monkeypatch.setattr(s, "advance_next_runs", lambda ids: 1) + monkeypatch.setattr(s, "claim_job_for_fire", lambda _job_id, **_kwargs: True) s.tick(verbose=False, sync=True) @@ -54,6 +54,16 @@ def test_tick_process_job_sequence(monkeypatch): assert calls[-1] == ("mark", "j1", True) +def test_tick_skips_job_when_durable_fire_claim_is_lost(monkeypatch): + """A manual/external fire that wins the shared CAS must exclude ticker.""" + calls = _patch_pipeline(monkeypatch) + monkeypatch.setattr(s, "get_due_jobs", lambda: [{"id": "j1", "name": "t"}]) + monkeypatch.setattr(s, "claim_job_for_fire", lambda _job_id: False) + + assert s.tick(verbose=False, sync=True) == 0 + assert calls == [] + + def test_run_one_job_success_sequence(monkeypatch): """The extracted helper runs the same execute→save→deliver→mark sequence for a successful job.""" diff --git a/tests/cron/test_scheduler.py b/tests/cron/test_scheduler.py index b8e5cce652636..007da444a46ad 100644 --- a/tests/cron/test_scheduler.py +++ b/tests/cron/test_scheduler.py @@ -662,13 +662,345 @@ def test_tick_skips_due_jobs_while_dispatch_is_paused(self, tmp_path): "enabled": True, } with patch("cron.scheduler.get_due_jobs", return_value=[job]), patch( - "cron.scheduler.advance_next_runs" - ) as advance, patch("cron.scheduler.run_one_job") as run_one: + "cron.scheduler.claim_job_for_fire", return_value=True + ) as claim, patch("cron.scheduler.run_one_job") as run_one: assert tick(verbose=False, sync=True, can_dispatch=lambda: False) == 0 - advance.assert_not_called() + claim.assert_not_called() run_one.assert_not_called() + def test_tick_marks_empty_response_as_error(self, tmp_path): + """When run_job returns success=True but final_response is empty, + tick() should mark the job as error so last_status != 'ok'. + (issue #8585) + """ + from cron.scheduler import tick + + job = { + "id": "empty-job", + "name": "empty-test", + "prompt": "do something", + "schedule": "every 1h", + "enabled": True, + "next_run_at": "2020-01-01T00:00:00", + "deliver": "local", + "last_status": None, + } + + fake_db = MagicMock() + + with patch("cron.scheduler._hermes_home", tmp_path), \ + patch("cron.scheduler.get_due_jobs", return_value=[job]), \ + patch("cron.scheduler.claim_job_for_fire", return_value=True), \ + patch("cron.scheduler.mark_job_run") as mock_mark, \ + patch("cron.scheduler.save_job_output", return_value="/tmp/out.md"), \ + patch("cron.scheduler._resolve_origin", return_value=None), \ + patch("cron.scheduler.run_job", return_value=(True, "output", "", None)): + tick(verbose=False) + + # Should be called with success=False because final_response is empty + mock_mark.assert_called_once() + call_args = mock_mark.call_args + assert call_args[0][0] == "empty-job" + assert call_args[0][1] is False # success should be False + assert "empty" in call_args[0][2].lower() # error should mention empty + + def test_run_job_sets_auto_delivery_env_from_dotenv_home_channel(self, tmp_path, monkeypatch): + job = { + "id": "test-job", + "name": "test", + "prompt": "hello", + "deliver": "telegram", + } + fake_db = MagicMock() + seen = {} + + (tmp_path / ".env").write_text("TELEGRAM_HOME_CHANNEL=-2002\n") + monkeypatch.delenv("TELEGRAM_HOME_CHANNEL", raising=False) + monkeypatch.delenv("HERMES_CRON_AUTO_DELIVER_PLATFORM", raising=False) + monkeypatch.delenv("HERMES_CRON_AUTO_DELIVER_CHAT_ID", raising=False) + monkeypatch.delenv("HERMES_CRON_AUTO_DELIVER_THREAD_ID", raising=False) + + class FakeAgent: + def __init__(self, *args, **kwargs): + pass + + def run_conversation(self, *args, **kwargs): + from gateway.session_context import get_session_env + seen["platform"] = get_session_env("HERMES_CRON_AUTO_DELIVER_PLATFORM") or None + seen["chat_id"] = get_session_env("HERMES_CRON_AUTO_DELIVER_CHAT_ID") or None + seen["thread_id"] = get_session_env("HERMES_CRON_AUTO_DELIVER_THREAD_ID") or None + return {"final_response": "ok"} + + with patch("cron.scheduler._hermes_home", tmp_path), \ + patch("cron.scheduler._preflight_job_config", return_value=None), \ + patch("hermes_state.SessionDB", return_value=fake_db), \ + patch( + "hermes_cli.runtime_provider.resolve_runtime_provider", + return_value={ + "api_key": "***", + "base_url": "https://example.invalid/v1", + "provider": "openrouter", + "api_mode": "chat_completions", + }, + ), \ + patch("run_agent.AIAgent", FakeAgent): + success, output, final_response, error = run_job(job) + + assert success is True + assert error is None + assert final_response == "ok" + assert "ok" in output + assert seen == { + "platform": "telegram", + "chat_id": "-2002", + "thread_id": None, + } + assert os.getenv("HERMES_CRON_AUTO_DELIVER_PLATFORM") is None + assert os.getenv("HERMES_CRON_AUTO_DELIVER_CHAT_ID") is None + assert os.getenv("HERMES_CRON_AUTO_DELIVER_THREAD_ID") is None + fake_db.close.assert_called_once() + + def test_run_job_preserves_slack_origin_thread_for_same_explicit_channel(self, tmp_path, monkeypatch): + job = { + "id": "slack-thread-job", + "name": "slack-thread", + "prompt": "hello", + "deliver": "slack:C0B3KEP3SD6", + "origin": { + "platform": "slack", + "chat_id": "C0B3KEP3SD6", + "thread_id": "1778485067.844139", + }, + } + fake_db = MagicMock() + seen = {} + + monkeypatch.delenv("HERMES_CRON_AUTO_DELIVER_PLATFORM", raising=False) + monkeypatch.delenv("HERMES_CRON_AUTO_DELIVER_CHAT_ID", raising=False) + monkeypatch.delenv("HERMES_CRON_AUTO_DELIVER_THREAD_ID", raising=False) + + class FakeAgent: + def __init__(self, *args, **kwargs): + pass + + def run_conversation(self, *args, **kwargs): + from gateway.session_context import get_session_env + + seen["platform"] = get_session_env("HERMES_CRON_AUTO_DELIVER_PLATFORM") or None + seen["chat_id"] = get_session_env("HERMES_CRON_AUTO_DELIVER_CHAT_ID") or None + seen["thread_id"] = get_session_env("HERMES_CRON_AUTO_DELIVER_THREAD_ID") or None + return {"final_response": "ok"} + + with patch("cron.scheduler._hermes_home", tmp_path), \ + patch("cron.scheduler._preflight_job_config", return_value=None), \ + patch("hermes_state.SessionDB", return_value=fake_db), \ + patch( + "hermes_cli.runtime_provider.resolve_runtime_provider", + return_value={ + "api_key": "***", + "base_url": "https://example.invalid/v1", + "provider": "openrouter", + "api_mode": "chat_completions", + }, + ), \ + patch("run_agent.AIAgent", FakeAgent): + success, output, final_response, error = run_job(job) + + assert success is True + assert error is None + assert final_response == "ok" + assert "ok" in output + assert seen == { + "platform": "slack", + "chat_id": "C0B3KEP3SD6", + "thread_id": "1778485067.844139", + } + assert os.getenv("HERMES_CRON_AUTO_DELIVER_PLATFORM") is None + assert os.getenv("HERMES_CRON_AUTO_DELIVER_CHAT_ID") is None + assert os.getenv("HERMES_CRON_AUTO_DELIVER_THREAD_ID") is None + fake_db.close.assert_called_once() + + @pytest.mark.parametrize("timeout_value", ["600", "0"]) + def test_run_job_heartbeats_oneshot_claim_in_both_wait_modes( + self, tmp_path, monkeypatch, timeout_value + ): + """Timed and unlimited one-shot monitors both refresh their owned claim.""" + job = { + "id": "heartbeat-job", + "name": "heartbeat", + "prompt": "hello", + "schedule": {"kind": "once", "run_at": "2026-07-10T12:00:00Z"}, + "run_claim": {"at": "2026-07-10T12:00:00Z", "by": "owner-token"}, + } + fake_db = MagicMock() + + class FakeAgent: + def __init__(self, *args, **kwargs): + pass + + def run_conversation(self, *args, **kwargs): + return {"final_response": "ok"} + + class FakeFuture: + def result(self): + return {"final_response": "ok"} + + fake_future = FakeFuture() + fake_pool = MagicMock() + fake_pool.submit.return_value = fake_future + wait_results = [(set(), set()), ({fake_future}, set())] + monotonic_ticks = itertools.count(step=61.0) + monkeypatch.setenv("HERMES_CRON_TIMEOUT", timeout_value) + + with patch("cron.scheduler._hermes_home", tmp_path), \ + patch("cron.scheduler._preflight_job_config", return_value=None), \ + patch("hermes_state.SessionDB", return_value=fake_db), \ + patch( + "hermes_cli.runtime_provider.resolve_runtime_provider", + return_value={ + "api_key": "***", + "base_url": "https://example.invalid/v1", + "provider": "openrouter", + "api_mode": "chat_completions", + }, + ), \ + patch("run_agent.AIAgent", FakeAgent), \ + patch("cron.scheduler.concurrent.futures.ThreadPoolExecutor", return_value=fake_pool), \ + patch("cron.scheduler.concurrent.futures.wait", side_effect=wait_results), \ + patch("cron.scheduler.time.monotonic", side_effect=monotonic_ticks.__next__), \ + patch("cron.scheduler.heartbeat_run_claim", return_value=True) as heartbeat: + success, _output, final_response, error = run_job(job) + + assert success is True + assert error is None + assert final_response == "ok" + heartbeat.assert_called_once_with( + "heartbeat-job", expected_owner="owner-token" + ) + + def test_run_job_resets_secret_source_cache_before_reload(self, tmp_path, monkeypatch): + """Each run must clear the secret-source cache before re-reading the + env, so a long-running gateway re-resolves Bitwarden/BSM-backed secrets + instead of leaving the startup .env placeholder in place (#33465). + + A bare ``load_dotenv`` re-load can't do this: startup already recorded + this HERMES_HOME in ``_APPLIED_HOMES``, so the external-secret pull + no-ops and only the placeholder is re-applied. The scheduler must call + ``reset_secret_source_cache()`` (forcing the re-pull) and route through + ``load_hermes_dotenv`` (which then re-applies external secret sources). + """ + job = {"id": "bsm-job", "name": "bsm", "prompt": "hello"} + fake_db = MagicMock() + call_order = [] + + def _record_reset(): + call_order.append("reset") + + def _record_load(*args, **kwargs): + call_order.append("load") + return [] + + with patch("cron.scheduler._hermes_home", tmp_path), \ + patch("cron.scheduler._resolve_origin", return_value=None), \ + patch("hermes_cli.env_loader.reset_secret_source_cache", _record_reset), \ + patch("hermes_cli.env_loader.load_hermes_dotenv", _record_load), \ + patch("hermes_state.SessionDB", return_value=fake_db), \ + patch( + "hermes_cli.runtime_provider.resolve_runtime_provider", + return_value={ + "api_key": "***", + "base_url": "https://example.invalid/v1", + "provider": "openrouter", + "api_mode": "chat_completions", + }, + ), \ + patch("run_agent.AIAgent") as mock_agent_cls: + mock_agent = MagicMock() + mock_agent.run_conversation.return_value = {"final_response": "ok"} + mock_agent_cls.return_value = mock_agent + success, _output, _final, error = run_job(job) + + assert success is True + assert error is None + # reset MUST precede the reload, else _APPLIED_HOMES no-ops the re-pull. + assert call_order[:2] == ["reset", "load"], call_order + + def test_run_job_clears_stale_auto_delivery_thread_id_between_jobs(self, tmp_path, monkeypatch): + jobs = [ + { + "id": "threaded-job", + "name": "threaded", + "prompt": "hello", + "deliver": "telegram:-1001:42", + }, + { + "id": "threadless-job", + "name": "threadless", + "prompt": "hello again", + "deliver": "telegram:-2002", + }, + ] + fake_db = MagicMock() + seen = [] + + monkeypatch.delenv("HERMES_CRON_AUTO_DELIVER_PLATFORM", raising=False) + monkeypatch.delenv("HERMES_CRON_AUTO_DELIVER_CHAT_ID", raising=False) + monkeypatch.delenv("HERMES_CRON_AUTO_DELIVER_THREAD_ID", raising=False) + + class FakeAgent: + def __init__(self, *args, **kwargs): + pass + + def run_conversation(self, *args, **kwargs): + from gateway.session_context import get_session_env + + seen.append( + { + "platform": get_session_env("HERMES_CRON_AUTO_DELIVER_PLATFORM") or None, + "chat_id": get_session_env("HERMES_CRON_AUTO_DELIVER_CHAT_ID") or None, + "thread_id": get_session_env("HERMES_CRON_AUTO_DELIVER_THREAD_ID") or None, + } + ) + return {"final_response": "ok"} + + with patch("cron.scheduler._hermes_home", tmp_path), \ + patch("cron.scheduler._preflight_job_config", return_value=None), \ + patch("hermes_state.SessionDB", return_value=fake_db), \ + patch( + "hermes_cli.runtime_provider.resolve_runtime_provider", + return_value={ + "api_key": "***", + "base_url": "https://example.invalid/v1", + "provider": "openrouter", + "api_mode": "chat_completions", + }, + ), \ + patch("run_agent.AIAgent", FakeAgent): + for job in jobs: + success, output, final_response, error = run_job(job) + assert success is True + assert error is None + assert final_response == "ok" + assert "ok" in output + + assert seen == [ + { + "platform": "telegram", + "chat_id": "-1001", + "thread_id": "42", + }, + { + "platform": "telegram", + "chat_id": "-2002", + "thread_id": None, + }, + ] + assert os.getenv("HERMES_CRON_AUTO_DELIVER_PLATFORM") is None + assert os.getenv("HERMES_CRON_AUTO_DELIVER_CHAT_ID") is None + assert os.getenv("HERMES_CRON_AUTO_DELIVER_THREAD_ID") is None + assert fake_db.close.call_count == 2 + class TestRunJobConfigLogging: """Verify that config.yaml parse failures are logged, not silently swallowed.""" @@ -1028,6 +1360,7 @@ def _make_job(self): def test_silent_response_suppresses_delivery(self, caplog): with patch("cron.scheduler.get_due_jobs", return_value=[self._make_job()]), \ + patch("cron.scheduler.claim_job_for_fire", return_value=True), \ patch("cron.scheduler.run_job", return_value=(True, "# output", "[SILENT]", None)), \ patch("cron.scheduler.save_job_output", return_value="/tmp/out.md"), \ patch("cron.scheduler._deliver_result") as deliver_mock, \ @@ -1038,12 +1371,61 @@ def test_silent_response_suppresses_delivery(self, caplog): deliver_mock.assert_not_called() assert any(SILENT_MARKER in r.message for r in caplog.records) + def test_silent_with_note_suppresses_delivery(self): + with patch("cron.scheduler.get_due_jobs", return_value=[self._make_job()]), \ + patch("cron.scheduler.claim_job_for_fire", return_value=True), \ + patch("cron.scheduler.run_job", return_value=(True, "# output", "[SILENT] No changes detected", None)), \ + patch("cron.scheduler.save_job_output", return_value="/tmp/out.md"), \ + patch("cron.scheduler._deliver_result") as deliver_mock, \ + patch("cron.scheduler.mark_job_run"): + from cron.scheduler import tick + tick(verbose=False) + deliver_mock.assert_not_called() + + def test_silent_trailing_suppresses_delivery(self): + """Agent appended [SILENT] after explanation text — must still suppress.""" + response = "2 deals filtered out (like<10, reply<15).\n\n[SILENT]" + with patch("cron.scheduler.get_due_jobs", return_value=[self._make_job()]), \ + patch("cron.scheduler.claim_job_for_fire", return_value=True), \ + patch("cron.scheduler.run_job", return_value=(True, "# output", response, None)), \ + patch("cron.scheduler.save_job_output", return_value="/tmp/out.md"), \ + patch("cron.scheduler._deliver_result") as deliver_mock, \ + patch("cron.scheduler.mark_job_run"): + from cron.scheduler import tick + tick(verbose=False) + deliver_mock.assert_not_called() + + def test_silent_is_case_insensitive(self): + with patch("cron.scheduler.get_due_jobs", return_value=[self._make_job()]), \ + patch("cron.scheduler.claim_job_for_fire", return_value=True), \ + patch("cron.scheduler.run_job", return_value=(True, "# output", "[silent] nothing new", None)), \ + patch("cron.scheduler.save_job_output", return_value="/tmp/out.md"), \ + patch("cron.scheduler._deliver_result") as deliver_mock, \ + patch("cron.scheduler.mark_job_run"): + from cron.scheduler import tick + tick(verbose=False) + deliver_mock.assert_not_called() + + def test_bracketless_silent_variants_suppress(self): + """Bracketless near-markers the model emits when it drops brackets + must still suppress delivery (#51438, #46917).""" + from cron.scheduler import tick + for marker in ("SILENT", "NO_REPLY", "NO REPLY", "no_reply"): + with patch("cron.scheduler.get_due_jobs", return_value=[self._make_job()]), \ + patch("cron.scheduler.claim_job_for_fire", return_value=True), \ + patch("cron.scheduler.run_job", return_value=(True, "# output", marker, None)), \ + patch("cron.scheduler.save_job_output", return_value="/tmp/out.md"), \ + patch("cron.scheduler._deliver_result") as deliver_mock, \ + patch("cron.scheduler.mark_job_run"): + tick(verbose=False) + deliver_mock.assert_not_called() def test_report_quoting_marker_mid_sentence_still_delivers(self): """A genuine report that merely mentions the token mid-sentence must be delivered — the old substring check wrongly swallowed it.""" response = "I considered staying [SILENT] but here is the summary: 3 items merged." with patch("cron.scheduler.get_due_jobs", return_value=[self._make_job()]), \ + patch("cron.scheduler.claim_job_for_fire", return_value=True), \ patch("cron.scheduler.run_job", return_value=(True, "# output", response, None)), \ patch("cron.scheduler.save_job_output", return_value="/tmp/out.md"), \ patch("cron.scheduler._deliver_result") as deliver_mock, \ @@ -1056,6 +1438,7 @@ def test_report_quoting_marker_mid_sentence_still_delivers(self): def test_failed_job_always_delivers(self): """Failed jobs deliver regardless of [SILENT] in output.""" with patch("cron.scheduler.get_due_jobs", return_value=[self._make_job()]), \ + patch("cron.scheduler.claim_job_for_fire", return_value=True), \ patch("cron.scheduler.run_job", return_value=(False, "# output", "", "some error")), \ patch("cron.scheduler.save_job_output", return_value="/tmp/out.md"), \ patch("cron.scheduler._deliver_result") as deliver_mock, \ @@ -1064,10 +1447,23 @@ def test_failed_job_always_delivers(self): tick(verbose=False) deliver_mock.assert_called_once() + def test_output_saved_even_when_delivery_suppressed(self): + with patch("cron.scheduler.get_due_jobs", return_value=[self._make_job()]), \ + patch("cron.scheduler.claim_job_for_fire", return_value=True), \ + patch("cron.scheduler.run_job", return_value=(True, "# full output", "[SILENT]", None)), \ + patch("cron.scheduler.save_job_output") as save_mock, \ + patch("cron.scheduler._deliver_result") as deliver_mock, \ + patch("cron.scheduler.mark_job_run"): + save_mock.return_value = "/tmp/out.md" + from cron.scheduler import tick + tick(verbose=False) + save_mock.assert_called_once_with("monitor-job", "# full output") + deliver_mock.assert_not_called() def test_whitespace_only_response_is_marked_failed_not_delivered(self): """Whitespace-only final responses should behave like empty responses.""" with patch("cron.scheduler.get_due_jobs", return_value=[self._make_job()]), \ + patch("cron.scheduler.claim_job_for_fire", return_value=True), \ patch("cron.scheduler.run_job", return_value=(True, "# output", " \n\t ", None)), \ patch("cron.scheduler.save_job_output", return_value="/tmp/out.md"), \ patch("cron.scheduler._deliver_result") as deliver_mock, \ @@ -1101,6 +1497,7 @@ def _oneshot(self): def test_claim_runs_before_run_job(self): order = [] with patch("cron.scheduler.get_due_jobs", return_value=[self._oneshot()]), \ + patch("cron.scheduler.claim_job_for_fire", return_value=True), \ patch("cron.scheduler.claim_dispatch", side_effect=lambda _id: order.append("claim") or True), \ patch("cron.scheduler.run_job", side_effect=lambda _j, **_kw: order.append("run") or (True, "# out", "ok", None)), \ patch("cron.scheduler.save_job_output", return_value="/tmp/out.md"), \ @@ -1110,6 +1507,20 @@ def test_claim_runs_before_run_job(self): tick(verbose=False) assert order == ["claim", "run"] # claim strictly before side effect + def test_refused_claim_skips_run_job(self): + with patch("cron.scheduler.get_due_jobs", return_value=[self._oneshot()]), \ + patch("cron.scheduler.claim_job_for_fire", return_value=True), \ + patch("cron.scheduler.claim_dispatch", return_value=False), \ + patch("cron.scheduler.run_job") as run_mock, \ + patch("cron.scheduler.save_job_output"), \ + patch("cron.scheduler._deliver_result") as deliver_mock, \ + patch("cron.scheduler.mark_job_run") as mark_mock: + from cron.scheduler import tick + tick(verbose=False) + run_mock.assert_not_called() + deliver_mock.assert_not_called() + mark_mock.assert_not_called() + class TestBuildJobPromptSilentHint: """Verify _build_job_prompt always injects [SILENT] guidance.""" @@ -1355,7 +1766,7 @@ def mock_run_job(job, *, defer_agent_teardown=None, **kw): ] with patch("cron.scheduler.get_due_jobs", return_value=jobs), \ - patch("cron.scheduler.advance_next_runs"), \ + patch("cron.scheduler.claim_job_for_fire", return_value=True), \ patch("cron.scheduler.run_job", side_effect=mock_run_job), \ patch("cron.scheduler.save_job_output", return_value="/tmp/out.md"), \ patch("cron.scheduler._deliver_result", return_value=None), \ @@ -1400,7 +1811,7 @@ def mock_run_job(job, *, defer_agent_teardown=None, **kw): ] with patch("cron.scheduler.get_due_jobs", return_value=jobs), \ - patch("cron.scheduler.advance_next_runs"), \ + patch("cron.scheduler.claim_job_for_fire", return_value=True), \ patch("cron.scheduler.run_job", side_effect=mock_run_job), \ patch("cron.scheduler.save_job_output", return_value="/tmp/out.md"), \ patch("cron.scheduler._deliver_result", return_value=None), \ @@ -1411,6 +1822,38 @@ def mock_run_job(job, *, defer_agent_teardown=None, **kw): assert seen["tg-job"] == {"platform": "telegram", "chat_id": "111"} assert seen["dc-job"] == {"platform": "discord", "chat_id": "222"} + def test_max_parallel_env_var(self, monkeypatch): + """HERMES_CRON_MAX_PARALLEL=1 should restore serial behaviour.""" + monkeypatch.setenv("HERMES_CRON_MAX_PARALLEL", "1") + call_times = [] + + def mock_run_job(job, *, defer_agent_teardown=None, **_kw): + import time + call_times.append(("start", job["id"], time.monotonic())) + time.sleep(0.05) + call_times.append(("end", job["id"], time.monotonic())) + return (True, "output", "response", None) + + jobs = [ + {"id": "s1", "name": "s1", "deliver": "local"}, + {"id": "s2", "name": "s2", "deliver": "local"}, + ] + + with patch("cron.scheduler.get_due_jobs", return_value=jobs), \ + patch("cron.scheduler.claim_job_for_fire", return_value=True), \ + patch("cron.scheduler.run_job", side_effect=mock_run_job), \ + patch("cron.scheduler.save_job_output", return_value="/tmp/out.md"), \ + patch("cron.scheduler._deliver_result", return_value=None), \ + patch("cron.scheduler.mark_job_run"): + from cron.scheduler import tick + result = tick(verbose=False) + + assert result == 2 + # With max_workers=1, second job starts after first ends + end_s1 = [t for action, jid, t in call_times if action == "end" and jid == "s1"][0] + start_s2 = [t for action, jid, t in call_times if action == "start" and jid == "s2"][0] + assert start_s2 >= end_s1, "Jobs ran concurrently despite max_parallel=1" + class TestDeliverResultTimeoutCancelsFuture: """When future.result(timeout=60) raises TimeoutError in the live adapter diff --git a/tests/cron/test_scheduler_provider.py b/tests/cron/test_scheduler_provider.py index 0229d8dd48c56..552fbdf5a11dd 100644 --- a/tests/cron/test_scheduler_provider.py +++ b/tests/cron/test_scheduler_provider.py @@ -129,6 +129,39 @@ def test_abc_growth_stays_additive(): ) +def test_force_fire_capability_detects_legacy_override(): + from cron.scheduler_provider import CronScheduler + + class Current(CronScheduler): + @property + def name(self): + return "current" + + def start(self, stop_event, **kw): + pass + + class Legacy(Current): + def fire_due( # type: ignore[invalid-method-override] + self, job_id, *, adapters=None, loop=None + ): + return True + + class PositionalOnly(Current): + def fire_due( # type: ignore[invalid-method-override] + self, job_id, force=False, / + ): + return True + + class KeywordSink(Current): + def fire_due(self, job_id, **kwargs): + return True + + assert Current().supports_force_fire is True + assert Legacy().supports_force_fire is False + assert PositionalOnly().supports_force_fire is False + assert KeywordSink().supports_force_fire is True + + def test_inprocess_provider_ticks_and_stops(): """The built-in provider drives cron.scheduler.tick(sync=False) on a loop and exits promptly when stop_event is set — same contract as the raw @@ -214,6 +247,98 @@ def test_resolve_defaults_to_builtin(monkeypatch): assert prov.name == "builtin" +def test_resolve_no_cron_section_falls_back_to_builtin(monkeypatch): + """Config with no cron section at all → built-in (cfg_get returns default).""" + import hermes_cli.config as cfg + from cron import scheduler_provider as sp + + monkeypatch.setattr(cfg, "load_config", lambda: {}) + prov = sp.resolve_cron_scheduler() + assert prov.name == "builtin" + + +def test_resolve_unknown_provider_falls_back_to_builtin(monkeypatch): + """A named provider that doesn't exist → built-in (cron never dies).""" + import hermes_cli.config as cfg + from cron import scheduler_provider as sp + + monkeypatch.setattr(cfg, "load_config", lambda: {"cron": {"provider": "nope-not-real"}}) + prov = sp.resolve_cron_scheduler() + assert prov.name == "builtin" + + +def test_resolve_unavailable_provider_falls_back(monkeypatch): + """A provider that loads but reports is_available()==False → built-in.""" + import hermes_cli.config as cfg + import plugins.cron_providers as pc + from cron import scheduler_provider as sp + from cron.scheduler_provider import CronScheduler + + class Unavailable(CronScheduler): + @property + def name(self): + return "unavailable" + + def is_available(self): + return False + + def start(self, stop_event, **kw): + pass + + monkeypatch.setattr(cfg, "load_config", lambda: {"cron": {"provider": "unavailable"}}) + monkeypatch.setattr(pc, "load_cron_scheduler", lambda n: Unavailable()) + prov = sp.resolve_cron_scheduler() + assert prov.name == "builtin" + + +def test_resolve_available_provider_is_used(monkeypatch): + """A provider that loads and is available is returned (not the fallback).""" + import hermes_cli.config as cfg + import plugins.cron_providers as pc + from cron import scheduler_provider as sp + from cron.scheduler_provider import CronScheduler + + class Fake(CronScheduler): + @property + def name(self): + return "fake" + + def is_available(self): + return True + + def start(self, stop_event, **kw): + pass + + monkeypatch.setattr(cfg, "load_config", lambda: {"cron": {"provider": "fake"}}) + monkeypatch.setattr(pc, "load_cron_scheduler", lambda n: Fake()) + prov = sp.resolve_cron_scheduler() + assert prov.name == "fake" + + +def test_external_provider_falls_back_to_builtin_under_multiplex(): + from cron.scheduler_provider import ( + CronScheduler, + InProcessCronScheduler, + scheduler_for_profile_mode, + ) + + class External(CronScheduler): + @property + def name(self): + return "external" + + def start(self, stop_event, **kwargs): + return None + + external = External() + + assert scheduler_for_profile_mode(external, multiplex_profiles=False) is external + assert isinstance( + scheduler_for_profile_mode(external, multiplex_profiles=True), + InProcessCronScheduler, + ) + + # ── Phase 4B: additive hooks (on_jobs_changed / fire_due / reconcile) ──────── @@ -238,19 +363,120 @@ def test_builtin_inherits_hook_defaults(): def test_fire_due_default_claims_then_runs(monkeypatch): - """The default fire_due claims via the store CAS, fetches the job, and runs - it through the shared run_one_job body.""" + """The default fire_due runs the exact owner-bearing CAS snapshot.""" import cron.jobs as jobs import cron.scheduler as sched from cron.scheduler_provider import InProcessCronScheduler ran = [] - monkeypatch.setattr(jobs, "claim_job_for_fire", lambda jid: True, raising=False) - monkeypatch.setattr(jobs, "get_job", lambda jid: {"id": jid, "name": "t"}) - monkeypatch.setattr(sched, "run_one_job", lambda job, **kw: ran.append(job["id"]) or True) + claims = [] + monkeypatch.setattr( + jobs, + "claim_job_for_fire", + lambda jid, **kw: claims.append((jid, kw)) + or {"id": jid, "name": "t", "fire_claim": {"by": "exact-owner"}}, + raising=False, + ) + monkeypatch.setattr( + sched, + "run_one_job", + lambda job, **kw: ran.append((job["id"], job["fire_claim"]["by"])) or True, + ) assert InProcessCronScheduler().fire_due("j1") is True - assert ran == ["j1"] + assert claims == [("j1", {"return_job": True})] + assert ran == [("j1", "exact-owner")] + + +def test_claim_fire_persists_attempt_before_fire_claimed(monkeypatch): + import cron.executions as executions + import cron.jobs as jobs + import cron.scheduler as sched + from cron.scheduler_provider import InProcessCronScheduler + + events = [] + monkeypatch.setattr( + jobs, + "claim_job_for_fire", + lambda jid, **kwargs: events.append("claim") + or {"id": jid, "fire_claim": {"by": "owner"}}, + ) + monkeypatch.setattr( + executions, + "create_execution", + lambda jid, source: events.append("ledger") or {"id": "exec-1"}, + ) + monkeypatch.setattr( + sched, + "run_one_job", + lambda job, **kwargs: events.append(("run", job["execution_id"])) or True, + ) + + provider = InProcessCronScheduler() + claimed = provider.claim_fire("j1") + + assert events == ["ledger", "claim"] + assert claimed is not None + assert claimed["execution_id"] == "exec-1" + assert provider.fire_claimed(claimed) is True + assert events == ["ledger", "claim", ("run", "exec-1")] + + +def test_fire_due_forwards_manual_force_to_store_claim(monkeypatch): + import cron.jobs as jobs + import cron.scheduler as sched + from cron.scheduler_provider import InProcessCronScheduler + + claims = [] + monkeypatch.setattr( + jobs, + "claim_job_for_fire", + lambda jid, **kw: claims.append((jid, kw)) + or {"id": jid, "name": "t", "fire_claim": {"by": "manual-owner"}}, + ) + monkeypatch.setattr(sched, "run_one_job", lambda job, **kw: True) + + assert InProcessCronScheduler().fire_due("j1", force=True) is True + assert claims == [("j1", {"force": True, "return_job": True})] + + +def test_fire_due_lost_claim_does_not_run(monkeypatch): + """If the CAS claim is lost (another machine/retry won), fire_due returns + False and never runs the job.""" + import cron.jobs as jobs + import cron.scheduler as sched + from cron.scheduler_provider import InProcessCronScheduler + + ran = [] + monkeypatch.setattr( + jobs, + "claim_job_for_fire", + lambda jid, **kw: False, + raising=False, + ) + monkeypatch.setattr(sched, "run_one_job", lambda job, **kw: ran.append(job["id"]) or True) + + assert InProcessCronScheduler().fire_due("j1") is False + assert ran == [] + + +def test_fire_due_missing_job_does_not_run(monkeypatch): + """If the job vanished before atomic claim, fire_due does not run it.""" + import cron.jobs as jobs + import cron.scheduler as sched + from cron.scheduler_provider import InProcessCronScheduler + + ran = [] + monkeypatch.setattr( + jobs, + "claim_job_for_fire", + lambda jid, **kw: False, + raising=False, + ) + monkeypatch.setattr(sched, "run_one_job", lambda job, **kw: ran.append(job["id"]) or True) + + assert InProcessCronScheduler().fire_due("gone") is False + assert ran == [] # ── F2a: ticker liveness — survival, heartbeat, honest status (#32612, #32895) ── diff --git a/tests/cron/test_script_claim_heartbeat.py b/tests/cron/test_script_claim_heartbeat.py index 0caa7a95554be..da393b3ee35d7 100644 --- a/tests/cron/test_script_claim_heartbeat.py +++ b/tests/cron/test_script_claim_heartbeat.py @@ -1,12 +1,163 @@ """Regression coverage for one-shot claims during blocking cron scripts.""" from datetime import datetime, timedelta, timezone +import contextlib +import sys import threading +import time from unittest.mock import MagicMock, patch import pytest +def test_cancel_event_terminates_script_process_tree(tmp_path, monkeypatch): + """Losing a fire claim must stop both the script and its descendants.""" + import cron.scheduler as scheduler + + monkeypatch.setattr(scheduler, "_get_hermes_home", lambda: tmp_path) + scripts_dir = tmp_path / "scripts" + scripts_dir.mkdir() + started = tmp_path / "started" + child_done = tmp_path / "child-done" + script = scripts_dir / "blocking.py" + child_code = ( + "import time; from pathlib import Path; " + f"time.sleep(1); Path({str(child_done)!r}).write_text('done')" + ) + script.write_text( + "import subprocess, sys, time\n" + f"subprocess.Popen([sys.executable, '-c', {child_code!r}])\n" + f"open({str(started)!r}, 'w').close()\n" + "time.sleep(30)\n", + encoding="utf-8", + ) + + cancel = threading.Event() + result = [] + errors = [] + + def _run() -> None: + try: + result.append( + scheduler._run_job_script( + str(script), + workdir=str(tmp_path), + cancel_event=cancel, + ) + ) + except Exception as exc: + errors.append(exc) + + thread = threading.Thread(target=_run) + thread.start() + deadline = time.monotonic() + 5 + while not started.exists() and not errors and time.monotonic() < deadline: + time.sleep(0.01) + assert errors == [] + assert started.exists(), "script did not start" + + cancel.set() + thread.join(timeout=3) + + assert errors == [] + assert not thread.is_alive(), "script ignored cancellation" + assert result and result[0][0] is False + assert "cancel" in result[0][1].lower() + time.sleep(1.2) + assert not child_done.exists(), "script descendant survived cancellation" + + +@pytest.mark.skipif(sys.platform == "win32", reason="POSIX process-group semantics") +def test_cancel_event_kills_sigterm_ignoring_descendant(tmp_path, monkeypatch): + """A SIGTERM-ignoring grandchild must not wedge the cancellation path: + the tree kill escalates to SIGKILL for surviving group members, and the + pipe drain is bounded even if a descendant still holds the write ends.""" + import cron.scheduler as scheduler + + monkeypatch.setattr(scheduler, "_get_hermes_home", lambda: tmp_path) + scripts_dir = tmp_path / "scripts" + scripts_dir.mkdir() + started = tmp_path / "started" + script = scripts_dir / "stubborn.py" + child_code = ( + "import signal, time; " + "signal.signal(signal.SIGTERM, signal.SIG_IGN); " + f"open({str(started)!r}, 'w').close(); " + "time.sleep(60)" + ) + script.write_text( + "import subprocess, sys, time\n" + f"subprocess.Popen([sys.executable, '-c', {child_code!r}])\n" + "time.sleep(60)\n", + encoding="utf-8", + ) + + cancel = threading.Event() + result = [] + errors = [] + + def _run() -> None: + try: + result.append( + scheduler._run_job_script( + str(script), + workdir=str(tmp_path), + cancel_event=cancel, + ) + ) + except Exception as exc: + errors.append(exc) + + thread = threading.Thread(target=_run) + thread.start() + deadline = time.monotonic() + 5 + while not started.exists() and not errors and time.monotonic() < deadline: + time.sleep(0.01) + assert errors == [] + assert started.exists(), "script did not spawn its descendant" + + cancel.set() + # TERM grace (1s) + KILL + bounded drain (5s) + margin: must return well + # before the unbounded-communicate hang this regresses against. + thread.join(timeout=10) + + assert errors == [] + assert not thread.is_alive(), "cancellation wedged on a SIGTERM-ignoring descendant" + assert result and result[0][0] is False + assert "cancel" in result[0][1].lower() + + +def test_no_agent_forwards_cancel_event_to_script_runner(monkeypatch): + import cron.scheduler as scheduler + + cancel = threading.Event() + observed = [] + + def _script_runner(job, script_path, workdir=None, cancel_event=None): + observed.append(cancel_event) + return True, "" + + monkeypatch.setattr( + scheduler, + "_run_job_script_with_claim_heartbeat", + _script_runner, + ) + + success, _output, _response, error = scheduler.run_job( + { + "id": "cancel-aware-script", + "name": "cancel aware", + "script": "watchdog.py", + "no_agent": True, + }, + cancel_event=cancel, + ) + + assert success is True + assert error is None + assert observed == [cancel] + + @pytest.mark.parametrize( ("no_agent", "script_output"), [ @@ -163,3 +314,267 @@ def _blocking_script(_script_path: str, **kwargs) -> tuple[bool, str]: "at": replacement_timestamp, "by": "replacement-owner", } + + +def test_run_one_job_refreshes_fire_claim_in_profile_store(tmp_path, monkeypatch): + """The shared execute/save/deliver body keeps its durable fire claim alive.""" + import cron.jobs as jobs + import cron.scheduler as scheduler + + profile_home = tmp_path / "profile" + profile_home.mkdir() + with jobs.use_cron_store(profile_home): + job = jobs.create_job(prompt="x", schedule="every 5m", name="agent-run") + assert jobs.claim_job_for_fire(job["id"]) is True + claimed_job = jobs.get_job(job["id"]) + original_claim = dict(claimed_job["fire_claim"]) + + heartbeat_seen = threading.Event() + real_heartbeat = jobs.heartbeat_fire_claim + + def _observed_heartbeat(job_id: str, *, expected_owner: str) -> bool: + updated = real_heartbeat(job_id, expected_owner=expected_owner) + heartbeat_seen.set() + return updated + + def _blocking_body(job, **kwargs): + assert heartbeat_seen.wait(timeout=2) + return True + + monkeypatch.setattr(scheduler, "_RUN_CLAIM_HEARTBEAT_SECONDS", 0.01) + monkeypatch.setattr(scheduler, "heartbeat_fire_claim", _observed_heartbeat) + monkeypatch.setattr(scheduler, "_run_one_job_body", _blocking_body) + + with jobs.use_cron_store(profile_home): + assert isinstance(claimed_job, dict) + assert scheduler.run_one_job(claimed_job) is True + refreshed = jobs.get_job(job["id"])["fire_claim"] + + assert refreshed["at"] != original_claim["at"] + assert refreshed["by"] == original_claim["by"] + + +def test_lost_fire_claim_stops_stale_delivery(monkeypatch): + """A runner that loses its durable owner must not deliver its stale result.""" + import cron.scheduler as scheduler + + lost_seen = threading.Event() + heartbeat_calls = 0 + + def _heartbeat(job_id: str, *, expected_owner: str) -> bool: + nonlocal heartbeat_calls + heartbeat_calls += 1 + if heartbeat_calls == 1: + return True + lost_seen.set() + return False + + def _run_job(job, *, defer_agent_teardown=None, extra_prompt=None, cancel_event=None): + assert lost_seen.wait(timeout=2) + return True, "stale output", "stale response", None + + job = { + "id": "reclaimed-agent", + "name": "reclaimed agent", + "prompt": "work", + "execution_id": "stale-execution", + "fire_claim": {"at": "2026-07-12T12:00:00+00:00", "by": "stale-owner"}, + } + monkeypatch.setattr(scheduler, "_RUN_CLAIM_HEARTBEAT_SECONDS", 0.01) + monkeypatch.setattr(scheduler, "heartbeat_fire_claim", _heartbeat) + monkeypatch.setattr(scheduler, "run_job", _run_job) + monkeypatch.setattr(scheduler, "claim_dispatch", lambda job_id: True) + monkeypatch.setattr(scheduler, "mark_execution_running", lambda execution_id: None) + monkeypatch.setattr(scheduler, "finish_execution", lambda *args, **kwargs: None) + save_output = MagicMock() + deliver_result = MagicMock() + mark_run = MagicMock() + monkeypatch.setattr(scheduler, "save_job_output", save_output) + monkeypatch.setattr(scheduler, "_deliver_result", deliver_result) + monkeypatch.setattr(scheduler, "mark_job_run", mark_run) + + with patch("agent.secret_scope.set_secret_scope", return_value=None), \ + patch("agent.secret_scope.build_profile_secret_scope", return_value=None), \ + patch("agent.secret_scope.reset_secret_scope"): + assert scheduler.run_one_job(job) is True + + save_output.assert_not_called() + deliver_result.assert_not_called() + mark_run.assert_not_called() + + +def test_initially_lost_fire_claim_finishes_execution_without_running(monkeypatch): + """A stale claimed snapshot rejected before body entry must close its ledger row.""" + import cron.scheduler as scheduler + + run_body = MagicMock(return_value=True) + finish = MagicMock() + job = { + "id": "already-reclaimed", + "execution_id": "stale-execution", + "fire_claim": {"at": "2026-07-12T12:00:00+00:00", "by": "stale-owner"}, + } + monkeypatch.setattr(scheduler, "heartbeat_fire_claim", lambda *args, **kwargs: False) + monkeypatch.setattr(scheduler, "_run_one_job_body", run_body) + monkeypatch.setattr(scheduler, "finish_execution", finish) + + assert scheduler.run_one_job(job) is True + + run_body.assert_not_called() + finish.assert_called_once_with( + "stale-execution", + success=False, + error="Fire claim ownership lost before execution started.", + ) + + +def test_initially_lost_claim_does_not_run_when_ledger_write_fails(monkeypatch): + """A ledger I/O error cannot turn a confirmed ownership loss into execution.""" + import cron.scheduler as scheduler + + run_body = MagicMock(return_value=True) + job = { + "id": "already-reclaimed", + "execution_id": "stale-execution", + "fire_claim": {"at": "2026-07-12T12:00:00+00:00", "by": "stale-owner"}, + } + monkeypatch.setattr(scheduler, "heartbeat_fire_claim", lambda *args, **kwargs: False) + monkeypatch.setattr(scheduler, "_run_one_job_body", run_body) + monkeypatch.setattr( + scheduler, + "finish_execution", + MagicMock(side_effect=OSError("ledger unavailable")), + ) + + assert scheduler.run_one_job(job) is True + run_body.assert_not_called() + + +def test_initial_heartbeat_exception_does_not_start_execution(monkeypatch): + """Unconfirmed initial ownership must fail closed before any side effect.""" + import cron.scheduler as scheduler + + run_body = MagicMock(return_value=True) + finish = MagicMock() + job = { + "id": "validation-error", + "execution_id": "validation-execution", + "fire_claim": {"at": "2026-07-12T12:00:00+00:00", "by": "owner"}, + } + monkeypatch.setattr( + scheduler, + "heartbeat_fire_claim", + MagicMock(side_effect=OSError("store unavailable")), + ) + monkeypatch.setattr(scheduler, "_run_one_job_body", run_body) + monkeypatch.setattr(scheduler, "finish_execution", finish) + + assert scheduler.run_one_job(job) is True + + run_body.assert_not_called() + finish.assert_called_once_with( + "validation-execution", + success=False, + error="Fire claim ownership could not be validated before execution started.", + ) + + +def test_heartbeat_thread_start_failure_does_not_start_execution(monkeypatch): + """A claimed job cannot run when no renewal monitor protects its lease.""" + import cron.scheduler as scheduler + + run_body = MagicMock(return_value=True) + finish = MagicMock() + job = { + "id": "thread-start-error", + "execution_id": "thread-execution", + "fire_claim": {"at": "2026-07-12T12:00:00+00:00", "by": "owner"}, + } + monkeypatch.setattr(scheduler, "heartbeat_fire_claim", lambda *args, **kwargs: True) + monkeypatch.setattr(scheduler, "_run_one_job_body", run_body) + monkeypatch.setattr(scheduler, "finish_execution", finish) + monkeypatch.setattr( + scheduler.threading.Thread, + "start", + MagicMock(side_effect=RuntimeError("cannot start thread")), + ) + + assert scheduler.run_one_job(job) is True + + run_body.assert_not_called() + finish.assert_called_once_with( + "thread-execution", + success=False, + error="Fire claim heartbeat could not be started; execution was not run.", + ) + + +def test_repeated_heartbeat_errors_cancel_after_bounded_grace(monkeypatch): + """Store uncertainty cannot let a run outlive its last confirmed lease forever.""" + import cron.scheduler as scheduler + + calls = 0 + + def heartbeat(*_args, **_kwargs): + nonlocal calls + calls += 1 + if calls == 1: + return True + raise OSError("store unavailable") + + def run_body(_job, **kwargs): + assert kwargs["fire_claim_lost"].wait(timeout=0.5) + return True + + job = { + "id": "heartbeat-errors", + "fire_claim": {"at": "2026-07-12T12:00:00+00:00", "by": "owner"}, + } + monkeypatch.setattr(scheduler, "heartbeat_fire_claim", heartbeat) + monkeypatch.setattr(scheduler, "_run_one_job_body", run_body) + monkeypatch.setattr(scheduler, "_RUN_CLAIM_HEARTBEAT_SECONDS", 0.01) + monkeypatch.setattr(scheduler, "_FIRE_CLAIM_HEARTBEAT_GRACE_SECONDS", 0.03) + + assert scheduler.run_one_job(job) is True + assert calls >= 3 + + +def test_terminal_owner_cas_failure_marks_ledger_ownership_lost(monkeypatch): + """A replacement owner cannot leave the stale ledger recorded as success.""" + import cron.scheduler as scheduler + + @contextlib.contextmanager + def owned_fence(*_args, **_kwargs): + yield True + + job = { + "id": "terminal-cas", + "execution_id": "execution-cas", + "name": "terminal-cas", + "fire_claim": {"at": "2026-07-12T12:00:00+00:00", "by": "owner"}, + } + finish = MagicMock() + monkeypatch.setattr(scheduler, "heartbeat_fire_claim", lambda *args, **kwargs: True) + monkeypatch.setattr(scheduler, "claim_dispatch", lambda *_args, **_kwargs: True) + monkeypatch.setattr(scheduler, "mark_execution_running", lambda *_args: None) + monkeypatch.setattr( + scheduler, + "run_job", + lambda *_args, **_kwargs: (True, "output", "response", None), + ) + monkeypatch.setattr(scheduler, "fire_claim_fence", owned_fence, raising=False) + monkeypatch.setattr(scheduler, "save_job_output", lambda *_args: "output.md") + monkeypatch.setattr(scheduler, "_deliver_result", lambda *_args, **_kwargs: None) + monkeypatch.setattr(scheduler, "mark_job_run", lambda *_args, **_kwargs: False) + monkeypatch.setattr(scheduler, "finish_execution", finish) + + with patch("agent.secret_scope.set_secret_scope", return_value=None), \ + patch("agent.secret_scope.build_profile_secret_scope", return_value=None), \ + patch("agent.secret_scope.reset_secret_scope"): + assert scheduler.run_one_job(job) is True + + finish.assert_called_once_with( + "execution-cas", + success=False, + error="Fire claim ownership lost before terminal completion.", + ) diff --git a/tests/cron/test_sessiondb_init_hang.py b/tests/cron/test_sessiondb_init_hang.py index 9f89574309b8d..ce959325c5f0f 100644 --- a/tests/cron/test_sessiondb_init_hang.py +++ b/tests/cron/test_sessiondb_init_hang.py @@ -222,7 +222,7 @@ def test_guard_is_released_and_job_refires_after_sessiondb_hang(self, tmp_path, side_effect=_session_db_executor(timeouts), ), \ patch.object(sched, "get_due_jobs", return_value=[job]), \ - patch.object(sched, "advance_next_runs"), \ + patch.object(sched, "claim_job_for_fire", return_value=True), \ patch.object(sched, "save_job_output", return_value="/tmp/out"), \ patch.object(sched, "mark_job_run"), \ patch.object(sched, "_deliver_result", return_value=None): diff --git a/tests/cron/test_shutdown_interrupt.py b/tests/cron/test_shutdown_interrupt.py index edafdb741acf6..a8bb3bf28f135 100644 --- a/tests/cron/test_shutdown_interrupt.py +++ b/tests/cron/test_shutdown_interrupt.py @@ -11,6 +11,7 @@ result AFTER its tool was already killed out from under it """ +import threading from unittest.mock import patch import pytest @@ -23,9 +24,11 @@ def _reset_scheduler_state(): import cron.scheduler as sched sched._running_job_ids.clear() + sched._running_fire_owners.clear() sched._interrupted_job_ids.clear() yield sched._running_job_ids.clear() + sched._running_fire_owners.clear() sched._interrupted_job_ids.clear() @@ -72,8 +75,15 @@ def test_marks_every_in_flight_job(self): import cron.scheduler as sched sched._running_job_ids.update({"job-1", "job-2"}) - - with patch("cron.scheduler.mark_job_run") as mock_mark: + profile_home = sched._get_hermes_home().resolve() + sched._running_fire_owners.update( + { + "job-1": {object(): ("owner-1", profile_home)}, + "job-2": {object(): ("owner-2", profile_home)}, + } + ) + + with patch("cron.scheduler.mark_job_run", return_value=True) as mock_mark: marked = sched.mark_running_jobs_interrupted("gateway shutdown (final-cleanup)") assert sorted(marked) == ["job-1", "job-2"] @@ -84,6 +94,7 @@ def test_marks_every_in_flight_job(self): # success must be False -- an interrupted run is never "ok". assert c.args[1] is False assert "gateway shutdown" in c.args[2] + assert c.kwargs["expected_fire_owner"] in {"owner-1", "owner-2"} def test_sets_interrupted_flag_for_consumption_by_run_one_job(self): import cron.scheduler as sched @@ -102,16 +113,146 @@ def test_one_job_marking_failure_does_not_block_the_others(self): import cron.scheduler as sched sched._running_job_ids.update({"job-1", "job-2"}) + profile_home = sched._get_hermes_home().resolve() + sched._running_fire_owners.update( + { + "job-1": {object(): ("owner-1", profile_home)}, + "job-2": {object(): ("owner-2", profile_home)}, + } + ) def _side_effect(job_id, success, reason, **kwargs): if job_id == "job-1": raise OSError("disk full") + return True with patch("cron.scheduler.mark_job_run", side_effect=_side_effect): marked = sched.mark_running_jobs_interrupted("shutdown") assert marked == ["job-2"] + def test_stale_shutdown_cannot_clear_replacement_owner(self, tmp_path): + import cron.jobs as jobs + import cron.scheduler as sched + + profile_home = tmp_path / "profile" + profile_home.mkdir() + with jobs.use_cron_store(profile_home): + created = jobs.create_job(prompt="x", schedule="every 5m", name="owned") + claimed = jobs.claim_job_for_fire(created["id"], force=True, return_job=True) + assert isinstance(claimed, dict) + stale_owner = claimed["fire_claim"]["by"] + original_status = claimed["last_status"] + replacement_claim = { + "at": "2026-07-12T12:30:00+00:00", + "by": "replacement-owner", + } + replacement = {**claimed, "fire_claim": replacement_claim} + jobs.save_jobs([replacement]) + + sched._running_job_ids.add(created["id"]) + sched._running_fire_owners[created["id"]] = { + object(): (stale_owner, profile_home) + } + marked = sched.mark_running_jobs_interrupted("shutdown") + refreshed = jobs.get_job(created["id"]) + + assert marked == [] + assert isinstance(refreshed, dict) + assert refreshed["fire_claim"] == replacement_claim + assert refreshed["last_status"] == original_status + + +class TestRunningFireOwnerRegistry: + def test_run_one_job_registers_owner_only_while_active(self): + import cron.scheduler as sched + + job = { + "id": "owned-job", + "fire_claim": {"at": "2026-07-12T12:00:00+00:00", "by": "owner-1"}, + } + + def _observe_registry(current_job, run): + assert list(sched._running_fire_owners[current_job["id"]].values()) == [ + ("owner-1", sched._get_hermes_home().resolve()) + ] + return True + + with patch("cron.scheduler._run_with_fire_claim_heartbeat", side_effect=_observe_registry): + assert sched.run_one_job(job) is True + + assert job["id"] not in sched._running_fire_owners + + def test_shutdown_sees_all_concurrent_direct_fire_owners(self, monkeypatch): + """Direct entry points and replacement owners share one token registry.""" + import cron.scheduler as sched + + entered = threading.Barrier(3) + release = threading.Event() + marked_owners: list[str] = [] + + def hold_run(_job, _run): + entered.wait(timeout=2) + release.wait(timeout=2) + return True + + def mark(_job_id, _success, _reason, *, expected_fire_owner): + marked_owners.append(expected_fire_owner) + return True + + monkeypatch.setattr(sched, "_run_with_fire_claim_heartbeat", hold_run) + monkeypatch.setattr(sched, "mark_job_run", mark) + + jobs = [ + {"id": "same-job", "fire_claim": {"by": "old-owner"}}, + {"id": "same-job", "fire_claim": {"by": "replacement-owner"}}, + ] + threads = [threading.Thread(target=sched.run_one_job, args=(job,)) for job in jobs] + for thread in threads: + thread.start() + entered.wait(timeout=2) + + assert sched.get_running_job_ids() == frozenset({"same-job"}) + assert sched.mark_running_jobs_interrupted("shutdown") == ["same-job", "same-job"] + assert set(marked_owners) == {"old-owner", "replacement-owner"} + + release.set() + for thread in threads: + thread.join(timeout=2) + assert not thread.is_alive() + assert "same-job" not in sched.get_running_job_ids() + + def test_shutdown_marks_each_owner_in_its_profile_store(self, monkeypatch, tmp_path): + import cron.jobs as cron_jobs + import cron.scheduler as sched + + profile_a = tmp_path / "a" + profile_b = tmp_path / "b" + observed = [] + sched._running_fire_owners["same-job"] = { + object(): ("owner-a", profile_a), + object(): ("owner-b", profile_b), + } + + def mark(job_id, success, reason, *, expected_fire_owner): + observed.append( + ( + job_id, + success, + expected_fire_owner, + cron_jobs._current_cron_store().jobs_file, + ) + ) + return True + + monkeypatch.setattr(sched, "mark_job_run", mark) + + assert sched.mark_running_jobs_interrupted("shutdown") == ["same-job", "same-job"] + assert set(observed) == { + ("same-job", False, "owner-a", profile_a / "cron" / "jobs.json"), + ("same-job", False, "owner-b", profile_b / "cron" / "jobs.json"), + } + class TestIsInterrupted: """Peek-only check used at the delivery gate -- must NOT clear the @@ -155,6 +296,247 @@ def test_true_and_clears_when_marked(self): assert sched._consume_interrupted_flag("job-1") is False +class TestExecutionScopedInterruption: + """Interruption flags must target ONE execution, not the job ID. + + Owner-registered executions are recorded by their unique execution + token, so a fresh run that reuses the same job ID (recurring fire, + replacement claim owner) never consumes a flag that targeted its + dead predecessor. + """ + + def test_interruption_targets_only_the_interrupted_execution(self): + import cron.scheduler as sched + + profile_home = sched._get_hermes_home().resolve() + old_token = object() + sched._running_fire_owners["job-1"] = { + old_token: ("owner-1", profile_home), + } + + with patch("cron.scheduler.mark_job_run", return_value=True): + sched.mark_running_jobs_interrupted("shutdown") + + assert sched._is_interrupted("job-1", old_token) is True + new_token = object() + assert sched._is_interrupted("job-1", new_token) is False + # A new execution must not steal (and thereby clear) the old flag. + assert sched._consume_interrupted_flag("job-1", new_token) is False + assert sched._consume_interrupted_flag("job-1", old_token) is True + assert sched._is_interrupted("job-1", old_token) is False + + def test_only_owners_marks_only_targeted_executions(self): + import cron.scheduler as sched + + profile_home = sched._get_hermes_home().resolve() + token_a, token_b = object(), object() + sched._running_fire_owners["job-a"] = {token_a: ("owner-a", profile_home)} + sched._running_fire_owners["job-b"] = {token_b: ("owner-b", profile_home)} + + with patch("cron.scheduler.mark_job_run", return_value=True) as mock_mark: + marked = sched.mark_running_jobs_interrupted( + "dashboard shutdown", + only_owners={("job-a", "owner-a")}, + ) + + assert marked == ["job-a"] + assert mock_mark.call_count == 1 + assert mock_mark.call_args.kwargs["expected_fire_owner"] == "owner-a" + assert sched._is_interrupted("job-a", token_a) is True + assert sched._is_interrupted("job-b", token_b) is False + + def test_replacement_execution_of_same_job_is_not_poisoned(self): + """A replacement owner starting while the stale flag exists must + complete through the normal mark path, not the interrupted one.""" + import cron.scheduler as sched + + profile_home = sched._get_hermes_home().resolve() + stale_token = object() + sched._running_fire_owners["job-1"] = { + stale_token: ("stale-owner", profile_home), + } + with patch("cron.scheduler.mark_job_run", return_value=True): + sched.mark_running_jobs_interrupted("shutdown") + sched._running_fire_owners.clear() + + job = { + "id": "job-1", + "name": "test job", + "prompt": "do work", + "fire_claim": {"by": "replacement-owner"}, + } + with patch("cron.scheduler.claim_dispatch", return_value=True), \ + patch("agent.secret_scope.set_secret_scope", return_value=None), \ + patch("agent.secret_scope.build_profile_secret_scope", return_value=None), \ + patch("agent.secret_scope.reset_secret_scope"), \ + patch( + "cron.scheduler.run_job", + return_value=(True, "full output", "final response", None), + ), \ + patch("cron.scheduler.save_job_output", return_value="/tmp/out.md"), \ + patch("cron.scheduler._is_cron_silence_response", return_value=False), \ + patch("cron.scheduler._deliver_result", return_value=None), \ + patch("cron.scheduler.fire_claim_fence"), \ + patch("cron.scheduler.heartbeat_fire_claim", return_value=True), \ + patch("cron.scheduler.mark_job_run", return_value=True) as mock_mark: + result = sched.run_one_job(job) + + assert result is True + mock_mark.assert_called_once() + + +class TestCombinedCancelEvent: + def test_or_semantics(self): + import cron.scheduler as sched + + a, b = threading.Event(), threading.Event() + combined = sched._CombinedCancelEvent(a, b) + assert combined.is_set() is False + b.set() + assert combined.is_set() is True + + def test_set_propagates_to_all(self): + import cron.scheduler as sched + + a, b = threading.Event(), threading.Event() + combined = sched._CombinedCancelEvent(a, b) + combined.set() + assert a.is_set() and b.is_set() + + def test_run_one_job_forwards_external_cancel_event(self): + import cron.scheduler as sched + + external = threading.Event() + job = {"id": "job-x", "name": "x", "prompt": "p"} + + with patch.object( + sched, + "_run_with_fire_claim_heartbeat", + side_effect=lambda job_arg, run: run(threading.Event()), + ), patch.object(sched, "_run_one_job_body", return_value=True) as body: + assert sched.run_one_job(job, cancel_event=external) is True + + combined = body.call_args.kwargs["fire_claim_lost"] + assert combined.is_set() is False + external.set() + assert combined.is_set() is True + + +class TestBaseExceptionThroughOwnerFencedFlow: + """#73973 (sweeper review on #70638): a BaseException escaping run_job + must still record a failed run through the owner-fenced terminal path — + and a stale worker must not record over a replacement claim owner.""" + + def _job(self): + return { + "id": "job-be", + "name": "base exc", + "prompt": "p", + "fire_claim": {"by": "owner-be"}, + } + + def _patches(self, run_side_effect): + return ( + patch("cron.scheduler.claim_dispatch", return_value=True), + patch("agent.secret_scope.set_secret_scope", return_value=None), + patch("agent.secret_scope.build_profile_secret_scope", return_value=None), + patch("agent.secret_scope.reset_secret_scope"), + patch("cron.scheduler.run_job", side_effect=run_side_effect), + patch("cron.scheduler.heartbeat_fire_claim", return_value=True), + ) + + def test_cancelled_error_records_failure_and_reraises(self): + import asyncio + + import cron.scheduler as sched + + p1, p2, p3, p4, p5, p6 = self._patches(asyncio.CancelledError()) + with p1, p2, p3, p4, p5, p6, \ + patch("cron.scheduler.mark_job_run", return_value=True) as mock_mark, \ + patch("cron.scheduler.finish_execution") as mock_finish: + try: + sched.run_one_job(self._job()) + raised = False + except asyncio.CancelledError: + raised = True + + assert raised, "non-Exception BaseException must propagate" + mock_mark.assert_called_once() + assert mock_mark.call_args.args[:3] == ("job-be", False, "CancelledError") + assert mock_mark.call_args.kwargs["expected_fire_owner"] == "owner-be" + assert mock_finish.call_args.kwargs["success"] is False + + def test_keyboard_interrupt_records_failure_and_reraises(self): + import cron.scheduler as sched + + p1, p2, p3, p4, p5, p6 = self._patches(KeyboardInterrupt()) + with p1, p2, p3, p4, p5, p6, \ + patch("cron.scheduler.mark_job_run", return_value=True) as mock_mark, \ + patch("cron.scheduler.finish_execution"): + try: + sched.run_one_job(self._job()) + raised = False + except KeyboardInterrupt: + raised = True + + assert raised + mock_mark.assert_called_once() + assert mock_mark.call_args.kwargs["expected_fire_owner"] == "owner-be" + + def test_base_exception_from_stale_owner_is_fenced_out(self): + """A replacement owner reclaimed the job: the stale worker's + BaseException path must NOT write terminal state over it.""" + import asyncio + + import cron.scheduler as sched + + p1, p2, p3, p4, p5, p6 = self._patches(asyncio.CancelledError()) + with p1, p2, p3, p4, p5, p6, \ + patch("cron.scheduler.mark_job_run", return_value=False) as mock_mark, \ + patch("cron.scheduler.finish_execution"): + try: + sched.run_one_job(self._job()) + except asyncio.CancelledError: + pass + + mock_mark.assert_called_once() + # fenced write was attempted with the stale owner and discarded by + # the store (return False) — and the code accepted that verdict + # without retrying or writing anything else. + assert mock_mark.call_args.kwargs["expected_fire_owner"] == "owner-be" + + +class TestCallerLossAfterClaimAcquisition: + """cirwel's integration assertion on #70638: if the HTTP/CLI caller is + lost AFTER the claim was acquired, the gateway owner must produce at + most one terminal ledger/artifact/delivery, clear only its own claim, + and block retries while that ownership is live.""" + + def test_second_fire_cannot_claim_while_first_ownership_live(self, tmp_path): + import cron.jobs as jobs + + with jobs.use_cron_store(tmp_path): + job = jobs.create_job(prompt="x", schedule="every 5m", name="owned") + claimed = jobs.claim_job_for_fire(job["id"], force=True, return_job=True) + assert isinstance(claimed, dict) + + # Caller died here — the claim outlives it. A retry (NAS/webhook + # or manual) must be refused while the lease is fresh. + retry = jobs.claim_job_for_fire(job["id"], return_job=True) + assert retry is False or not isinstance(retry, dict) + + # The live owner still heartbeats and terminally marks — exactly + # one terminal write, and only its own claim is cleared. + owner = claimed["fire_claim"]["by"] + assert jobs.heartbeat_fire_claim(job["id"], expected_owner=owner) is True + assert jobs.mark_job_run( + job["id"], True, expected_fire_owner=owner, + ) is True + refreshed = jobs.get_job(job["id"]) + assert refreshed["fire_claim"] is None + assert refreshed["last_status"] == "ok" + + class TestRunOneJobHonoursInterruptedFlag: """run_one_job() must not let a job's own completion overwrite a status the shutdown path already wrote for the same run.""" From 1a8625abee3a3e51e403775f3cef2c4c0b2ec4c2 Mon Sep 17 00:00:00 2001 From: Evgenii <413011+smwbev@users.noreply.github.com> Date: Sun, 9 Aug 2026 07:22:14 +0000 Subject: [PATCH 075/376] fix(cron): harden gateway fire admission and provider compatibility - The gateway api_server fire webhook acknowledges 202 only after a durable claim + execution row exist (admission failure stays retryable as 503; a live claim answers 200 duplicate), then dispatches the claimed snapshot with the live runner adapters (delivery parity with the built-in ticker, including relay-fronted and E2EE platforms). - Legacy single-phase providers (a documented fire_due override without split hooks) keep being driven through their own hook. Capability detection now credits claim_fire AND fire_claimed overrides, so Chronos is correctly classified split-aware (its re-arm lives in fire_claimed; the redundant fire_due passthrough override is removed). - Multi-profile dashboards fail closed for external providers: an unscoped reconcile would disarm other profiles' armed one-shots in the shared NAS registry. - Manual runs (cronjob run) carry the owner-bearing claimed snapshot through every entry point, composing with upstream's manual-run heartbeat (#76502) and background dispatch. Note: current main moved the dashboard NAS webhook to a pure forward-to-gateway design (the gateway owns execution and live adapters), so the dashboard-side claim/tracking machinery from earlier revisions of this PR is dropped; the durable admission contract lives in the gateway webhook path. --- gateway/platforms/api_server.py | 56 +- hermes_cli/web_routers/cron.py | 9 +- hermes_cli/web_server.py | 140 ++- plugins/cron_providers/chronos/__init__.py | 31 +- tests/gateway/test_cron_active_work_drain.py | 5 + tests/gateway/test_cron_fire_webhook.py | 88 +- tests/hermes_cli/test_cron_fire_dashboard.py | 5 +- .../test_web_server_cron_profiles.py | 860 ++++++++++++++++++ tests/plugins/test_chronos_cron.py | 115 ++- tests/tools/test_cronjob_run_background.py | 16 +- tests/tools/test_cronjob_run_immediate.py | 109 ++- tools/cronjob_tools.py | 34 +- 12 files changed, 1405 insertions(+), 63 deletions(-) diff --git a/gateway/platforms/api_server.py b/gateway/platforms/api_server.py index 53fb4a34f14c4..13c59ab0de841 100644 --- a/gateway/platforms/api_server.py +++ b/gateway/platforms/api_server.py @@ -5918,7 +5918,10 @@ async def _handle_cron_fire(self, request: "web.Request") -> "web.Response": if not job_id: return web.json_response({"error": "missing job_id"}, status=400) - from cron.scheduler_provider import resolve_cron_scheduler + from cron.scheduler_provider import ( + provider_supports_split_fire, + resolve_cron_scheduler, + ) provider = resolve_cron_scheduler() loop = asyncio.get_running_loop() @@ -5940,10 +5943,55 @@ async def _handle_cron_fire(self, request: "web.Request") -> "web.Response": runner = None adapters = getattr(runner, "adapters", None) or None - # Fire in the background (202 immediately). fire_due claims via the - # store CAS, so a retry while this is in flight is de-duped. + if not provider_supports_split_fire(provider): + # Legacy single-phase provider: it overrides the documented + # ``fire_due`` hook (custom claim/re-arm/telemetry) but + # inherits the base ``claim_fire`` — driving it through the + # split claim path would silently bypass that override. + task = asyncio.create_task( + asyncio.to_thread( + provider.fire_due, + job_id, + adapters=adapters, + loop=loop, + ) + ) + reservation["detached"] = True + task.add_done_callback( + lambda _task: _release_pending_api_work(self, reservation) + ) + try: + self._background_tasks.add(task) + task.add_done_callback(self._background_tasks.discard) + except (TypeError, AttributeError): + pass + return web.json_response( + {"status": "accepted", "job_id": job_id}, status=202 + ) + + # Persist the attempt and exact store owner before acknowledging NAS. + # A failure here is retryable and the reservation remains attached. + try: + claimed_job = await asyncio.to_thread(provider.claim_fire, job_id) + except Exception as exc: + logger.error("cron fire admission failed for %s: %s", job_id, exc) + return web.json_response( + {"error": "cron fire admission failed", "job_id": job_id}, + status=503, + ) + if claimed_job is None: + return web.json_response( + {"status": "duplicate", "job_id": job_id}, + status=200, + ) + task = asyncio.create_task( - asyncio.to_thread(provider.fire_due, job_id, adapters=adapters, loop=loop) + asyncio.to_thread( + provider.fire_claimed, + claimed_job, + adapters=adapters, + loop=loop, + ) ) reservation["detached"] = True task.add_done_callback( diff --git a/hermes_cli/web_routers/cron.py b/hermes_cli/web_routers/cron.py index e92b3c5ce53de..abe11f12cf859 100644 --- a/hermes_cli/web_routers/cron.py +++ b/hermes_cli/web_routers/cron.py @@ -44,6 +44,7 @@ _find_cron_job_profile = late("_find_cron_job_profile") _fire_cron_job_for_profile = late("_fire_cron_job_for_profile") _forward_cron_fire_to_gateway = late("_forward_cron_fire_to_gateway") +_notify_cron_provider_for_profile = late("_notify_cron_provider_for_profile") _call_cron_for_profile = late("_call_cron_for_profile") _raise_if_cron_registration_error = late("_raise_if_cron_registration_error") load_config = late("load_config") @@ -251,7 +252,13 @@ async def instantiate_blueprint(body: AutomationBlueprintInstantiate, profile: s # like the sibling cron endpoints (partial avoids **spec keys ever # colliding with the wrapper's own parameters). _create = functools.partial(_call_cron_for_profile, profile, "create_job", **spec) - return await _run_cron_dashboard_io(_create) + created = await _run_cron_dashboard_io(_create) + # Same contract as the other dashboard mutations: reconcile the + # profile-scoped provider (best-effort; fail-closed for external + # providers on a multi-profile dashboard). Off the event loop — + # a Chronos reconcile does file I/O plus NAS network calls. + await _run_cron_dashboard_io(_notify_cron_provider_for_profile, profile) + return created except HTTPException: raise except Exception as e: diff --git a/hermes_cli/web_server.py b/hermes_cli/web_server.py index e64e06d30d5ad..14af2cde45eaa 100644 --- a/hermes_cli/web_server.py +++ b/hermes_cli/web_server.py @@ -365,12 +365,12 @@ async def _lifespan(app: "FastAPI"): try: yield finally: + if cron_stop is not None: + cron_stop.set() pty_reaper_task.cancel() selftest_task.cancel() auto_archive_task.cancel() await PTY_REGISTRY.close_all() - if cron_stop is not None: - cron_stop.set() if os.getenv("HERMES_DESKTOP") == "1": _terminate_desktop_managed_gateway() @@ -417,6 +417,7 @@ def _get_pty_active_session_files(app: "FastAPI") -> dict[str, Path]: app = FastAPI(title="Hermes Agent", version=__version__, lifespan=_lifespan) + # Memory-provider OAuth connect routes live in the memory layer, not here. from hermes_cli.memory_oauth import router as _memory_oauth_router # noqa: E402 @@ -12044,6 +12045,73 @@ def _call_cron_for_profile(target_profile: Optional[str], func_name: str, *args, return result +def _notify_cron_provider_for_profile(target_profile: Optional[str]) -> None: + """Best-effort provider reconcile against one profile's job store. + + Fail-closed for external providers on a multi-profile dashboard: an + external provider's ``reconcile`` converges its REMOTE registry toward + one profile's jobs.json, and its orphan cleanup cancels every remote + entry absent from that store. The NAS registry is not profile-scoped, + so reconciling profile B would silently disarm profile A's one-shots. + Until the provider contract carries a profile identity through + arm/cancel/list, a multi-profile dashboard must not drive unscoped + external reconciles at all — the affected profile simply re-arms on + its next fire/start (idempotent via dedup_key). The built-in provider + re-reads jobs.json each tick and stays a no-op here. + """ + try: + _profile_name, home = _cron_profile_home(target_profile) + from cron import jobs as cron_jobs + from cron.scheduler_provider import ( + InProcessCronScheduler, + resolve_cron_scheduler, + ) + from hermes_constants import ( + reset_hermes_home_override, + set_hermes_home_override, + ) + + token = set_hermes_home_override(str(home)) + try: + with cron_jobs.use_cron_store(home): + provider = resolve_cron_scheduler() + if not isinstance(provider, InProcessCronScheduler): + profile_names = [ + str(p.get("name") or "") + for p in _cron_profile_dicts() + ] + if len([n for n in profile_names if n]) > 1: + _log.warning( + "Skipping cron provider reconcile for profile %s: " + "external provider '%s' reconcile is not " + "profile-scoped and would disarm other profiles' " + "armed one-shots. The mutated profile re-arms " + "idempotently on its next fire/start.", + target_profile, + provider.name, + ) + return + provider.on_jobs_changed() + finally: + reset_hermes_home_override(token) + except Exception: + _log.debug( + "Cron provider reconciliation failed for profile %s", + target_profile, + exc_info=True, + ) + + +def _mutate_cron_for_profile( + target_profile: Optional[str], func_name: str, *args, **kwargs +): + """Apply a cron store mutation and reconcile its scheduler provider.""" + result = _call_cron_for_profile(target_profile, func_name, *args, **kwargs) + if result: + _notify_cron_provider_for_profile(target_profile) + return result + + def _find_cron_job_profile(job_id: str) -> Optional[str]: for profile in _cron_profile_dicts(): name = str(profile.get("name") or "") @@ -12188,7 +12256,7 @@ def _create_cron_job_sync(body: CronJobCreate, profile: Optional[str] = None): "script": script, "no_agent": no_agent, }) - return _call_cron_for_profile( + return _mutate_cron_for_profile( profile_name, "create_job", prompt=body.prompt or "", @@ -12241,7 +12309,7 @@ def _update_cron_job_sync(job_id: str, body: CronJobUpdate, profile: Optional[st if "skills" in updates and "skill" not in updates: effective["skill"] = None _validate_dashboard_cron_effective_job(effective) - job = _call_cron_for_profile(profile_name, "update_job", job_id, updates) + job = _mutate_cron_for_profile(profile_name, "update_job", job_id, updates) except HTTPException: raise except ValueError as exc: @@ -12257,7 +12325,7 @@ def _pause_cron_job_sync(job_id: str, profile: Optional[str] = None): selected = profile or _find_cron_job_profile(job_id) if not selected: raise HTTPException(status_code=404, detail="Job not found") - job = _call_cron_for_profile(selected, "pause_job", job_id) + job = _mutate_cron_for_profile(selected, "pause_job", job_id) if not job: raise HTTPException(status_code=404, detail="Job not found") return job @@ -12269,7 +12337,7 @@ def _resume_cron_job_sync(job_id: str, profile: Optional[str] = None): selected = profile or _find_cron_job_profile(job_id) if not selected: raise HTTPException(status_code=404, detail="Job not found") - job = _call_cron_for_profile(selected, "resume_job", job_id) + job = _mutate_cron_for_profile(selected, "resume_job", job_id) if not job: raise HTTPException(status_code=404, detail="Job not found") return job @@ -12281,10 +12349,34 @@ def _trigger_cron_job_sync(job_id: str, profile: Optional[str] = None): selected = profile or _find_cron_job_profile(job_id) if not selected: raise HTTPException(status_code=404, detail="Job not found") - job = _call_cron_for_profile(selected, "trigger_job", job_id) + job = _call_cron_for_profile(selected, "resolve_job_ref", job_id) if not job: raise HTTPException(status_code=404, detail="Job not found") - return job + # Do not expose the job as due before claiming it: the built-in ticker and + # external/manual fire paths share the same durable claim, so only one can + # execute this selected run even if they race across processes. Active jobs + # keep the legacy provider call shape; paused jobs need the explicit force + # flag to resume and claim atomically. + force = not job.get("enabled", True) or job.get("state") == "paused" + ran = _fire_cron_job_for_profile(selected, job["id"], force=force) + refreshed = _call_cron_for_profile(selected, "get_job", job["id"]) + if refreshed and refreshed.get("last_run_at") != job.get("last_run_at"): + return refreshed + if not ran: + raise HTTPException( + status_code=409, + detail="Job is already running or was claimed by another scheduler", + ) + if refreshed: + return refreshed + # A one-shot may remove itself after exhausting repeat=1. Keep the response + # shape compatible without inventing an outcome that is no longer present + # in the job store; authoritative list refresh removes the completed row. + return { + **job, + "enabled": False, + "state": "completed", + } @@ -12294,7 +12386,7 @@ def _delete_cron_job_sync(job_id: str, profile: Optional[str] = None): if not selected: raise HTTPException(status_code=404, detail="Job not found") try: - removed = _call_cron_for_profile(selected, "remove_job", job_id) + removed = _mutate_cron_for_profile(selected, "remove_job", job_id) except ValueError as exc: raise HTTPException(status_code=400, detail=str(exc)) from exc if not removed: @@ -12304,8 +12396,17 @@ def _delete_cron_job_sync(job_id: str, profile: Optional[str] = None): -def _fire_cron_job_for_profile(profile: str, job_id: str) -> bool: - """DEPRECATED — retained only until callers migrate; do not add new uses. +def _fire_cron_job_for_profile( + profile: str, + job_id: str, + *, + force: bool = False, +) -> bool: + """DEPRECATED for NAS webhook fires (superseded by gateway forwarding); + retained for the dashboard trigger path — do not add new uses. + + Run ONE due cron job end-to-end for ``profile`` via the resolved + scheduler provider's ``fire_due`` (store CAS claim + ``run_one_job``). Superseded by :func:`_forward_cron_fire_to_gateway`: cron fires must execute in the GATEWAY process (which owns the live platform adapters), @@ -12317,7 +12418,10 @@ def _fire_cron_job_for_profile(profile: str, job_id: str) -> bool: """ _profile_name, home = _cron_profile_home(profile) from cron import jobs as cron_jobs - from cron.scheduler_provider import resolve_cron_scheduler + from cron.scheduler_provider import ( + provider_supports_force_fire, + resolve_cron_scheduler, + ) from hermes_constants import ( reset_hermes_home_override, set_hermes_home_override, @@ -12327,6 +12431,18 @@ def _fire_cron_job_for_profile(profile: str, job_id: str) -> bool: try: with cron_jobs.use_cron_store(home): provider = resolve_cron_scheduler() + if force: + if not provider_supports_force_fire(provider): + raise HTTPException( + status_code=409, + detail=( + f"Cron provider '{getattr(provider, 'name', 'custom')}' " + "does not support atomic forced firing of paused jobs" + ), + ) + return bool( + provider.fire_due(job_id, adapters=None, loop=None, force=True) + ) return bool(provider.fire_due(job_id, adapters=None, loop=None)) finally: reset_hermes_home_override(token) diff --git a/plugins/cron_providers/chronos/__init__.py b/plugins/cron_providers/chronos/__init__.py index b46d1d6dc5834..a4c80977a4b64 100644 --- a/plugins/cron_providers/chronos/__init__.py +++ b/plugins/cron_providers/chronos/__init__.py @@ -225,15 +225,28 @@ def reconcile(self) -> None: # -- fire ------------------------------------------------------------- - def fire_due(self, job_id: str, *, adapters: Any = None, loop: Any = None) -> bool: - """Run the due job (claim + run_one_job via the ABC default), then - re-arm the NEXT one-shot through NAS. - - Re-arm happens AFTER the run so next_run_at reflects the completed fire. - If the job is gone (one-shot completed / repeat-N exhausted), get_job - returns None → nothing to re-arm (the schedule naturally stops). - """ - ran = super().fire_due(job_id, adapters=adapters, loop=loop) + # NOTE: no ``fire_due`` override on purpose. The base implementation + # virtually dispatches through ``self.claim_fire``/``self.fire_claimed``, + # and ``provider_supports_split_fire`` treats ANY ``fire_due`` override + # (even a pure ``super()`` delegate) as the legacy single-phase signal — + # overriding it here would silently opt Chronos out of claim admission, + # duplicate detection, and the cancel-aware drain on the fire webhook. + + def fire_claimed( + self, + claimed_job: dict, + *, + adapters: Any = None, + loop: Any = None, + cancel_event: Any = None, + ) -> bool: + job_id = claimed_job["id"] + ran = super().fire_claimed( + claimed_job, + adapters=adapters, + loop=loop, + cancel_event=cancel_event, + ) if ran: from cron.jobs import get_job job = get_job(job_id) diff --git a/tests/gateway/test_cron_active_work_drain.py b/tests/gateway/test_cron_active_work_drain.py index 07616a7880a88..0a025288bc031 100644 --- a/tests/gateway/test_cron_active_work_drain.py +++ b/tests/gateway/test_cron_active_work_drain.py @@ -30,9 +30,11 @@ def _reset_cron_running_set(): import cron.scheduler as sched sched._running_job_ids.clear() + sched._running_fire_owners.clear() sched._interrupted_job_ids.clear() yield sched._running_job_ids.clear() + sched._running_fire_owners.clear() sched._interrupted_job_ids.clear() @@ -88,6 +90,9 @@ async def test_in_flight_cron_job_marked_interrupted_on_forced_kill(self, monkey adapter.disconnect = _make_async_noop() sched._running_job_ids.add("job-1") + sched._running_fire_owners["job-1"] = { + object(): ("owner-1", sched._get_hermes_home().resolve()) + } monkeypatch.setattr(_pr.process_registry, "kill_all", lambda task_id=None: 1) monkeypatch.setattr(_tt, "cleanup_all_environments", lambda: None) diff --git a/tests/gateway/test_cron_fire_webhook.py b/tests/gateway/test_cron_fire_webhook.py index f6a9f94a7f294..e315030a30065 100644 --- a/tests/gateway/test_cron_fire_webhook.py +++ b/tests/gateway/test_cron_fire_webhook.py @@ -39,13 +39,18 @@ def adapter(): class _SpyProvider: - """Records fire_due calls; stands in for the resolved provider.""" + """Records durable admission and claimed dispatch calls.""" def __init__(self): + self.claimed = [] self.fired = [] - def fire_due(self, job_id, *, adapters=None, loop=None): - self.fired.append(job_id) + def claim_fire(self, job_id): + self.claimed.append(job_id) + return {"id": job_id, "execution_id": f"exec-{job_id}"} + + def fire_claimed(self, job, *, adapters=None, loop=None): + self.fired.append(job["id"]) return True @@ -58,7 +63,10 @@ async def test_valid_fire_reservation_blocks_drain_before_body_and_task(adapter, release_fire = threading.Event() class BlockingProvider: - def fire_due(self, job_id, *, adapters=None, loop=None): + def claim_fire(self, job_id): + return {"id": job_id, "execution_id": "exec-1"} + + def fire_claimed(self, job, *, adapters=None, loop=None): fired.set() release_fire.wait(timeout=2) return True @@ -104,6 +112,78 @@ async def delayed_json(request): assert adapter.active_agent_work_count() == 0 +@pytest.mark.asyncio +async def test_admission_failure_is_retryable_and_never_dispatches(adapter, monkeypatch): + class FailingProvider(_SpyProvider): + def claim_fire(self, job_id): + raise OSError("ledger unavailable") + + provider = FailingProvider() + monkeypatch.setattr("cron.scheduler_provider.resolve_cron_scheduler", lambda: provider) + monkeypatch.setattr( + "plugins.cron_providers.chronos.verify.get_fire_verifier", + lambda: (lambda **kw: {"purpose": "cron_fire"}), + ) + + app = _create_app(adapter) + async with TestClient(TestServer(app)) as cli: + response = await cli.post( + "/api/cron/fire", + headers={"Authorization": "Bearer good"}, + json={"job_id": "abc123"}, + ) + + assert response.status == 503 + assert provider.fired == [] + assert adapter.active_agent_work_count() == 0 + + +@pytest.mark.asyncio +async def test_accepted_response_waits_for_durable_admission(adapter, monkeypatch): + claim_started = threading.Event() + release_claim = threading.Event() + + class BlockingAdmissionProvider(_SpyProvider): + def claim_fire(self, job_id): + claim_started.set() + release_claim.wait(timeout=2) + return super().claim_fire(job_id) + + provider = BlockingAdmissionProvider() + monkeypatch.setattr("cron.scheduler_provider.resolve_cron_scheduler", lambda: provider) + monkeypatch.setattr( + "plugins.cron_providers.chronos.verify.get_fire_verifier", + lambda: (lambda **kw: {"purpose": "cron_fire"}), + ) + + app = _create_app(adapter) + async with TestClient(TestServer(app)) as cli: + request_task = asyncio.create_task( + cli.post( + "/api/cron/fire", + headers={"Authorization": "Bearer good"}, + json={"job_id": "abc123"}, + ) + ) + assert await asyncio.to_thread(claim_started.wait, 2) + await asyncio.sleep(0) + assert not request_task.done() + + release_claim.set() + response = await request_task + # The 202 guarantees durable ADMISSION only — the fire itself runs as + # tracked background work, so wait for it to actually land (fast + # locally, but CI scheduling can lose this race). + for _ in range(200): + if provider.fired: + break + await asyncio.sleep(0.01) + + assert response.status == 202 + assert provider.claimed == ["abc123"] + assert provider.fired == ["abc123"] + + @pytest.mark.asyncio async def test_missing_job_id_400(adapter, monkeypatch): """Valid token but no job_id → 400, no fire.""" diff --git a/tests/hermes_cli/test_cron_fire_dashboard.py b/tests/hermes_cli/test_cron_fire_dashboard.py index 9ec70d4ade6cb..bfbadfce91437 100644 --- a/tests/hermes_cli/test_cron_fire_dashboard.py +++ b/tests/hermes_cli/test_cron_fire_dashboard.py @@ -9,8 +9,9 @@ the JWT verifier runs, - reject a bad/missing NAS-JWT with 401 (the JWT is the real gate), - 400 on missing job_id, - - on a valid token, resolve the job's profile and run fire_due in the - background, returning 202. + - on a valid token, FORWARD the fire to the gateway api_server (which owns + cron execution and the live delivery adapters) and pass its response + through — 503 when the gateway is unreachable so NAS retries. """ import pytest diff --git a/tests/hermes_cli/test_web_server_cron_profiles.py b/tests/hermes_cli/test_web_server_cron_profiles.py index 8e49f4a294b90..adc3ddafcd309 100644 --- a/tests/hermes_cli/test_web_server_cron_profiles.py +++ b/tests/hermes_cli/test_web_server_cron_profiles.py @@ -167,6 +167,175 @@ def fail_create(*args, **kwargs): assert "private callback URL and token" not in str(exc_info.value.detail) +def test_notify_cron_provider_scopes_store_and_runtime_home_together( + isolated_profiles, + monkeypatch, +): + """Provider reconciliation must observe the mutated profile, not default.""" + from cron import jobs as cron_jobs + from cron import scheduler + from hermes_cli import web_server + + from hermes_constants import ( + reset_hermes_home_override, + set_hermes_home_override, + ) + + default_home = isolated_profiles["default"] + worker_home = isolated_profiles["worker_alpha"] + monkeypatch.setattr(scheduler, "_hermes_home", None) + monkeypatch.setattr( + web_server, + "_cron_profile_dicts", + lambda: [{"name": "worker_alpha"}], + ) + captured = {} + + class RecordingProvider: + def on_jobs_changed(self): + captured["runtime_home"] = scheduler._get_hermes_home() + captured["jobs_file"] = cron_jobs._current_cron_store().jobs_file + + monkeypatch.setattr( + "cron.scheduler_provider.resolve_cron_scheduler", + lambda: RecordingProvider(), + ) + + outer_token = set_hermes_home_override(default_home) + try: + web_server._notify_cron_provider_for_profile("worker_alpha") + assert captured == { + "runtime_home": worker_home, + "jobs_file": worker_home / "cron" / "jobs.json", + } + assert scheduler._get_hermes_home() == default_home + finally: + reset_hermes_home_override(outer_token) + + +def test_notify_cron_provider_failure_is_best_effort( + isolated_profiles, + monkeypatch, +): + from hermes_cli import web_server + + class FailNotifyProvider: + @property + def name(self): + return "fail-notify" + + def register_job(self, job): + return None + + def on_jobs_changed(self): + raise RuntimeError("provider unavailable") + + monkeypatch.setattr( + "cron.scheduler_provider.resolve_cron_scheduler", + lambda: FailNotifyProvider(), + ) + + created = web_server._mutate_cron_for_profile( + "worker_alpha", + "create_job", + prompt="survives provider failure", + schedule="every 1h", + name="best-effort-notify", + ) + + assert created["profile"] == "worker_alpha" + assert created["name"] == "best-effort-notify" + + +def test_external_provider_reconcile_fails_closed_with_multiple_profiles( + isolated_profiles, + monkeypatch, +): + """Multi-profile dashboard + external provider: the unscoped reconcile + must NOT run — its orphan cleanup would disarm the other profiles' + armed one-shots in the shared NAS registry. The mutation itself still + succeeds (fail-closed only skips the remote converge).""" + from cron import scheduler + from hermes_cli import web_server + + monkeypatch.setattr(scheduler, "_hermes_home", None) + monkeypatch.setattr( + web_server, + "_cron_profile_dicts", + lambda: [{"name": "default"}, {"name": "worker_alpha"}], + ) + notified = [] + + class ExternalProvider: + @property + def name(self): + return "chronos" + + def register_job(self, job): + return None + + def on_jobs_changed(self): + notified.append(True) + + monkeypatch.setattr( + "cron.scheduler_provider.resolve_cron_scheduler", + lambda: ExternalProvider(), + ) + + created = web_server._mutate_cron_for_profile( + "worker_alpha", + "create_job", + prompt="must not disarm siblings", + schedule="every 1h", + name="multi-profile-guard", + ) + + assert created["profile"] == "worker_alpha" + assert notified == [], ( + "external provider reconcile must stay fail-closed on a " + "multi-profile dashboard" + ) + + +def test_builtin_provider_hook_still_fires_with_multiple_profiles( + isolated_profiles, + monkeypatch, +): + """The built-in provider re-reads jobs.json per tick — its hook is a + safe no-op and must NOT be blocked by the multi-profile guard.""" + from cron import scheduler + from cron.scheduler_provider import InProcessCronScheduler + from hermes_cli import web_server + + monkeypatch.setattr(scheduler, "_hermes_home", None) + monkeypatch.setattr( + web_server, + "_cron_profile_dicts", + lambda: [{"name": "default"}, {"name": "worker_alpha"}], + ) + notified = [] + + class BuiltinProbe(InProcessCronScheduler): + def on_jobs_changed(self): + notified.append(True) + + monkeypatch.setattr( + "cron.scheduler_provider.resolve_cron_scheduler", + lambda: BuiltinProbe(), + ) + + created = web_server._mutate_cron_for_profile( + "worker_alpha", + "create_job", + prompt="builtin notify", + schedule="every 1h", + name="builtin-notify", + ) + + assert created["profile"] == "worker_alpha" + assert notified == [True] + + def test_profile_call_cannot_retarget_ticker_store_mid_write( isolated_profiles, monkeypatch, @@ -281,6 +450,482 @@ async def test_cron_mutation_without_profile_finds_named_profile_job(isolated_pr assert worker_jobs[0]["enabled"] is False +@pytest.mark.asyncio +async def test_dashboard_cron_mutations_notify_selected_profile_provider( + isolated_profiles, + monkeypatch, +): + from hermes_cli import web_server + + notified_profiles = [] + monkeypatch.setattr( + web_server, + "_notify_cron_provider_for_profile", + notified_profiles.append, + ) + + created = await web_server.create_cron_job( + web_server.CronJobCreate( + prompt="managed by named profile", + schedule="every 1h", + name="provider-notify-job", + ), + profile="worker_alpha", + ) + await web_server.update_cron_job( + created["id"], + web_server.CronJobUpdate(updates={"name": "provider-notify-job-updated"}), + profile="worker_alpha", + ) + await web_server.pause_cron_job(created["id"], profile="worker_alpha") + await web_server.resume_cron_job(created["id"], profile="worker_alpha") + await web_server.delete_cron_job(created["id"], profile="worker_alpha") + + assert notified_profiles == ["worker_alpha"] * 5 + + +@pytest.mark.asyncio +async def test_blueprint_instantiation_notifies_selected_profile_provider( + isolated_profiles, + monkeypatch, +): + from hermes_cli import web_server + + notified_profiles = [] + monkeypatch.setattr( + web_server, + "_notify_cron_provider_for_profile", + notified_profiles.append, + ) + + created = await web_server.instantiate_blueprint( + web_server.AutomationBlueprintInstantiate( + blueprint="morning-brief", + values={"time": "07:30", "deliver": "local"}, + ), + profile="worker_alpha", + ) + + assert created["profile"] == "worker_alpha" + assert notified_profiles == ["worker_alpha"] + + +@pytest.mark.asyncio +async def test_trigger_cron_job_fires_only_selected_job_and_returns_refreshed_state( + isolated_profiles, + monkeypatch, +): + from cron import jobs as cron_jobs + from hermes_cli import web_server + + selected = web_server._call_cron_for_profile( + "worker_alpha", + "create_job", + prompt="run immediately", + schedule="every 1h", + name="selected-trigger-job", + ) + sibling = web_server._call_cron_for_profile( + "worker_alpha", + "create_job", + prompt="leave scheduled", + schedule="every 1h", + name="sibling-job", + ) + fired = [] + + class RecordingProvider: + def fire_due(self, job_id, *, adapters=None, loop=None, force=False): + fired.append( + { + "job_id": job_id, + "jobs_file": cron_jobs._current_cron_store().jobs_file, + "force": force, + } + ) + cron_jobs.mark_job_run(job_id, success=True) + return True + + monkeypatch.setattr( + "cron.scheduler_provider.resolve_cron_scheduler", + lambda: RecordingProvider(), + ) + monkeypatch.setattr( + cron_jobs, + "trigger_job", + lambda _job_id: (_ for _ in ()).throw( + AssertionError("manual fire must not expose the job to the ticker first") + ), + ) + + triggered = await web_server.trigger_cron_job( + selected["id"], + profile="worker_alpha", + ) + + assert fired == [ + { + "job_id": selected["id"], + "jobs_file": isolated_profiles["worker_alpha"] / "cron" / "jobs.json", + "force": False, + } + ] + assert triggered["last_status"] == "ok" + assert triggered["last_run_at"] is not None + untouched = web_server._call_cron_for_profile( + "worker_alpha", + "get_job", + sibling["id"], + ) + assert untouched["last_run_at"] is None + + +@pytest.mark.asyncio +async def test_trigger_cron_job_reports_lost_claim_as_conflict( + isolated_profiles, + monkeypatch, +): + from hermes_cli import web_server + + job = web_server._call_cron_for_profile( + "worker_alpha", + "create_job", + prompt="already running", + schedule="every 1h", + name="claimed-trigger-job", + ) + + class ClaimLostProvider: + def fire_due(self, job_id, *, adapters=None, loop=None, force=False): + return False + + monkeypatch.setattr( + "cron.scheduler_provider.resolve_cron_scheduler", + lambda: ClaimLostProvider(), + ) + + with pytest.raises(HTTPException) as exc: + await web_server.trigger_cron_job(job["id"], profile="worker_alpha") + + assert exc.value.status_code == 409 + assert "already running" in exc.value.detail + + +@pytest.mark.asyncio +async def test_trigger_cron_job_forces_paused_job_atomically( + isolated_profiles, + monkeypatch, +): + from cron import jobs as cron_jobs + from hermes_cli import web_server + + job = web_server._call_cron_for_profile( + "worker_alpha", + "create_job", + prompt="resume me", + schedule="every 1h", + name="paused-trigger-job", + ) + web_server._call_cron_for_profile("worker_alpha", "pause_job", job["id"]) + observed = {} + + class ForceProvider: + def fire_due(self, job_id, *, adapters=None, loop=None, force=False): + observed["force"] = force + assert cron_jobs.claim_job_for_fire(job_id, force=force) is True + cron_jobs.mark_job_run(job_id, success=True) + return True + + monkeypatch.setattr( + "cron.scheduler_provider.resolve_cron_scheduler", + lambda: ForceProvider(), + ) + + triggered = await web_server.trigger_cron_job( + job["id"], + profile="worker_alpha", + ) + + assert observed["force"] is True + assert triggered["enabled"] is True + assert triggered["state"] == "scheduled" + assert triggered["last_status"] == "ok" + + +@pytest.mark.asyncio +async def test_trigger_paused_job_rejects_legacy_provider_without_mutating_job( + isolated_profiles, + monkeypatch, +): + from fastapi import HTTPException + from hermes_cli import web_server + + job = web_server._call_cron_for_profile( + "worker_alpha", + "create_job", + prompt="stay paused", + schedule="every 1h", + name="legacy-paused-trigger-job", + ) + web_server._call_cron_for_profile("worker_alpha", "pause_job", job["id"]) + calls = [] + + class LegacyProvider: + def fire_due(self, job_id, *, adapters=None, loop=None): + calls.append(job_id) + return True + + monkeypatch.setattr( + "cron.scheduler_provider.resolve_cron_scheduler", + lambda: LegacyProvider(), + ) + + with pytest.raises(HTTPException) as exc: + await web_server.trigger_cron_job(job["id"], profile="worker_alpha") + + assert exc.value.status_code == 409 + assert "forced" in exc.value.detail.lower() + assert calls == [] + persisted = web_server._call_cron_for_profile( + "worker_alpha", + "get_job", + job["id"], + ) + assert persisted["state"] == "paused" + assert persisted["enabled"] is False + + +@pytest.mark.asyncio +async def test_trigger_cron_job_returns_refreshed_execution_failure( + isolated_profiles, + monkeypatch, +): + from cron import jobs as cron_jobs + from hermes_cli import web_server + + job = web_server._call_cron_for_profile( + "worker_alpha", + "create_job", + prompt="fail visibly", + schedule="every 1h", + name="failed-trigger-job", + ) + + class FailedProvider: + def fire_due(self, job_id, *, adapters=None, loop=None, force=False): + cron_jobs.mark_job_run(job_id, success=False, error="expected failure") + return False + + monkeypatch.setattr( + "cron.scheduler_provider.resolve_cron_scheduler", + lambda: FailedProvider(), + ) + + triggered = await web_server.trigger_cron_job( + job["id"], + profile="worker_alpha", + ) + + assert triggered["last_status"] == "error" + assert triggered["last_error"] == "expected failure" + + +@pytest.mark.asyncio +async def test_trigger_cron_job_returns_completed_snapshot_for_retained_oneshot( + isolated_profiles, + monkeypatch, +): + from cron import jobs as cron_jobs + from hermes_cli import web_server + + job = web_server._call_cron_for_profile( + "worker_alpha", + "create_job", + prompt="run once", + schedule="30m", + name="completed-trigger-job", + ) + + class SuccessfulProvider: + def fire_due(self, job_id, *, adapters=None, loop=None, force=False): + cron_jobs.mark_job_run(job_id, success=True) + return True + + monkeypatch.setattr( + "cron.scheduler_provider.resolve_cron_scheduler", + lambda: SuccessfulProvider(), + ) + + triggered = await web_server.trigger_cron_job( + job["id"], + profile="worker_alpha", + ) + + assert triggered["state"] == "completed" + assert triggered["enabled"] is False + # Completed one-shots are retained for the retention window (#80624) with + # their terminal status inspectable — the trigger response is the real + # record, not a synthetic pre-removal snapshot. + assert triggered["last_status"] == "ok" + assert triggered["last_run_at"] is not None + retained = web_server._call_cron_for_profile( + "worker_alpha", + "get_job", + job["id"], + ) + assert retained is not None + assert retained["state"] == "completed" + + +@pytest.mark.asyncio +async def test_cron_profile_scan_runs_off_event_loop(isolated_profiles, monkeypatch): + from hermes_cli import web_server + + worker_job = web_server._call_cron_for_profile( + "worker_alpha", + "create_job", + prompt="managed by named profile", + schedule="every 1h", + name="thread-offload-job", + ) + + event_loop_thread = threading.get_ident() + profile_scan_threads = SimpleQueue() + worker_threads = SimpleQueue() + original_profile_dicts = web_server._cron_profile_dicts + original_find = web_server._find_cron_job_profile + + def tracking_profile_dicts(): + profile_scan_threads.put(threading.get_ident()) + return original_profile_dicts() + + def tracking_find(job_id): + worker_threads.put(threading.get_ident()) + return original_find(job_id) + + monkeypatch.setattr(web_server, "_cron_profile_dicts", tracking_profile_dicts) + monkeypatch.setattr(web_server, "_find_cron_job_profile", tracking_find) + + jobs = await web_server.list_cron_jobs(profile="all") + paused = await web_server.pause_cron_job(worker_job["id"]) + + assert any(job["id"] == worker_job["id"] for job in jobs) + assert paused["profile"] == "worker_alpha" + profile_scan_thread_ids = _drain_queue(profile_scan_threads) + worker_thread_ids = _drain_queue(worker_threads) + assert profile_scan_thread_ids + assert worker_thread_ids + assert all(thread_id != event_loop_thread for thread_id in profile_scan_thread_ids) + assert all(thread_id != event_loop_thread for thread_id in worker_thread_ids) + + +@pytest.mark.asyncio +async def test_cron_dashboard_io_rejects_async_callables(): + from hermes_cli import web_server + + async def async_callable(): + return "nope" + + with pytest.raises(TypeError, match="only accepts sync callables"): + await web_server._run_cron_dashboard_io(async_callable) + + + +@pytest.mark.asyncio +async def test_update_cron_job_normalizes_dashboard_core_fields(isolated_profiles, tmp_path): + from hermes_cli import web_server + + scripts_dir = isolated_profiles["worker_alpha"] / "scripts" + scripts_dir.mkdir() + (scripts_dir / "collect.py").write_text("print('ok')\n", encoding="utf-8") + job = web_server._call_cron_for_profile( + "worker_alpha", + "create_job", + prompt="managed by named profile", + schedule="every 1h", + name="normalizes-dashboard-fields", + ) + + updated = await web_server.update_cron_job( + job["id"], + web_server.CronJobUpdate( + updates={ + "base_url": "https://example.invalid/v1/", + "script": str(scripts_dir / "collect.py"), + "context_from": "", + "no_agent": True, + } + ), + profile="worker_alpha", + ) + + assert updated["base_url"] == "https://example.invalid/v1" + assert updated["script"] == "collect.py" + assert updated["context_from"] is None + assert updated["no_agent"] is True + + +@pytest.mark.asyncio +async def test_create_cron_job_rejects_script_outside_profile_scripts( + isolated_profiles, tmp_path +): + from hermes_cli import web_server + + outside = tmp_path / "outside.py" + outside.write_text("print('nope')\n", encoding="utf-8") + + with pytest.raises(HTTPException) as exc: + await web_server.create_cron_job( + web_server.CronJobCreate( + schedule="every 1h", + script=str(outside), + no_agent=True, + ), + profile="worker_alpha", + ) + + assert exc.value.status_code == 400 + assert "inside" in exc.value.detail + + +@pytest.mark.asyncio +async def test_create_cron_job_rejects_empty_agent_job(isolated_profiles): + from hermes_cli import web_server + + with pytest.raises(HTTPException) as exc: + await web_server.create_cron_job( + web_server.CronJobCreate(schedule="every 1h"), + profile="worker_alpha", + ) + + assert exc.value.status_code == 400 + assert "prompt, skill, or script" in exc.value.detail + + +@pytest.mark.asyncio +async def test_update_cron_job_no_agent_reuses_existing_script(isolated_profiles): + from hermes_cli import web_server + + scripts_dir = isolated_profiles["worker_alpha"] / "scripts" + scripts_dir.mkdir() + (scripts_dir / "collect.py").write_text("print('ok')\n", encoding="utf-8") + + job = await web_server.create_cron_job( + web_server.CronJobCreate( + schedule="every 1h", + script=str(scripts_dir / "collect.py"), + ), + profile="worker_alpha", + ) + + updated = await web_server.update_cron_job( + job["id"], + web_server.CronJobUpdate(updates={"no_agent": True}), + profile="worker_alpha", + ) + + assert updated["no_agent"] is True + assert updated["script"] == "collect.py" @pytest.mark.asyncio @@ -327,3 +972,218 @@ async def test_dashboard_cron_rejects_missing_context_from(isolated_profiles): +@pytest.mark.asyncio +async def test_dashboard_cron_noop_inference_fields_keep_existing_snapshots( + isolated_profiles, + monkeypatch, +): + from hermes_cli import runtime_provider, web_server + + current_provider = {"name": "initial-provider"} + monkeypatch.setattr( + runtime_provider, + "resolve_runtime_provider", + lambda **kwargs: {"provider": current_provider["name"]}, + ) + + job = web_server._call_cron_for_profile( + "worker_alpha", + "create_job", + prompt="managed by named profile", + schedule="every 1h", + name="dashboard-edit-job", + ) + + assert job["provider_snapshot"] == "initial-provider" + assert job["model_snapshot"] == "test-model" + + current_provider["name"] = "changed-provider" + (isolated_profiles["worker_alpha"] / "config.yaml").write_text( + "model: changed-model\n", + encoding="utf-8", + ) + + updated = await web_server.update_cron_job( + job["id"], + web_server.CronJobUpdate( + updates={ + "name": "dashboard-edit-job-renamed", + "provider": None, + "model": None, + "base_url": None, + "no_agent": False, + } + ), + profile="worker_alpha", + ) + + assert updated["name"] == "dashboard-edit-job-renamed" + assert updated["provider_snapshot"] == "initial-provider" + assert updated["model_snapshot"] == "test-model" + + +@pytest.mark.asyncio +async def test_update_cron_job_clears_snapshots_for_no_agent( + isolated_profiles, + monkeypatch, +): + from hermes_cli import runtime_provider, web_server + + monkeypatch.setattr( + runtime_provider, + "resolve_runtime_provider", + lambda **kwargs: {"provider": "worker-provider"}, + ) + scripts_dir = isolated_profiles["worker_alpha"] / "scripts" + scripts_dir.mkdir() + (scripts_dir / "collect.py").write_text("print('ok')\n", encoding="utf-8") + + job = web_server._call_cron_for_profile( + "worker_alpha", + "create_job", + prompt="managed by named profile", + schedule="every 1h", + name="agent-to-script-job", + ) + + assert job["provider_snapshot"] == "worker-provider" + assert job["model_snapshot"] == "test-model" + + updated = await web_server.update_cron_job( + job["id"], + web_server.CronJobUpdate( + updates={ + "script": str(scripts_dir / "collect.py"), + "no_agent": True, + } + ), + profile="worker_alpha", + ) + + assert updated["provider_snapshot"] is None + assert updated["model_snapshot"] is None + + +@pytest.mark.asyncio +async def test_update_cron_job_rejects_id_mutation(isolated_profiles, monkeypatch): + """Dashboard surfaces a 400 (not a 500 or silent rename) when an + id-mutation attempt is rejected by cron/jobs.update_job.""" + from hermes_cli import web_server + + notified_profiles = [] + monkeypatch.setattr( + web_server, + "_notify_cron_provider_for_profile", + notified_profiles.append, + ) + worker_job = web_server._call_cron_for_profile( + "worker_alpha", + "create_job", + prompt="managed by named profile", + schedule="every 1h", + name="immutable-id-job", + ) + + with pytest.raises(HTTPException) as exc: + await web_server.update_cron_job( + worker_job["id"], + web_server.CronJobUpdate(updates={"id": "../escape"}), + profile="worker_alpha", + ) + + assert exc.value.status_code == 400 + assert "id" in exc.value.detail + assert notified_profiles == [] + worker_jobs = await web_server.list_cron_jobs(profile="worker_alpha") + assert [job["id"] for job in worker_jobs] == [worker_job["id"]] + + +@pytest.mark.asyncio +async def test_cron_delete_with_profile_deletes_only_target_profile(isolated_profiles): + from hermes_cli import web_server + + default_job = web_server._call_cron_for_profile( + "default", + "create_job", + prompt="same-ish default", + schedule="every 1h", + name="shared-name", + ) + worker_job = web_server._call_cron_for_profile( + "worker_alpha", + "create_job", + prompt="same-ish worker", + schedule="every 1h", + name="shared-name-worker", + ) + + deleted = await web_server.delete_cron_job(worker_job["id"], profile="worker_alpha") + assert deleted == {"ok": True} + + remaining_default = await web_server.list_cron_jobs(profile="default") + remaining_worker = await web_server.list_cron_jobs(profile="worker_alpha") + assert [job["id"] for job in remaining_default] == [default_job["id"]] + assert remaining_worker == [] + + +@pytest.mark.asyncio +async def test_cron_profile_validation_errors(isolated_profiles): + from hermes_cli import web_server + + with pytest.raises(HTTPException) as bad_name: + await web_server.list_cron_jobs(profile="../bad") + assert bad_name.value.status_code == 400 + + with pytest.raises(HTTPException) as missing: + await web_server.list_cron_jobs(profile="missing_profile") + assert missing.value.status_code == 404 + + +@pytest.mark.asyncio +async def test_create_cron_job_without_profile_uses_backend_own_profile( + isolated_profiles, monkeypatch +): + """A pool backend scoped to a named profile must not default creates to + ``~/.hermes`` when the request carries no explicit ``profile`` (the + Desktop app's pre-profileScoped clients sent none).""" + from hermes_cli import web_server + + monkeypatch.setenv( + "HERMES_HOME", str(isolated_profiles["worker_alpha"]) + ) + + job = await web_server.create_cron_job( + web_server.CronJobCreate( + prompt="runs in my own profile", + schedule="every 1h", + name="own-profile-job", + ), + profile=None, + ) + + assert job["profile"] == "worker_alpha" + assert (isolated_profiles["worker_alpha"] / "cron" / "jobs.json").exists() + assert not (isolated_profiles["default"] / "cron" / "jobs.json").exists() + + +@pytest.mark.asyncio +async def test_create_cron_job_without_profile_defaults_when_unscoped( + isolated_profiles, monkeypatch +): + """HERMES_HOME at the default home (or unrecognized) keeps the legacy + ``default`` fallback.""" + from hermes_cli import web_server + + monkeypatch.setenv("HERMES_HOME", str(isolated_profiles["default"])) + + job = await web_server.create_cron_job( + web_server.CronJobCreate( + prompt="runs in default", + schedule="every 1h", + name="default-job", + ), + profile=None, + ) + + assert job["profile"] == "default" + assert (isolated_profiles["default"] / "cron" / "jobs.json").exists() diff --git a/tests/plugins/test_chronos_cron.py b/tests/plugins/test_chronos_cron.py index bf0cdca797609..75971ea2d309e 100644 --- a/tests/plugins/test_chronos_cron.py +++ b/tests/plugins/test_chronos_cron.py @@ -118,9 +118,16 @@ def test_reconcile_arms_all_enabled(temp_home, chronos, monkeypatch): def test_fire_due_rearms_next_oneshot(chronos, monkeypatch): prov, fake = chronos - # super().fire_due runs the job; stub the ABC default to "ran". - monkeypatch.setattr("cron.scheduler_provider.CronScheduler.fire_due", - lambda self, jid, **kw: True) + # Keep the two-phase provider flow intact while stubbing durable admission + # and the shared runner body. + monkeypatch.setattr( + "cron.scheduler_provider.CronScheduler.claim_fire", + lambda self, jid, **kw: {"id": jid, "execution_id": "exec-1"}, + ) + monkeypatch.setattr( + "cron.scheduler_provider.CronScheduler.fire_claimed", + lambda self, job, **kw: True, + ) monkeypatch.setattr("cron.jobs.get_job", lambda jid: {"id": jid, "enabled": True, "next_run_at": "2026-06-18T12:05:00+00:00"}) @@ -128,3 +135,105 @@ def test_fire_due_rearms_next_oneshot(chronos, monkeypatch): assert [p["job_id"] for p in fake.provisions] == ["j1"] assert fake.provisions[0]["fire_at"] == "2026-06-18T12:05:00+00:00" + +def test_fire_due_rearms_after_claimed_job_failure(chronos, monkeypatch): + """A claimed attempt is consumed even when the job pipeline reports failure.""" + prov, fake = chronos + claimed = {"id": "j1", "fire_claim": {"by": "owner-1"}} + persisted = { + "id": "j1", + "enabled": True, + "next_run_at": "2026-06-18T12:05:00+00:00", + } + + monkeypatch.setattr("cron.jobs.claim_job_for_fire", lambda jid, **kw: claimed) + monkeypatch.setattr( + "cron.executions.create_execution", + lambda jid, source: {"id": "exec-1"}, + ) + monkeypatch.setattr("cron.scheduler.run_one_job", lambda *args, **kwargs: False) + monkeypatch.setattr("cron.jobs.get_job", lambda jid: persisted) + + assert prov.fire_due("j1") is True + assert [provision["job_id"] for provision in fake.provisions] == ["j1"] + + +def test_fire_due_forwards_manual_force_to_claim(chronos, monkeypatch): + """A manual force fire must reach the store claim as force=True.""" + prov, _fake = chronos + seen = [] + monkeypatch.setattr( + "cron.jobs.claim_job_for_fire", + lambda jid, **kw: seen.append(kw) or False, + ) + monkeypatch.setattr( + "cron.executions.create_execution", + lambda jid, source: {"id": "exec-1"}, + ) + + assert prov.fire_due("j1", force=True) is False + assert seen == [{"return_job": True, "force": True}] + + +def test_fire_due_no_rearm_when_job_gone(chronos, monkeypatch): + """repeat-N exhausted / one-shot completed → mark_job_run deleted the job → + get_job None → no re-arm (the schedule stops cleanly).""" + prov, fake = chronos + monkeypatch.setattr("cron.scheduler_provider.CronScheduler.fire_due", + lambda self, jid, **kw: True) + monkeypatch.setattr("cron.jobs.get_job", lambda jid: None) + + assert prov.fire_due("j1") is True + assert fake.provisions == [] + + +def test_fire_due_no_rearm_when_claim_lost(chronos, monkeypatch): + """If the run didn't happen (claim lost), don't re-arm.""" + prov, fake = chronos + monkeypatch.setattr("cron.scheduler_provider.CronScheduler.fire_due", + lambda self, jid, **kw: False) + + assert prov.fire_due("j1") is False + assert fake.provisions == [] + + +# -- provider capability classification ---------------------------------------- + +def test_chronos_is_split_fire_capable(chronos): + """Regression: Chronos must be classified as a split-aware provider so the + fire webhook uses durable claim admission (not the legacy fire_due path). + Chronos deliberately has NO fire_due override — its re-arm logic lives in + fire_claimed, which the split path invokes.""" + from cron.scheduler_provider import ( + provider_supports_fire_cancel, + provider_supports_force_fire, + provider_supports_split_fire, + ) + + prov, _fake = chronos + assert provider_supports_split_fire(prov) is True + assert provider_supports_force_fire(prov) is True + assert provider_supports_fire_cancel(prov) is True + + +def test_fire_claimed_no_rearm_when_run_failed(chronos, monkeypatch): + prov, fake = chronos + monkeypatch.setattr( + "cron.scheduler_provider.CronScheduler.fire_claimed", + lambda self, job, **kw: False, + ) + + assert prov.fire_claimed({"id": "j1"}) is False + assert fake.provisions == [] + + +def test_fire_claimed_no_rearm_when_job_gone(chronos, monkeypatch): + prov, fake = chronos + monkeypatch.setattr( + "cron.scheduler_provider.CronScheduler.fire_claimed", + lambda self, job, **kw: True, + ) + monkeypatch.setattr("cron.jobs.get_job", lambda jid: None) + + assert prov.fire_claimed({"id": "j1"}) is True + assert fake.provisions == [] diff --git a/tests/tools/test_cronjob_run_background.py b/tests/tools/test_cronjob_run_background.py index d33858c8b6615..a35d4c4a8fbca 100644 --- a/tests/tools/test_cronjob_run_background.py +++ b/tests/tools/test_cronjob_run_background.py @@ -67,7 +67,7 @@ def slow_run_one_job(job, **kw): return True with _bound_session_key(): - with patch("tools.cronjob_tools.claim_job_for_fire", return_value=True) as m_claim, \ + with patch("tools.cronjob_tools.claim_job_for_fire", side_effect=lambda jid, **kw: {**_job(jid), "fire_claim": {"by": "bg-owner"}}) as m_claim, \ patch("cron.scheduler.run_one_job", side_effect=slow_run_one_job), \ patch("tools.cronjob_tools.get_job", return_value={"last_status": "ok", "last_error": None}): @@ -79,7 +79,7 @@ def slow_run_one_job(job, **kw): assert res["claimed"] is True assert res["dispatched"] is True assert res["delegation_id"] - m_claim.assert_called_once_with("job-bg-01") + m_claim.assert_called_once_with("job-bg-01", return_job=True) # The job actually starts on the daemon executor. assert run_started.wait(timeout=5.0), "job never started in background" finally: @@ -95,7 +95,7 @@ def test_completion_event_reaches_shared_queue(self): # The runner executes on a daemon thread — the patches must stay # active until the completion event lands, so poll INSIDE the blocks. with _bound_session_key("agent:main:telegram:dm:777"): - with patch("tools.cronjob_tools.claim_job_for_fire", return_value=True), \ + with patch("tools.cronjob_tools.claim_job_for_fire", side_effect=lambda jid, **kw: {**_job(jid), "fire_claim": {"by": "bg-owner"}}), \ patch("cron.scheduler.run_one_job", return_value=True), \ patch("tools.cronjob_tools.get_job", return_value={"last_status": "ok", "last_error": None, @@ -128,7 +128,7 @@ def test_failed_run_reports_error_status_in_event(self): from tools.process_registry import process_registry with _bound_session_key("agent:main:telegram:dm:778"): - with patch("tools.cronjob_tools.claim_job_for_fire", return_value=True), \ + with patch("tools.cronjob_tools.claim_job_for_fire", side_effect=lambda jid, **kw: {**_job(jid), "fire_claim": {"by": "bg-owner"}}), \ patch("cron.scheduler.run_one_job", return_value=True), \ patch("tools.cronjob_tools.get_job", return_value={"last_status": "error", @@ -183,7 +183,7 @@ def test_async_delivery_unsupported_falls_back_to_sync(self): def test_pool_at_capacity_runs_inline(self): """A rejected dispatch must not strand the already-taken claim.""" with _bound_session_key(): - with patch("tools.cronjob_tools.claim_job_for_fire", return_value=True), \ + with patch("tools.cronjob_tools.claim_job_for_fire", side_effect=lambda jid, **kw: {**_job(jid), "fire_claim": {"by": "bg-owner"}}), \ patch("tools.async_delegation.dispatch_async_delegation", return_value={"status": "rejected", "error": "capacity"}), \ patch("cron.scheduler.run_one_job", return_value=True) as m_run, \ @@ -278,7 +278,7 @@ def test_run_action_returns_background_note(self): """cronjob(action='run') surfaces the handle + do-not-wait note.""" with _bound_session_key(): with patch("tools.cronjob_tools.resolve_job_ref", return_value=_job('job-bg-12')), \ - patch("tools.cronjob_tools.claim_job_for_fire", return_value=True), \ + patch("tools.cronjob_tools.claim_job_for_fire", side_effect=lambda jid, **kw: {**_job(jid), "fire_claim": {"by": "bg-owner"}}), \ patch("cron.scheduler.run_one_job", return_value=True), \ patch("tools.cronjob_tools.get_job", return_value={"id": "job-bg-12", "name": "bg run", @@ -296,7 +296,7 @@ def test_run_action_sync_path_unchanged_without_session(self): execution_success populated from the completed run).""" ran = {"job": "after-run", "last_status": "ok", "last_error": None} with patch("tools.cronjob_tools.resolve_job_ref", return_value=_job('job-bg-13')), \ - patch("tools.cronjob_tools.claim_job_for_fire", return_value=True) as m_claim, \ + patch("tools.cronjob_tools.claim_job_for_fire", side_effect=lambda jid, **kw: {**_job(jid), "fire_claim": {"by": "bg-owner"}}) as m_claim, \ patch("cron.scheduler.run_one_job", return_value=True) as m_run, \ patch("tools.cronjob_tools.get_job", return_value=ran): out = json.loads(cronjob(action="run", job_id="job-bg-13")) @@ -304,5 +304,5 @@ def test_run_action_sync_path_unchanged_without_session(self): assert out["success"] is True assert out["job"]["executed"] is True assert out["job"]["execution_success"] is True - m_claim.assert_called_once_with("job-bg-13") + m_claim.assert_called_once_with("job-bg-13", return_job=True) m_run.assert_called_once() diff --git a/tests/tools/test_cronjob_run_immediate.py b/tests/tools/test_cronjob_run_immediate.py index ad24d02f257bd..beb8c098c3cec 100644 --- a/tests/tools/test_cronjob_run_immediate.py +++ b/tests/tools/test_cronjob_run_immediate.py @@ -29,8 +29,9 @@ class TestCronjobRunExecutesImmediately: def test_run_action_claims_and_fires_via_run_one_job(self): """action='run' must claim the job then fire it through run_one_job.""" ran = {"job": "after-run", "last_status": "ok", "last_error": None} + claimed = {**_JOB, "fire_claim": {"by": "manual-owner"}} with patch("tools.cronjob_tools.resolve_job_ref", return_value=dict(_JOB)), \ - patch("tools.cronjob_tools.claim_job_for_fire", return_value=True) as m_claim, \ + patch("tools.cronjob_tools.claim_job_for_fire", return_value=claimed) as m_claim, \ patch("cron.scheduler.run_one_job", return_value=True) as m_run, \ patch("tools.cronjob_tools.get_job", return_value=ran): out = json.loads(cronjob(action="run", job_id="job-run-1")) @@ -38,9 +39,80 @@ def test_run_action_claims_and_fires_via_run_one_job(self): assert out["success"] is True assert out["job"]["executed"] is True assert out["job"]["execution_success"] is True - m_claim.assert_called_once_with("job-run-1") # at-most-once claim taken - m_run.assert_called_once() # fired via the shared body + m_claim.assert_called_once_with("job-run-1", return_job=True) + m_run.assert_called_once_with(claimed, adapters=None, loop=None, extra_prompt=None) + def test_run_reconciles_external_provider_after_claimed_execution(self): + """A direct run must re-arm Chronos after it advances next_run_at. + + Otherwise a scheduled Chronos fire that loses its claim to this direct + run is consumed without a successor one-shot, permanently stalling the + recurring job. + """ + order = [] + ran = {"id": "job-run-1", "last_status": "ok", "last_error": None} + claimed = {**_JOB, "fire_claim": {"by": "manual-owner"}} + with patch("tools.cronjob_tools.resolve_job_ref", return_value=dict(_JOB)), \ + patch("tools.cronjob_tools.claim_job_for_fire", return_value=claimed), \ + patch("cron.scheduler.run_one_job", + side_effect=lambda *a, **kw: order.append("run") or True), \ + patch("tools.cronjob_tools.get_job", return_value=ran), \ + patch("tools.cronjob_tools._notify_provider_jobs_changed_safe", + side_effect=lambda: order.append("notify")) as m_notify: + out = json.loads(cronjob(action="run", job_id="job-run-1")) + + assert out["job"]["executed"] is True + m_notify.assert_called_once_with() + # Reconcile only AFTER the run persisted its final state (mark_job_run + # inside run_one_job), so the provider arms the post-run next_run_at. + assert order == ["run", "notify"] + + def test_run_reconciles_external_provider_even_when_claimed_run_fails(self): + """A claimed direct run advances next_run_at at claim time, so the + provider must be reconciled even when the execution itself fails.""" + failed = {"id": "job-run-1", "last_status": "error", "last_error": "provider 500"} + claimed = {**_JOB, "fire_claim": {"by": "manual-owner"}} + with patch("tools.cronjob_tools.resolve_job_ref", return_value=dict(_JOB)), \ + patch("tools.cronjob_tools.claim_job_for_fire", return_value=claimed), \ + patch("cron.scheduler.run_one_job", side_effect=RuntimeError("boom")), \ + patch("tools.cronjob_tools.mark_job_run"), \ + patch("tools.cronjob_tools.get_job", return_value=failed), \ + patch("tools.cronjob_tools._notify_provider_jobs_changed_safe") as m_notify: + out = json.loads(cronjob(action="run", job_id="job-run-1")) + + assert out["job"]["executed"] is True + assert out["job"]["execution_success"] is False + m_notify.assert_called_once_with() + + def test_run_skips_when_claim_lost(self): + """If the scheduler already holds the fire claim, do NOT double-run.""" + with patch("tools.cronjob_tools.resolve_job_ref", return_value=dict(_JOB)), \ + patch("tools.cronjob_tools.claim_job_for_fire", return_value=False), \ + patch("cron.scheduler.run_one_job") as m_run, \ + patch("tools.cronjob_tools.get_job", return_value=dict(_JOB)), \ + patch("tools.cronjob_tools._notify_provider_jobs_changed_safe") as m_notify: + out = json.loads(cronjob(action="run", job_id="job-run-1")) + + assert out["success"] is True + assert out["job"]["executed"] is False + assert out["job"]["execution_success"] is False + assert "execution_skipped" in out["job"] + m_run.assert_not_called() # claim lost -> never fired + m_notify.assert_not_called() # the winning scheduler owns the re-arm + + def test_run_reports_failure_from_last_status(self): + """A failed run is reported via the re-read job's last_status/last_error.""" + failed = {"id": "job-run-1", "last_status": "error", "last_error": "provider 500"} + claimed = {**_JOB, "fire_claim": {"by": "manual-owner"}} + with patch("tools.cronjob_tools.resolve_job_ref", return_value=dict(_JOB)), \ + patch("tools.cronjob_tools.claim_job_for_fire", return_value=claimed), \ + patch("cron.scheduler.run_one_job", return_value=True), \ + patch("tools.cronjob_tools.get_job", return_value=failed): + out = json.loads(cronjob(action="run", job_id="job-run-1")) + + assert out["job"]["executed"] is True + assert out["job"]["execution_success"] is False + assert out["job"]["execution_error"] == "provider 500" def test_execute_job_now_bails_without_claim(self): """_execute_job_now never calls run_one_job when the claim is lost.""" @@ -58,7 +130,7 @@ def test_execute_job_now_passes_live_gateway_context_to_delivery(self): runner = SimpleNamespace(adapters=adapters, _gateway_loop=gateway_loop) completed = {"id": "job-run-1", "last_status": "ok", "last_error": None} - with patch("tools.cronjob_tools.claim_job_for_fire", return_value=True), \ + with patch("tools.cronjob_tools.claim_job_for_fire", return_value={**_JOB, "fire_claim": {"by": "manual-owner"}}), \ patch("gateway.run._gateway_runner_ref", return_value=runner), \ patch("cron.scheduler.run_one_job", return_value=True) as m_run, \ patch("tools.cronjob_tools.get_job", return_value=completed): @@ -66,7 +138,7 @@ def test_execute_job_now_passes_live_gateway_context_to_delivery(self): assert res["success"] is True m_run.assert_called_once_with( - _JOB, + {**_JOB, "fire_claim": {"by": "manual-owner"}}, adapters=adapters, loop=gateway_loop, extra_prompt=None, @@ -76,18 +148,24 @@ def test_execute_job_now_remains_standalone_without_gateway(self): """CLI-only runs retain the standalone delivery path.""" completed = {"id": "job-run-1", "last_status": "ok", "last_error": None} - with patch("tools.cronjob_tools.claim_job_for_fire", return_value=True), \ + with patch("tools.cronjob_tools.claim_job_for_fire", return_value={**_JOB, "fire_claim": {"by": "manual-owner"}}), \ patch.dict(sys.modules, {"gateway.run": None}), \ patch("cron.scheduler.run_one_job", return_value=True) as m_run, \ patch("tools.cronjob_tools.get_job", return_value=completed): res = _execute_job_now(dict(_JOB)) assert res["success"] is True - m_run.assert_called_once_with(_JOB, adapters=None, loop=None, extra_prompt=None) + m_run.assert_called_once_with( + {**_JOB, "fire_claim": {"by": "manual-owner"}}, + adapters=None, + loop=None, + extra_prompt=None, + ) def test_execute_job_now_marks_failure_on_exception(self): """An exception during fire is captured, marked failed, not propagated.""" - with patch("tools.cronjob_tools.claim_job_for_fire", return_value=True), \ + claimed = {**_JOB, "fire_claim": {"by": "manual-owner"}} + with patch("tools.cronjob_tools.claim_job_for_fire", return_value=claimed), \ patch("cron.scheduler.run_one_job", side_effect=RuntimeError("boom")), \ patch("tools.cronjob_tools.mark_job_run") as m_mark, \ patch("tools.cronjob_tools.get_job", return_value=dict(_JOB)): @@ -95,7 +173,12 @@ def test_execute_job_now_marks_failure_on_exception(self): assert res["claimed"] is True assert res["success"] is False assert "boom" in res["error"] - m_mark.assert_called_once() + m_mark.assert_called_once_with( + "job-run-1", + False, + "boom", + expected_fire_owner="manual-owner", + ) def test_execute_job_now_heartbeats_while_job_runs(self): """A manual run ticks the caller's activity tracker while the job @@ -116,7 +199,7 @@ def slow_run(job, **kw): assert heartbeat_seen.wait(timeout=5.0), "no heartbeat within 5s" return True - with patch("tools.cronjob_tools.claim_job_for_fire", return_value=True), \ + with patch("tools.cronjob_tools.claim_job_for_fire", return_value={**_JOB, "fire_claim": {"by": "manual-owner"}}), \ patch("tools.cronjob_tools._CRON_RUN_HEARTBEAT_INTERVAL", 0.05), \ patch("cron.scheduler.run_one_job", side_effect=slow_run) as m_run, \ patch("tools.cronjob_tools.get_job", @@ -134,7 +217,7 @@ def test_execute_job_now_without_callback_does_not_heartbeat(self): heartbeat thread is never started and behavior is unchanged.""" set_activity_callback(None) try: - with patch("tools.cronjob_tools.claim_job_for_fire", return_value=True), \ + with patch("tools.cronjob_tools.claim_job_for_fire", return_value={**_JOB, "fire_claim": {"by": "manual-owner"}}), \ patch("cron.scheduler.run_one_job", return_value=True) as m_run, \ patch("tools.cronjob_tools.get_job", return_value={"last_status": "ok", "last_error": None}), \ @@ -165,7 +248,7 @@ def slow_run(job, **kw): time.sleep(0.2) return True - with patch("tools.cronjob_tools.claim_job_for_fire", return_value=True), \ + with patch("tools.cronjob_tools.claim_job_for_fire", return_value={**_JOB, "fire_claim": {"by": "manual-owner"}}), \ patch("tools.cronjob_tools._CRON_RUN_HEARTBEAT_INTERVAL", 0.05), \ patch("tools.cronjob_tools._CRON_RUN_HEARTBEAT_CEILING", 0.0), \ patch("cron.scheduler.run_one_job", side_effect=slow_run), \ @@ -198,7 +281,7 @@ def slow_run(job, **kw): "heartbeat stopped after one callback exception" return True - with patch("tools.cronjob_tools.claim_job_for_fire", return_value=True), \ + with patch("tools.cronjob_tools.claim_job_for_fire", return_value={**_JOB, "fire_claim": {"by": "manual-owner"}}), \ patch("tools.cronjob_tools._CRON_RUN_HEARTBEAT_INTERVAL", 0.05), \ patch("cron.scheduler.run_one_job", side_effect=slow_run), \ patch("tools.cronjob_tools.get_job", diff --git a/tools/cronjob_tools.py b/tools/cronjob_tools.py index 6fb94b35b688c..15638fe0ab8b8 100644 --- a/tools/cronjob_tools.py +++ b/tools/cronjob_tools.py @@ -680,9 +680,11 @@ def _execute_job_now( Returns {"claimed": bool, "success": bool, "error": str|None}. """ job_id = job["id"] + claimed_job = None try: # At-most-once claim: bail without running if a tick/other fire owns it. - if not claim_job_for_fire(job_id): + claimed_job = claim_job_for_fire(job_id, return_job=True) + if not isinstance(claimed_job, dict): # claim_job_for_fire returns False for paused/disabled/missing # jobs too — don't mislabel those as "already being fired" # (#60703): that message sends the user chasing a phantom @@ -703,7 +705,7 @@ def _execute_job_now( pass return {"claimed": True, "success": False, "error": str(e)} - return _run_claimed_job(job, extra_prompt=extra_prompt) + return _run_claimed_job(claimed_job, extra_prompt=extra_prompt) def _run_claimed_job( @@ -720,6 +722,7 @@ def _run_claimed_job( """ job_id = job["id"] _registered = False + fire_owner = None try: from cron.scheduler import ( release_running_job, @@ -745,8 +748,13 @@ def _run_claimed_job( } _registered = True + claim = job.get("fire_claim") + fire_owner = str(claim.get("by") or "") if isinstance(claim, dict) else None + # run_one_job records last_run_at/last_status via mark_job_run (which # also clears the fire claim) and returns True iff it processed the job. + # ``job`` here is the exact claimed snapshot (owner-bearing), so the + # shared body fences every terminal write by that owner. # # A manual `run` executes the job synchronously on the caller's thread, # and a cron job is itself a full agent run that routinely takes @@ -854,10 +862,19 @@ def _heartbeat_loop() -> None: except Exception: pass try: - mark_job_run(job_id, False, str(e)) + mark_job_run( + job_id, + False, + str(e), + expected_fire_owner=fire_owner, + ) except Exception: pass - return {"claimed": True, "success": False, "error": str(e)} + return { + "claimed": True, + "success": False, + "error": str(e), + } def _latest_job_output_excerpt(job_id: str, max_chars: int = 2000) -> Optional[str]: @@ -978,7 +995,10 @@ def _try_dispatch_background_run( except Exception: pass - if not claim_job_for_fire(job_id): + # Same snapshot claim as _execute_job_now: carry the owner-bearing + # record into the run so terminal writes stay fenced by this owner. + claimed_job = claim_job_for_fire(job_id, return_job=True) + if not isinstance(claimed_job, dict): refreshed = get_job(job_id) if refreshed is None: reason = "Job no longer exists; nothing to run." @@ -1015,7 +1035,7 @@ def _try_dispatch_background_run( "cronjob run: async delegation registry unavailable (%s); " "running job '%s' inline.", e, job_name, ) - result = _run_claimed_job(job, extra_prompt=extra_prompt) + result = _run_claimed_job(claimed_job, extra_prompt=extra_prompt) result["dispatched"] = False return result @@ -1030,7 +1050,7 @@ def _try_dispatch_background_run( deliver = job.get("deliver", "local") def _runner() -> Dict[str, Any]: - res = _run_claimed_job(job, extra_prompt=extra_prompt) + res = _run_claimed_job(claimed_job, extra_prompt=extra_prompt) duration = round(time.time() - started_at, 2) refreshed = get_job(job_id) or {} lines = [ From f9d64b9a9d8b306f64851c1a13869d96ad5d7869 Mon Sep 17 00:00:00 2001 From: Evgenii <413011+smwbev@users.noreply.github.com> Date: Sun, 9 Aug 2026 07:22:14 +0000 Subject: [PATCH 076/376] fix(cron): add reliable trigger feedback in Web and Desktop clients MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Shared per-job trigger controller (apps/shared) coalesces duplicate clicks for the same profile+job inside a mounted client while letting unrelated jobs run independently; the backend durable claim remains authoritative across windows/processes. - Two-phase feedback everywhere: the action stays disabled/spinning while the request is in flight and the terminal success/error is reported once, after the HTTP response — no premature success toast (Web), matching the Desktop info notification. - Desktop keeps the 24h trigger timeout for the synchronous long operation and fences stale profile/list responses and unmounted surfaces; the sidebar trigger button shows a spinner while busy. --- .../app/chat/sidebar/cron-jobs-section.tsx | 54 ++++- apps/desktop/src/app/chat/sidebar/index.tsx | 2 +- apps/desktop/src/app/contrib/wiring.tsx | 13 +- .../desktop/src/app/cron/cron-actions.test.ts | 187 +++++++++++++++ apps/desktop/src/app/cron/cron-actions.ts | 98 ++++++++ apps/desktop/src/app/cron/index.tsx | 219 ++++++++++++++---- .../session/hooks/use-session-list-actions.ts | 9 +- apps/desktop/src/hermes.test.ts | 18 +- apps/desktop/src/hermes.ts | 8 +- apps/desktop/src/store/cron.test.ts | 45 ++++ apps/desktop/src/store/cron.ts | 74 +++++- apps/shared/src/cron-trigger-controller.ts | 40 ++++ apps/shared/src/index.ts | 5 + web/src/lib/cron-trigger-controller.test.ts | 114 +++++++++ web/src/pages/CronPage.tsx | 125 ++++++++-- 15 files changed, 928 insertions(+), 83 deletions(-) create mode 100644 apps/desktop/src/app/cron/cron-actions.test.ts create mode 100644 apps/desktop/src/app/cron/cron-actions.ts create mode 100644 apps/desktop/src/store/cron.test.ts create mode 100644 apps/shared/src/cron-trigger-controller.ts create mode 100644 web/src/lib/cron-trigger-controller.test.ts diff --git a/apps/desktop/src/app/chat/sidebar/cron-jobs-section.tsx b/apps/desktop/src/app/chat/sidebar/cron-jobs-section.tsx index 9fadcfbc42428..30fb25bde1e45 100644 --- a/apps/desktop/src/app/chat/sidebar/cron-jobs-section.tsx +++ b/apps/desktop/src/app/chat/sidebar/cron-jobs-section.tsx @@ -1,5 +1,6 @@ +import { createCronTriggerController, type CronTriggerController } from '@hermes/shared' import { useStore } from '@nanostores/react' -import { useEffect, useMemo, useState } from 'react' +import { useEffect, useMemo, useRef, useState } from 'react' import { usePaneVisible } from '@/components/pane-shell/pane-visibility' import { ActionsContextMenu, type MenuKit, renderActionItem } from '@/components/ui/actions-menu' @@ -72,7 +73,7 @@ interface SidebarCronJobsSectionProps { // Open the full Cron page focused on this job (manage / full history). onManageJob: (jobId: string) => void // Fire the job now. - onTriggerJob: (jobId: string) => void + onTriggerJob: (jobId: string) => Promise onToggle: () => void open: boolean } @@ -92,6 +93,45 @@ export function SidebarCronJobsSection({ const [peekJobId, setPeekJobId] = useState(null) // Rows revealed so far; starts compact, grows in steps via "load more". const [visibleCount, setVisibleCount] = useState(INITIAL_VISIBLE_JOBS) + const [triggeringJobIds, setTriggeringJobIds] = useState>(() => new Set()) + const triggerControllerRef = useRef(null) + + // eslint-disable-next-line no-restricted-syntax -- controller mount identity, not an atom mirror + useEffect(() => { + const controller = createCronTriggerController((jobId, running) => { + if (triggerControllerRef.current !== controller) { + return + } + + setTriggeringJobIds(current => { + const next = new Set(current) + + if (running) { + next.add(jobId) + } else { + next.delete(jobId) + } + + return next + }) + }) + + triggerControllerRef.current = controller + + return () => { + triggerControllerRef.current = null + } + }, []) + + const triggerJob = (jobId: string) => { + const controller = triggerControllerRef.current + + if (!controller) { + return + } + + void controller.run(jobId, () => onTriggerJob(jobId)).catch(() => undefined) + } const visible = usePaneVisible() @@ -153,6 +193,7 @@ export function SidebarCronJobsSection({ {shown.map(job => ( onManageJob(job.id)} onOpenRun={onOpenRun} onTogglePeek={() => setPeekJobId(prev => (prev === job.id ? null : job.id))} - onTrigger={() => onTriggerJob(job.id)} + onTrigger={() => triggerJob(job.id)} /> ))} {hiddenCount > 0 && ( @@ -176,6 +217,7 @@ export function SidebarCronJobsSection({ } function CronJobSidebarRow({ + busy, expanded, job, nowMs, @@ -184,6 +226,7 @@ function CronJobSidebarRow({ onTogglePeek, onTrigger }: { + busy: boolean expanded: boolean job: CronJob nowMs: number @@ -298,11 +341,12 @@ function CronJobSidebarRow({ diff --git a/apps/desktop/src/app/chat/sidebar/index.tsx b/apps/desktop/src/app/chat/sidebar/index.tsx index 6377cd75b8fd7..fc9d1f36091ab 100644 --- a/apps/desktop/src/app/chat/sidebar/index.tsx +++ b/apps/desktop/src/app/chat/sidebar/index.tsx @@ -284,7 +284,7 @@ interface ChatSidebarProps extends React.ComponentProps { /** Create a brand-new session and open it as a tile on `dir`. */ onNewSessionSplit: (dir: SplitDir) => void onManageCronJob: (jobId: string) => void - onTriggerCronJob: (jobId: string) => void + onTriggerCronJob: (jobId: string) => Promise } export function ChatSidebar({ diff --git a/apps/desktop/src/app/contrib/wiring.tsx b/apps/desktop/src/app/contrib/wiring.tsx index b670d831455eb..ea819a09606c8 100644 --- a/apps/desktop/src/app/contrib/wiring.tsx +++ b/apps/desktop/src/app/contrib/wiring.tsx @@ -24,7 +24,7 @@ import { $newSessionTabAction, registerPaneCloser } from '@/components/pane-shel import { FloatingPet } from '@/components/pet/floating-pet' import { RemoteDisplayBanner } from '@/components/remote-display-banner' import { emitGatewayEvent } from '@/contrib/events' -import { getLatestSessionMessages, triggerCronJob } from '@/hermes' +import { getLatestSessionMessages } from '@/hermes' import { type ChatMessage, chatMessageText, preserveLocalAssistantErrors, toChatMessages } from '@/lib/chat-messages' import { isMessagingSource } from '@/lib/session-source' import { latestSessionTodos } from '@/lib/todos' @@ -40,6 +40,7 @@ import { $activeGatewayProfile, $freshSessionRequest, $profileScope, + ALL_PROFILES, ensureGatewayProfile, newSessionInProfile, normalizeProfileKey, @@ -73,6 +74,7 @@ import { closeWorkspaceTab } from '../chat/close-tab' import { requestComposerInsert } from '../chat/composer/focus' import { useComposerActions } from '../chat/hooks/use-composer-actions' import { CommandPalette } from '../command-palette' +import { triggerAndRefreshCronJobs } from '../cron/cron-actions' import { useGatewayBoot } from '../gateway/hooks/use-gateway-boot' import { useGatewayRequest } from '../gateway/hooks/use-gateway-request' import { useKeybinds } from '../hooks/use-keybinds' @@ -901,11 +903,10 @@ export function ContribWiring({ children }: { children: ReactNode }) { onThreadMessagesChange: handleThreadMessagesChange, onToggleSelectedPin: toggleSelectedPin, onTranscribeAudio: transcribeVoiceAudio, - onTriggerCronJob: jobId => { - void triggerCronJob(jobId) - .then(() => refreshCronJobs()) - .catch(() => undefined) - }, + onTriggerCronJob: jobId => + triggerAndRefreshCronJobs(jobId, profileScope === ALL_PROFILES ? 'all' : profileScope) + .then(() => undefined) + .catch(() => undefined), getGateway: () => gatewayRef.current, openAgents, openCommandCenterSection, diff --git a/apps/desktop/src/app/cron/cron-actions.test.ts b/apps/desktop/src/app/cron/cron-actions.test.ts new file mode 100644 index 0000000000000..61e6dd1aa305f --- /dev/null +++ b/apps/desktop/src/app/cron/cron-actions.test.ts @@ -0,0 +1,187 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' + +const getCronJobs = vi.fn() +const triggerCronJob = vi.fn() + +vi.mock('@/hermes', () => ({ + getCronJobs: (...args: unknown[]) => getCronJobs(...args), + triggerCronJob: (...args: unknown[]) => triggerCronJob(...args) +})) + +import { beginCronJobsRequest } from '@/store/cron' + +import { mutateAndRefreshCronJobs, refreshCronJobs, triggerAndRefreshCronJobs } from './cron-actions' + +function deferred() { + let resolve!: (value: T) => void + + const promise = new Promise(res => { + resolve = res + }) + + return { promise, resolve } +} + +describe('triggerAndRefreshCronJobs', () => { + beforeEach(() => { + getCronJobs.mockReset() + triggerCronJob.mockReset() + }) + + it('replaces the local cache with the authoritative list after a trigger', async () => { + const authoritative = [{ id: 'recurring-job', state: 'scheduled' }] + triggerCronJob.mockResolvedValue({ id: 'deleted-one-shot', state: 'completed' }) + getCronJobs.mockResolvedValue(authoritative) + + const result = await triggerAndRefreshCronJobs('deleted-one-shot', 'work') + + expect(triggerCronJob).toHaveBeenCalledWith('deleted-one-shot') + expect(getCronJobs).toHaveBeenCalledWith('work') + expect(result).toEqual({ jobs: authoritative, refreshError: null, stale: false }) + }) + + it('reports refresh failure separately after a successful trigger', async () => { + const refreshError = new Error('refresh failed') + triggerCronJob.mockResolvedValue({ id: 'job-1', state: 'scheduled' }) + getCronJobs.mockRejectedValue(refreshError) + + const result = await triggerAndRefreshCronJobs('job-1', 'all') + + expect(result).toEqual({ jobs: null, refreshError, stale: false }) + }) + + it('still rejects when the trigger itself fails', async () => { + const triggerError = new Error('trigger failed') + triggerCronJob.mockRejectedValue(triggerError) + + await expect(triggerAndRefreshCronJobs('job-1', 'all')).rejects.toBe(triggerError) + expect(getCronJobs).not.toHaveBeenCalled() + }) + + it('discards a trigger failure after the profile scope changes', async () => { + const trigger = deferred() + triggerCronJob.mockReturnValue(trigger.promise) + + const resultPromise = triggerAndRefreshCronJobs('job-1', 'work') + beginCronJobsRequest('personal') + trigger.resolve(Promise.reject(new Error('old profile failed')) as never) + + await expect(resultPromise).resolves.toEqual({ jobs: null, refreshError: null, stale: true }) + expect(getCronJobs).not.toHaveBeenCalled() + }) + + it('discards a trigger refresh after the profile scope changes', async () => { + const refresh = deferred>() + triggerCronJob.mockResolvedValue({ id: 'job-1', state: 'scheduled' }) + getCronJobs.mockReturnValue(refresh.promise) + + const resultPromise = triggerAndRefreshCronJobs('job-1', 'work') + await triggerCronJob.mock.results[0]?.value + + beginCronJobsRequest('personal') + refresh.resolve([{ id: 'work-job' }]) + + await expect(resultPromise).resolves.toEqual({ jobs: null, refreshError: null, stale: true }) + }) + + it('discards an older ordinary refresh that completes after a trigger refresh', async () => { + const older = deferred>() + const newer = deferred>() + triggerCronJob.mockResolvedValue({ id: 'job-1', state: 'scheduled' }) + getCronJobs.mockReturnValueOnce(older.promise).mockReturnValueOnce(newer.promise) + + const olderPromise = refreshCronJobs('work') + const newerPromise = triggerAndRefreshCronJobs('job-1', 'work') + newer.resolve([{ id: 'newer' }]) + older.resolve([{ id: 'older' }]) + + await expect(newerPromise).resolves.toEqual({ + jobs: [{ id: 'newer' }], + refreshError: null, + stale: false + }) + await expect(olderPromise).resolves.toEqual({ jobs: null, refreshError: null, stale: true }) + }) +}) + +describe('mutateAndRefreshCronJobs', () => { + beforeEach(() => { + getCronJobs.mockReset() + }) + + it('does not refresh the old profile after a successful mutation switches scope', async () => { + const mutation = deferred<{ id: string }>() + const resultPromise = mutateAndRefreshCronJobs('work', () => mutation.promise) + + beginCronJobsRequest('personal') + mutation.resolve({ id: 'work-job' }) + + await expect(resultPromise).resolves.toEqual({ + jobs: null, + refreshError: null, + stale: true, + value: null + }) + expect(getCronJobs).not.toHaveBeenCalled() + }) + + it('suppresses a mutation error after the profile scope changes', async () => { + const mutation = deferred() + const resultPromise = mutateAndRefreshCronJobs('work', () => mutation.promise) + + beginCronJobsRequest('personal') + mutation.resolve(Promise.reject(new Error('old profile failed')) as never) + + await expect(resultPromise).resolves.toEqual({ + jobs: null, + refreshError: null, + stale: true, + value: null + }) + }) + + it('allows overlapping same-profile mutations to authoritatively refresh', async () => { + const first = deferred() + const second = deferred() + getCronJobs + .mockResolvedValueOnce([{ id: 'after-second' }]) + .mockResolvedValueOnce([{ id: 'after-both' }]) + + const firstResult = mutateAndRefreshCronJobs('work', () => first.promise) + const secondResult = mutateAndRefreshCronJobs('work', () => second.promise) + + second.resolve('second') + await expect(secondResult).resolves.toMatchObject({ stale: false, value: 'second' }) + + first.resolve('first') + await expect(firstResult).resolves.toMatchObject({ stale: false, value: 'first' }) + expect(getCronJobs).toHaveBeenCalledTimes(2) + }) + + it('preserves a successful mutation when a newer same-profile refresh supersedes its snapshot', async () => { + const mutation = deferred() + const mutationRefresh = deferred>() + const newerRefresh = deferred>() + getCronJobs.mockReturnValueOnce(mutationRefresh.promise).mockReturnValueOnce(newerRefresh.promise) + + const mutationResult = mutateAndRefreshCronJobs('work', () => mutation.promise) + mutation.resolve('created') + await vi.waitFor(() => expect(getCronJobs).toHaveBeenCalledTimes(1)) + + const newerResult = refreshCronJobs('work') + newerRefresh.resolve([{ id: 'newer' }]) + await expect(newerResult).resolves.toEqual({ + jobs: [{ id: 'newer' }], + refreshError: null, + stale: false + }) + + mutationRefresh.resolve([{ id: 'older' }]) + await expect(mutationResult).resolves.toEqual({ + jobs: null, + refreshError: null, + stale: false, + value: 'created' + }) + }) +}) diff --git a/apps/desktop/src/app/cron/cron-actions.ts b/apps/desktop/src/app/cron/cron-actions.ts new file mode 100644 index 0000000000000..95cc07c63e9d6 --- /dev/null +++ b/apps/desktop/src/app/cron/cron-actions.ts @@ -0,0 +1,98 @@ +import { type CronJob, getCronJobs, triggerCronJob } from '@/hermes' +import { + beginCronJobsAction, + beginCronJobsRequest, + commitCronJobsRequest, + type CronJobsRequest, + isCronJobsRequestCurrent, + isCronJobsScopeCurrent +} from '@/store/cron' + +export interface CronTriggerRefreshResult { + jobs: CronJob[] | null + refreshError: unknown | null + stale: boolean +} + +export interface CronMutationRefreshResult extends CronTriggerRefreshResult { + value: T | null +} + +async function refreshForGeneration( + profile: string, + request: CronJobsRequest +): Promise { + try { + const jobs = await getCronJobs(profile) + + if (!commitCronJobsRequest(request, jobs)) { + return { jobs: null, refreshError: null, stale: true } + } + + return { jobs, refreshError: null, stale: false } + } catch (refreshError) { + if (!isCronJobsRequestCurrent(request)) { + return { jobs: null, refreshError: null, stale: true } + } + + return { jobs: null, refreshError, stale: false } + } +} + +export function refreshCronJobs(profile: string): Promise { + return refreshForGeneration(profile, beginCronJobsRequest(profile)) +} + +export async function mutateAndRefreshCronJobs( + profile: string, + mutate: () => Promise +): Promise> { + const scopeToken = beginCronJobsAction(profile) + let value: T + + try { + value = await mutate() + } catch (mutationError) { + if (!isCronJobsScopeCurrent(scopeToken)) { + return { jobs: null, refreshError: null, stale: true, value: null } + } + + throw mutationError + } + + if (!isCronJobsScopeCurrent(scopeToken)) { + return { jobs: null, refreshError: null, stale: true, value: null } + } + + const refreshed = await refreshCronJobs(profile) + + if (!isCronJobsScopeCurrent(scopeToken)) { + return { jobs: null, refreshError: null, stale: true, value: null } + } + + // A newer request in the same scope may supersede this refresh after the + // mutation itself has already succeeded. Preserve the mutation result so + // callers can settle dialogs/toasts without publishing the older snapshot. + if (refreshed.stale) { + return { jobs: null, refreshError: null, stale: false, value } + } + + return { ...refreshed, value } +} + +/** + * Trigger a job synchronously, then replace the local view from the backend. + * A completed one-shot may have been deleted, so the trigger response alone is + * not an authoritative list update. Refresh failure is reported separately: + * the trigger already succeeded and must not be shown as failed. + */ +export async function triggerAndRefreshCronJobs( + jobId: string, + profile: 'all' | string +): Promise { + const { value: _value, ...result } = await mutateAndRefreshCronJobs(profile, () => + triggerCronJob(jobId) + ) + + return result +} \ No newline at end of file diff --git a/apps/desktop/src/app/cron/index.tsx b/apps/desktop/src/app/cron/index.tsx index 616a9b956dd66..1b6e6ab1566d2 100644 --- a/apps/desktop/src/app/cron/index.tsx +++ b/apps/desktop/src/app/cron/index.tsx @@ -1,3 +1,4 @@ +import { createCronTriggerController, type CronTriggerController } from '@hermes/shared' import { useStore } from '@nanostores/react' import { useQuery } from '@tanstack/react-query' import type * as React from 'react' @@ -36,19 +37,17 @@ import { getAutomationBlueprints, getCronDeliveryTargets, getCronJobRuns, - getCronJobs, instantiateAutomationBlueprint, pauseCronJob, resumeCronJob, type SessionInfo, - triggerCronJob, updateCronJob } from '@/hermes' import { type Translations, useI18n } from '@/i18n' import { AlertTriangle } from '@/lib/icons' import { requestModelOptions } from '@/lib/model-options' import { asText } from '@/lib/text' -import { $cronFocusJobId, $cronJobs, setCronFocusJobId, setCronJobs, updateCronJobs } from '@/store/cron' +import { $cronFocusJobId, $cronJobs, invalidateCronJobsRequests, setCronFocusJobId } from '@/store/cron' import { $changeEventsAvailable, $cronChangeTick } from '@/store/live-sync' import { notify, notifyError } from '@/store/notifications' import { $profileScope, ALL_PROFILES } from '@/store/profile' @@ -74,6 +73,7 @@ import { import type { SetStatusbarItemGroup } from '../shell/statusbar-controls' import { BlueprintSlotControl, blueprintSlotHelp, cleanBlueprintFieldError, initialBlueprintValues } from './blueprints' +import { mutateAndRefreshCronJobs, refreshCronJobs, triggerAndRefreshCronJobs } from './cron-actions' import { cronEditorUpdates, jobIsScriptOnly, @@ -93,6 +93,10 @@ const MODEL_DEFAULT_VALUE = '__default__' // blueprint key. Blueprint keys never collide with this sentinel. const CUSTOM_TEMPLATE = 'custom' +function cronProfileForScope(scope: string): string { + return scope === ALL_PROFILES ? 'all' : scope +} + const SCHEDULE_OPTIONS: ReadonlyArray = [ { expr: '0 9 * * *', value: 'daily' }, { expr: '0 9 * * 1-5', value: 'weekdays' }, @@ -299,7 +303,37 @@ export function CronView({ onClose, onOpenSession, setStatusbarItemGroup: _setSt const jobs = useStore($cronJobs) const [loading, setLoading] = useState(jobs.length === 0) const [query, setQuery] = useState('') - const [busyJobId, setBusyJobId] = useState(null) + const [busyJobTokens, setBusyJobTokens] = useState>(() => new Map()) + const [triggeringJobKeys, setTriggeringJobKeys] = useState>(() => new Set()) + const triggerControllerRef = useRef(null) + + // eslint-disable-next-line no-restricted-syntax -- controller mount identity, not an atom mirror + useEffect(() => { + const controller = createCronTriggerController((key, running) => { + if (triggerControllerRef.current !== controller) { + return + } + + setTriggeringJobKeys(current => { + const next = new Set(current) + + if (running) { + next.add(key) + } else { + next.delete(key) + } + + return next + }) + }) + + triggerControllerRef.current = controller + + return () => { + triggerControllerRef.current = null + } + }, []) + // Master/detail: the job whose schedule + run history fill the right pane. const [selectedJobId, setSelectedJobId] = useState(null) // Set when a job is opened from the sidebar so we scroll it into view once the @@ -315,21 +349,30 @@ export function CronView({ onClose, onOpenSession, setStatusbarItemGroup: _setSt // default — scope the fetch to the sidebar's profile scope so this overlay // and the sidebar (which share the $cronJobs atom) agree on what's shown. const profileScope = useStore($profileScope) + const profile = cronProfileForScope(profileScope) const refresh = useCallback(async () => { - try { - setCronJobs(await getCronJobs(profileScope === ALL_PROFILES ? 'all' : profileScope)) - } catch (err) { - notifyError(err, c.failedLoad) - } finally { - setLoading(false) + const { refreshError, stale } = await refreshCronJobs(profile) + + if (stale) { + return } - }, [c, profileScope]) + + if (refreshError) { + notifyError(refreshError, c.failedLoad) + } + + setLoading(false) + }, [c, profile]) useRefreshHotkey(refresh) useEffect(() => { void refresh() + // Fence the previous profile's request before the next profile effect, and + // fence every pending completion when the overlay unmounts. + + return () => invalidateCronJobsRequests() }, [refresh]) // Sidebar → "open this job": resolve the focus id (or name) to a job, select @@ -395,13 +438,46 @@ export function CronView({ onClose, onOpenSession, setStatusbarItemGroup: _setSt const totalCount = jobs.length + function beginJobBusy(jobId: string): symbol { + const token = Symbol(jobId) + + setBusyJobTokens(current => new Map(current).set(jobId, token)) + + return token + } + + function endJobBusy(jobId: string, token: symbol): void { + setBusyJobTokens(current => { + if (current.get(jobId) !== token) { + return current + } + + const next = new Map(current) + + next.delete(jobId) + + return next + }) + } + async function handlePauseResume(job: CronJob) { - setBusyJobId(job.id) + const busyToken = beginJobBusy(job.id) try { const isPaused = jobState(job) === 'paused' - const updated = isPaused ? await resumeCronJob(job.id) : await pauseCronJob(job.id) - updateCronJobs(rows => rows.map(row => (row.id === job.id ? updated : row))) + + const { refreshError, stale } = await mutateAndRefreshCronJobs(profile, () => + isPaused ? resumeCronJob(job.id) : pauseCronJob(job.id) + ) + + if (stale) { + return + } + + if (refreshError) { + notifyError(refreshError, c.failedLoad) + } + notify({ kind: 'success', title: isPaused ? c.resumed : c.paused, @@ -410,21 +486,50 @@ export function CronView({ onClose, onOpenSession, setStatusbarItemGroup: _setSt } catch (err) { notifyError(err, c.failedUpdate) } finally { - setBusyJobId(null) + endJobBusy(job.id, busyToken) } } async function handleTrigger(job: CronJob) { - setBusyJobId(job.id) + const viewProfile = profile + const key = `${viewProfile}:${job.id}` + const controller = triggerControllerRef.current + + if (!controller) { + return + } try { - const updated = await triggerCronJob(job.id) - updateCronJobs(rows => rows.map(row => (row.id === job.id ? updated : row))) + const run = await controller.run( + key, + () => triggerAndRefreshCronJobs(job.id, viewProfile), + () => notify({ kind: 'info', title: c.triggerNow, message: truncate(jobTitle(job), 60) }) + ) + + if ( + triggerControllerRef.current !== controller || + cronProfileForScope($profileScope.get()) !== viewProfile || + !run.started || + !run.value + ) { + return + } + + const { refreshError, stale } = run.value + + if (stale) { + return + } + + if (refreshError) { + notifyError(refreshError, c.failedLoad) + } + notify({ kind: 'success', title: c.triggered, message: truncate(jobTitle(job), 60) }) } catch (err) { - notifyError(err, c.failedTrigger) - } finally { - setBusyJobId(null) + if (triggerControllerRef.current === controller && cronProfileForScope($profileScope.get()) === viewProfile) { + notifyError(err, c.failedTrigger) + } } } @@ -436,8 +541,18 @@ export function CronView({ onClose, onOpenSession, setStatusbarItemGroup: _setSt setDeleting(true) try { - await deleteCronJob(pendingDelete.id) - updateCronJobs(rows => rows.filter(row => row.id !== pendingDelete.id)) + const { refreshError, stale } = await mutateAndRefreshCronJobs(profile, () => + deleteCronJob(pendingDelete.id) + ) + + if (stale) { + return + } + + if (refreshError) { + notifyError(refreshError, c.failedLoad) + } + notify({ kind: 'success', title: c.deleted, message: truncate(jobTitle(pendingDelete), 60) }) setPendingDelete(null) } catch (err) { @@ -449,22 +564,40 @@ export function CronView({ onClose, onOpenSession, setStatusbarItemGroup: _setSt async function handleEditorSave(values: EditorValues) { if (editor.mode === 'create') { - const created = await createCronJob({ - prompt: values.prompt, - schedule: values.schedule, - name: values.name || undefined, - deliver: values.deliver || DEFAULT_DELIVER, - ...(values.model.trim() ? { model: values.model.trim(), provider: values.provider.trim() || undefined } : {}) - }) + const { value: created, refreshError, stale } = await mutateAndRefreshCronJobs(profile, () => + createCronJob({ + prompt: values.prompt, + schedule: values.schedule, + name: values.name || undefined, + deliver: values.deliver || DEFAULT_DELIVER, + ...(values.model.trim() ? { model: values.model.trim(), provider: values.provider.trim() || undefined } : {}) + }) + ) + + if (stale || !created) { + return + } + + if (refreshError) { + notifyError(refreshError, c.failedLoad) + } - updateCronJobs(rows => [...rows, created]) notify({ kind: 'success', title: c.created, message: truncate(jobTitle(created), 60) }) } else if (editor.mode === 'edit') { const scriptOnlyJob = jobIsScriptOnly(editor.job) - const updated = await updateCronJob(editor.job.id, cronEditorUpdates(values, { scriptOnlyJob })) + const { value: updated, refreshError, stale } = await mutateAndRefreshCronJobs(profile, () => + updateCronJob(editor.job.id, cronEditorUpdates(values, { scriptOnlyJob })) + ) + + if (stale || !updated) { + return + } + + if (refreshError) { + notifyError(refreshError, c.failedLoad) + } - updateCronJobs(rows => rows.map(row => (row.id === updated.id ? updated : row))) notify({ kind: 'success', title: c.updated, message: truncate(jobTitle(updated), 60) }) } @@ -477,14 +610,20 @@ export function CronView({ onClose, onOpenSession, setStatusbarItemGroup: _setSt // real per-profile job, and "all" is not a writable target — collapse it to // 'default', matching the manual create path in handleEditorSave. async function handleBlueprintCreate(blueprint: AutomationBlueprint, values: Record) { - const profile = profileScope === ALL_PROFILES ? 'default' : profileScope - const job = await instantiateAutomationBlueprint({ blueprint: blueprint.key, values }, profile) + const writableProfile = profileScope === ALL_PROFILES ? 'default' : profileScope - updateCronJobs(rows => { - const rest = rows.filter(row => row.id !== job.id) + const { value: job, refreshError, stale } = await mutateAndRefreshCronJobs(profile, () => + instantiateAutomationBlueprint({ blueprint: blueprint.key, values }, writableProfile) + ) + + if (stale || !job) { + return + } + + if (refreshError) { + notifyError(refreshError, c.failedLoad) + } - return [...rest, job] - }) notify({ kind: 'success', title: c.blueprints.scheduled, message: asText(job.schedule_display) || blueprint.title }) setEditor({ mode: 'closed' }) } @@ -555,7 +694,7 @@ export function CronView({ onClose, onOpenSession, setStatusbarItemGroup: _setSt {selectedJob ? ( { try { - const jobs = await getCronJobs(profileScope === ALL_PROFILES ? 'all' : profileScope) - - setCronJobs(jobs) + await refreshCronJobsStore(profileScope === ALL_PROFILES ? 'all' : profileScope) } catch { // Non-fatal: the cron section just keeps its last-known jobs. } diff --git a/apps/desktop/src/hermes.test.ts b/apps/desktop/src/hermes.test.ts index dff0379e103c2..da39116d06e41 100644 --- a/apps/desktop/src/hermes.test.ts +++ b/apps/desktop/src/hermes.test.ts @@ -22,7 +22,8 @@ import { resetSidebarBatchCapability, setApiRequestProfile, speakText, - transcribeAudio + transcribeAudio, + triggerCronJob } from './hermes' import { refreshActiveProfile } from './store/profile' @@ -310,6 +311,21 @@ describe('Hermes REST helpers', () => { } }) + it('waits for synchronous cron triggers as a long-running operation', async () => { + api.mockResolvedValue({ id: 'job-1' }) + + await triggerCronJob('job-1') + + const request = api.mock.calls[0]?.[0] + expect(request).toEqual( + expect.objectContaining({ + path: '/api/cron/jobs/job-1/trigger', + method: 'POST' + }) + ) + expect(request.timeoutMs).toBeGreaterThanOrEqual(60 * 60 * 1000) + }) + it('keeps the liveness poll on the short default so a dead backend fails fast', async () => { api.mockResolvedValue({}) api.mockClear() diff --git a/apps/desktop/src/hermes.ts b/apps/desktop/src/hermes.ts index d19b958603cca..b747bffa749a8 100644 --- a/apps/desktop/src/hermes.ts +++ b/apps/desktop/src/hermes.ts @@ -86,6 +86,11 @@ import type { export const STARTUP_REQUEST_TIMEOUT_MS = 60_000 const DEFAULT_GATEWAY_REQUEST_TIMEOUT_MS = 30_000 const SESSION_LIST_REQUEST_TIMEOUT_MS = 60_000 +// The cron trigger endpoint intentionally waits for the whole job so its +// response reflects the persisted execution result. Agent jobs can run far +// longer than the Electron fetch default; keep this override local to the one +// synchronous long-operation endpoint rather than weakening all API timeouts. +export const CRON_TRIGGER_REQUEST_TIMEOUT_MS = 24 * 60 * 60 * 1000 // prompt.submit is effectively fire-and-forget: turn completion is signaled by // stream / message.complete events, NOT by the RPC return. A long turn (MoA // presets running references + aggregator in series, deep reasoning, large tool @@ -1487,7 +1492,8 @@ export function triggerCronJob(jobId: string): Promise { return window.hermesDesktop.api({ ...profileScoped(), path: `/api/cron/jobs/${encodeURIComponent(jobId)}/trigger`, - method: 'POST' + method: 'POST', + timeoutMs: CRON_TRIGGER_REQUEST_TIMEOUT_MS }) } diff --git a/apps/desktop/src/store/cron.test.ts b/apps/desktop/src/store/cron.test.ts new file mode 100644 index 0000000000000..e4db02e1f5bc0 --- /dev/null +++ b/apps/desktop/src/store/cron.test.ts @@ -0,0 +1,45 @@ +import { beforeEach, describe, expect, it } from 'vitest' + +import { + $cronJobs, + beginCronJobsRequest, + commitCronJobsRequest, + setCronJobs, + updateCronJobs +} from './cron' + +const oldJob = { id: 'old' } as never +const newJob = { id: 'new' } as never + +describe('cron jobs request fencing', () => { + beforeEach(() => { + setCronJobs([]) + }) + + it('rejects an older refresh after a newer refresh commits', () => { + const older = beginCronJobsRequest('all') + const newer = beginCronJobsRequest('all') + + expect(commitCronJobsRequest(newer, [newJob])).toBe(true) + expect(commitCronJobsRequest(older, [oldJob])).toBe(false) + expect($cronJobs.get()).toEqual([newJob]) + }) + + it('rejects a refresh from the previous profile scope', () => { + const work = beginCronJobsRequest('work') + + beginCronJobsRequest('personal') + + expect(commitCronJobsRequest(work, [oldJob])).toBe(false) + expect($cronJobs.get()).toEqual([]) + }) + + it('rejects an in-flight poll after a local mutation', () => { + const poll = beginCronJobsRequest('all') + + updateCronJobs(() => [newJob]) + + expect(commitCronJobsRequest(poll, [oldJob])).toBe(false) + expect($cronJobs.get()).toEqual([newJob]) + }) +}) diff --git a/apps/desktop/src/store/cron.ts b/apps/desktop/src/store/cron.ts index f017f60dc1281..b6df8cb0fc586 100644 --- a/apps/desktop/src/store/cron.ts +++ b/apps/desktop/src/store/cron.ts @@ -6,11 +6,81 @@ import type { CronJob } from '@/types/hermes' // the job — schedule, state, live next-run countdown — makes the job the // first-class entity; its runs (sessions) resolve under it in the cron detail. export const $cronJobs = atom([]) -export const setCronJobs = (jobs: CronJob[]) => $cronJobs.set(jobs) + +export interface CronJobsRequest { + generation: number + scope: string +} + +export interface CronJobsScopeToken { + generation: number + scope: string +} + +let cronJobsRequestGeneration = 0 +let cronJobsRequestScope = '' +let cronJobsScopeGeneration = 0 + +function activateCronJobsScope(scope: string): void { + if (scope === cronJobsRequestScope) { + return + } + + cronJobsRequestScope = scope + cronJobsRequestGeneration += 1 + cronJobsScopeGeneration += 1 +} + +export function beginCronJobsRequest(scope: string): CronJobsRequest { + activateCronJobsScope(scope) + cronJobsRequestGeneration += 1 + + return { generation: cronJobsRequestGeneration, scope } +} + +export function beginCronJobsAction(scope: string): CronJobsScopeToken { + activateCronJobsScope(scope) + + return { generation: cronJobsScopeGeneration, scope } +} + +export function isCronJobsScopeCurrent(token: CronJobsScopeToken): boolean { + return token.scope === cronJobsRequestScope && token.generation === cronJobsScopeGeneration +} + +export function isCronJobsRequestCurrent(request: CronJobsRequest): boolean { + return request.scope === cronJobsRequestScope && request.generation === cronJobsRequestGeneration +} + +export function invalidateCronJobsRequests(): void { + cronJobsRequestGeneration += 1 + cronJobsScopeGeneration += 1 +} + +export function commitCronJobsRequest(request: CronJobsRequest, jobs: CronJob[]): boolean { + if (!isCronJobsRequestCurrent(request)) { + return false + } + + // Consume the token so neither a duplicate completion nor any older request + // can publish after this authoritative snapshot. + cronJobsRequestGeneration += 1 + $cronJobs.set(jobs) + + return true +} + +export const setCronJobs = (jobs: CronJob[]) => { + cronJobsRequestGeneration += 1 + $cronJobs.set(jobs) +} // In-place edit so the cron overlay's mutations (create/edit/delete/pause/…) // land in the same atom the sidebar renders — no stale list until the next poll. -export const updateCronJobs = (fn: (jobs: CronJob[]) => CronJob[]) => $cronJobs.set(fn($cronJobs.get())) +export const updateCronJobs = (fn: (jobs: CronJob[]) => CronJob[]) => { + cronJobsRequestGeneration += 1 + $cronJobs.set(fn($cronJobs.get())) +} // One-shot focus target: clicking "Manage" on a job sets this, then opens the // cron overlay, which reads it once to select + scroll to that job. Cleared diff --git a/apps/shared/src/cron-trigger-controller.ts b/apps/shared/src/cron-trigger-controller.ts new file mode 100644 index 0000000000000..032231bd38dc3 --- /dev/null +++ b/apps/shared/src/cron-trigger-controller.ts @@ -0,0 +1,40 @@ +export interface CronTriggerRunResult { + started: boolean + value: T | null +} + +export interface CronTriggerController { + isRunning(key: string): boolean + run(key: string, action: () => Promise, onStarted?: () => void): Promise> +} + +// This is an interaction guard for one mounted UI surface. Cross-window and +// cross-process exclusion remains the backend's responsibility via its durable +// cron claim; a renderer-local Set must never be treated as the execution lock. +export function createCronTriggerController( + onRunningChange: (key: string, running: boolean) => void = () => undefined +): CronTriggerController { + const running = new Set() + + return { + isRunning: key => running.has(key), + async run(key: string, action: () => Promise, onStarted?: () => void) { + if (running.has(key)) { + return { started: false, value: null } + } + + running.add(key) + + try { + onRunningChange(key, true) + + onStarted?.() + + return { started: true, value: await action() } + } finally { + running.delete(key) + onRunningChange(key, false) + } + } + } +} diff --git a/apps/shared/src/index.ts b/apps/shared/src/index.ts index 391a3715bd207..8a173429613ae 100644 --- a/apps/shared/src/index.ts +++ b/apps/shared/src/index.ts @@ -34,6 +34,11 @@ export { type SettlementDeps, type SettlementOutcome } from './charge-settlement' +export { + createCronTriggerController, + type CronTriggerController, + type CronTriggerRunResult +} from './cron-trigger-controller' export { type ConnectionState, type GatewayClientOptions, diff --git a/web/src/lib/cron-trigger-controller.test.ts b/web/src/lib/cron-trigger-controller.test.ts new file mode 100644 index 0000000000000..2b527eb17d9e8 --- /dev/null +++ b/web/src/lib/cron-trigger-controller.test.ts @@ -0,0 +1,114 @@ +import { describe, expect, it, vi } from "vitest"; + +import { createCronTriggerController } from "@hermes/shared"; + +function deferred() { + let resolve!: (value: T) => void; + let reject!: (error: unknown) => void; + const promise = new Promise((res, rej) => { + resolve = res; + reject = rej; + }); + + return { promise, reject, resolve }; +} + +describe("createCronTriggerController", () => { + it("announces immediately and coalesces the same job while it is running", async () => { + const request = deferred(); + const order: string[] = []; + const action = vi.fn(() => { + order.push("action"); + return request.promise; + }); + const onStarted = vi.fn(() => order.push("started")); + const onRunningChange = vi.fn(); + const controller = createCronTriggerController(onRunningChange); + + const first = controller.run("profile-a:job-1", action, onStarted); + const duplicate = await controller.run("profile-a:job-1", action, onStarted); + + expect(action).toHaveBeenCalledTimes(1); + expect(onStarted).toHaveBeenCalledTimes(1); + expect(order).toEqual(["started", "action"]); + expect(onRunningChange).toHaveBeenNthCalledWith(1, "profile-a:job-1", true); + expect(duplicate).toEqual({ started: false, value: null }); + + request.resolve("done"); + await expect(first).resolves.toEqual({ started: true, value: "done" }); + expect(onRunningChange).toHaveBeenLastCalledWith("profile-a:job-1", false); + }); + + it("releases the job after failure so a retry can start", async () => { + const request = deferred(); + const controller = createCronTriggerController(); + + const failed = controller.run("job-1", () => request.promise); + request.reject(new Error("failed")); + + await expect(failed).rejects.toThrow("failed"); + await expect(controller.run("job-1", async () => "retried")).resolves.toEqual({ + started: true, + value: "retried", + }); + }); + + it("releases the job when the immediate feedback callback fails", async () => { + const controller = createCronTriggerController(); + + await expect( + controller.run("job-1", async () => "not-called", () => { + throw new Error("toast failed"); + }), + ).rejects.toThrow("toast failed"); + + await expect(controller.run("job-1", async () => "retried")).resolves.toEqual({ + started: true, + value: "retried", + }); + }); + + it("allows the same job id in different profiles to run concurrently", async () => { + const defaultRequest = deferred(); + const workRequest = deferred(); + const defaultAction = vi.fn(() => defaultRequest.promise); + const workAction = vi.fn(() => workRequest.promise); + const controller = createCronTriggerController(); + + const defaultRun = controller.run("default:job-1", defaultAction); + const workRun = controller.run("work:job-1", workAction); + + expect(defaultAction).toHaveBeenCalledTimes(1); + expect(workAction).toHaveBeenCalledTimes(1); + + defaultRequest.resolve("default"); + workRequest.resolve("work"); + + await expect(defaultRun).resolves.toEqual({ started: true, value: "default" }); + await expect(workRun).resolves.toEqual({ started: true, value: "work" }); + }); + + it("releases the job when the running-state callback fails", async () => { + const controller = createCronTriggerController((_key, running) => { + if (running) throw new Error("state callback failed"); + }); + + await expect(controller.run("job-1", async () => "not-called")).rejects.toThrow( + "state callback failed", + ); + expect(controller.isRunning("job-1")).toBe(false); + }); + + it("releases the job before the stopped-state callback runs", async () => { + const action = vi.fn(async () => "done"); + const controller = createCronTriggerController((_key, running) => { + if (!running) throw new Error("stopped callback failed"); + }); + + await expect(controller.run("job-1", action)).rejects.toThrow("stopped callback failed"); + expect(controller.isRunning("job-1")).toBe(false); + + await expect(controller.run("job-1", action)).rejects.toThrow("stopped callback failed"); + expect(action).toHaveBeenCalledTimes(2); + }); +}); diff --git a/web/src/pages/CronPage.tsx b/web/src/pages/CronPage.tsx index 2f55680fb5c3f..8d7b81ae7c1ed 100644 --- a/web/src/pages/CronPage.tsx +++ b/web/src/pages/CronPage.tsx @@ -1,4 +1,8 @@ -import { useCallback, useEffect, useLayoutEffect, useState } from "react"; +import { useCallback, useEffect, useLayoutEffect, useRef, useState } from "react"; +import { + type CronTriggerController, + createCronTriggerController, +} from "@hermes/shared"; import { Clock, Pause, Pencil, Play, Trash2, X, Zap } from "lucide-react"; import { Badge } from "@nous-research/ui/ui/components/badge"; import { Button } from "@nous-research/ui/ui/components/button"; @@ -510,6 +514,27 @@ const STATUS_TONE: Record = { export default function CronPage() { const [jobs, setJobs] = useState([]); + const [triggeringJobKeys, setTriggeringJobKeys] = useState>( + () => new Set(), + ); + const triggerControllerRef = useRef(null); + + useEffect(() => { + const controller = createCronTriggerController((key, running) => { + if (triggerControllerRef.current !== controller) return; + setTriggeringJobKeys((current) => { + const next = new Set(current); + if (running) next.add(key); + else next.delete(key); + return next; + }); + }); + triggerControllerRef.current = controller; + + return () => { + triggerControllerRef.current = null; + }; + }, []); const [profiles, setProfiles] = useState([]); const [selectedProfile, setSelectedProfile] = useState("all"); const [view, setView] = useState<"jobs" | "blueprints">("jobs"); @@ -576,13 +601,38 @@ export default function CronPage() { setEditForm(editorFormFromJob(job)); }, []); - const loadJobs = useCallback(() => { + const selectedProfileRef = useRef(selectedProfile); + const jobsRequestGenerationRef = useRef(0); + const jobsActiveRef = useRef(false); + + const loadJobs = useCallback((profile: string) => { + if (!jobsActiveRef.current || selectedProfileRef.current !== profile) return; + + const generation = ++jobsRequestGenerationRef.current; + api - .getCronJobs(selectedProfile) - .then(setJobs) - .catch(() => showToast(t.common.loading, "error")) - .finally(() => setLoading(false)); - }, [selectedProfile, showToast, t.common.loading]); + .getCronJobs(profile) + .then((nextJobs) => { + if ( + jobsRequestGenerationRef.current === generation && + selectedProfileRef.current === profile + ) setJobs(nextJobs); + }) + .catch(() => { + if ( + jobsRequestGenerationRef.current === generation && + selectedProfileRef.current === profile + ) { + showToast(t.common.loading, "error"); + } + }) + .finally(() => { + if ( + jobsRequestGenerationRef.current === generation && + selectedProfileRef.current === profile + ) setLoading(false); + }); + }, [showToast, t.common.loading]); useEffect(() => { api @@ -604,8 +654,15 @@ export default function CronPage() { }, []); useEffect(() => { - loadJobs(); - }, [loadJobs]); + jobsActiveRef.current = true; + selectedProfileRef.current = selectedProfile; + loadJobs(selectedProfile); + + return () => { + jobsActiveRef.current = false; + jobsRequestGenerationRef.current += 1; + }; + }, [loadJobs, selectedProfile]); // Load resources from the profile the create/edit form actually targets. // Pass "default" explicitly so the global dashboard profile switch cannot @@ -646,7 +703,7 @@ export default function CronPage() { showToast(t.common.create + " ✓", "success"); setCreateForm(emptyCronJobForm()); setCreateModalOpen(false); - loadJobs(); + loadJobs(selectedProfile); } catch (e) { showToast(`${t.config.failedToSave}: ${e}`, "error"); } finally { @@ -677,7 +734,7 @@ export default function CronPage() { ); showToast("Saved changes ✓", "success"); setEditJob(null); - loadJobs(); + loadJobs(selectedProfile); } catch (e) { showToast(`${t.config.failedToSave}: ${e}`, "error"); } finally { @@ -702,22 +759,45 @@ export default function CronPage() { "success", ); } - loadJobs(); + loadJobs(selectedProfile); } catch (e) { showToast(`${t.status.error}: ${e}`, "error"); } }; const handleTrigger = async (job: CronJob) => { + const jobKey = getJobKey(job); + const label = `${t.cron.triggerNow}: "${truncateText(getJobTitle(job), 30)}"`; + const viewProfile = selectedProfile; + const controller = triggerControllerRef.current; + + if (!controller) return; + try { - await api.triggerCronJob(job.id, getJobProfile(job)); - showToast( - `${t.cron.triggerNow}: "${truncateText(getJobTitle(job), 30)}"`, - "success", + // No pre-request toast: the controller's running state already gives + // immediate in-progress feedback (disabled + spinning action), and a + // success-styled toast before the HTTP response would claim a result + // the request has not produced yet. Terminal feedback only. + const result = await controller.run( + jobKey, + () => api.triggerCronJob(job.id, getJobProfile(job)), ); - loadJobs(); + + if ( + triggerControllerRef.current !== controller || + selectedProfileRef.current !== viewProfile || + !result.started + ) return; + + showToast(`${label} ✓`, "success"); + loadJobs(viewProfile); } catch (e) { - showToast(`${t.status.error}: ${e}`, "error"); + if ( + triggerControllerRef.current === controller && + selectedProfileRef.current === viewProfile + ) { + showToast(`${t.status.error}: ${e}`, "error"); + } } }; @@ -732,13 +812,13 @@ export default function CronPage() { `${t.common.delete}: "${job ? truncateText(getJobTitle(job), 30) : id}"`, "success", ); - loadJobs(); + loadJobs(selectedProfile); } catch (e) { showToast(`${t.status.error}: ${e}`, "error"); throw e; } }, - [jobs, loadJobs, showToast, t.common.delete, t.status.error], + [jobs, loadJobs, selectedProfile, showToast, t.common.delete, t.status.error], ), }); @@ -790,7 +870,7 @@ export default function CronPage() { {view === "blueprints" && ( loadJobs(selectedProfile)} /> )} @@ -1094,11 +1174,12 @@ export default function CronPage() { + {!isPrimary && ( + + )} + {conn.kind !== 'local' && ( + <> + + + + )} +
+ } + description={ + conn.kind === 'ssh' + ? `${kindMeta[conn.kind].label} · ${conn.user ? `${conn.user}@` : ''}${conn.host}${conn.port ? `:${conn.port}` : ''}` + : conn.url + ? `${kindMeta[conn.kind].label} · ${conn.url}` + : kindMeta[conn.kind].desc + } + key={conn.id} + title={ + + + {conn.label} + {isPrimary && {s.primaryPill}} + {conn.kind === 'local' && {s.managedPill}} + + } + /> + ) + }) + )} + + {editor ? ( +
+
+ {(['remote', 'cloud', 'ssh'] as const).map(kind => ( + + ))} +
+

{kindMeta[editor.kind].desc}

+ + setEditor({ ...editor, label: e.target.value })} + placeholder={s.labelPlaceholder} + value={editor.label} + /> + } + description={s.labelDesc} + title={s.labelTitle} + /> + + {(editor.kind === 'remote' || editor.kind === 'cloud') && ( + setEditor({ ...editor, url: e.target.value })} + placeholder="http://homelab.lan:9119" + value={editor.url} + /> + } + title={s.urlTitle} + /> + )} + + {editor.kind === 'remote' && ( + <> + + {(['token', 'oauth'] as const).map(mode => ( + + ))} +
+ } + title={t.settings.gateway.authTitle} + /> + {editor.authMode === 'token' && ( + setEditor({ ...editor, token: e.target.value })} + placeholder={t.settings.gateway.pasteSessionToken} + type="password" + value={editor.token} + /> + } + description={t.settings.gateway.tokenDesc} + title={t.settings.gateway.tokenTitle} + /> + )} + + )} + + {editor.kind === 'ssh' && ( + setEditor({ ...editor, host: e.target.value })} + placeholder="user@host:22" + value={editor.host} + /> + } + title={s.sshHostTitle} + /> + )} + +
+ + +
+ + ) : ( +
+ +
+ )} + + setRemoveTarget(null)} + onConfirm={() => remove()} + open={Boolean(removeTarget)} + title={s.removeConfirmTitle} + /> + + ) +} diff --git a/apps/desktop/src/app/settings/index.tsx b/apps/desktop/src/app/settings/index.tsx index f55503702b48f..7a8d45c7f561d 100644 --- a/apps/desktop/src/app/settings/index.tsx +++ b/apps/desktop/src/app/settings/index.tsx @@ -15,6 +15,7 @@ import { Info, Keyboard, KeyRound, + Network, Package, RefreshCw, Settings2, @@ -34,6 +35,7 @@ import { AboutSettings } from './about-settings' import { AppearanceSettings } from './appearance-settings' import { BillingSettings } from './billing' import { ConfigSettings } from './config-settings' +import { ConnectionsSettings } from './connections-settings' import { SECTIONS } from './constants' import { GatewaySettings } from './gateway-settings' import { KeybindSettings } from './keybind-settings' @@ -48,6 +50,7 @@ const SETTINGS_VIEWS: readonly SettingsViewId[] = [ ...SECTIONS.map(s => `config:${s.id}` as SettingsViewId), 'providers', 'gateway', + 'connections', 'keybinds', 'keys', 'notifications', @@ -206,6 +209,13 @@ export function SettingsView({ onClose, onConfigSaved, onMainModelChanged }: Set label: t.settings.nav.gateway, onSelect: () => setActiveView('gateway') }, + { + active: activeView === 'connections', + icon: Network, + id: 'connections', + label: t.settings.nav.connections, + onSelect: () => setActiveView('connections') + }, { active: activeView === 'keybinds', icon: Keyboard, @@ -305,6 +315,8 @@ export function SettingsView({ onClose, onConfigSaved, onMainModelChanged }: Set ) : activeView === 'gateway' ? ( + ) : activeView === 'connections' ? ( + ) : activeView === 'keybinds' ? ( ) : activeView.startsWith('config:') ? ( diff --git a/apps/desktop/src/app/settings/types.ts b/apps/desktop/src/app/settings/types.ts index 2828609ef63fb..5ecdf27a4a2a3 100644 --- a/apps/desktop/src/app/settings/types.ts +++ b/apps/desktop/src/app/settings/types.ts @@ -7,6 +7,7 @@ import type { EnvVarInfo } from '@/types/hermes' export type SettingsView = | 'about' | 'billing' + | 'connections' | 'gateway' | 'keybinds' | 'keys' diff --git a/apps/desktop/src/global.d.ts b/apps/desktop/src/global.d.ts index 6d383b98ca677..4c504c84ba624 100644 --- a/apps/desktop/src/global.d.ts +++ b/apps/desktop/src/global.d.ts @@ -117,6 +117,17 @@ declare global { saveConnectionConfig: (payload: DesktopConnectionConfigInput) => Promise applyConnectionConfig: (payload: DesktopConnectionConfigInput) => Promise testConnectionConfig: (payload: DesktopConnectionConfigInput) => Promise + // v2 multi-connection registry: named agent sources, all persisted + // together (local + any number of remote/cloud/ssh instances). + connections: { + list: () => Promise + save: ( + payload: DesktopRegistryConnectionInput + ) => Promise<{ ok: boolean; connection: DesktopRegistryConnection; registry: DesktopConnectionsRegistry }> + remove: (id: string) => Promise<{ ok: boolean; registry: DesktopConnectionsRegistry }> + setPrimary: (id: string) => Promise<{ ok: boolean; registry: DesktopConnectionsRegistry }> + test: (id: string) => Promise + } sshConfigHosts: () => Promise sshResolveHost: (host: string) => Promise probeConnectionConfig: (remoteUrl: string) => Promise @@ -657,6 +668,56 @@ export interface DesktopConnectionTestResult { remotePlatform?: string } +// ── v2 multi-connection registry (named agent sources) ───────────────────── + +export type DesktopConnectionKind = 'cloud' | 'local' | 'remote' | 'ssh' + +// A registered agent source as the renderer sees it: token bytes never cross +// the IPC boundary (preview + set flag instead, like DesktopConnectionConfig). +export interface DesktopRegistryConnection { + id: string + kind: DesktopConnectionKind + // Required, registry-unique device name ("Homelab", "Work laptop"). + label: string + url?: string + authMode?: 'oauth' | 'token' + org?: string + host?: string + user?: string + port?: number + keyPath?: string + remoteHermesPath?: string + remoteProfile?: string + tokenSet: boolean + tokenPreview: null | string +} + +export interface DesktopConnectionsRegistry { + version: number + // id of the connection that owns the window/primary backend. + primary: string + connections: DesktopRegistryConnection[] +} + +export interface DesktopRegistryConnectionInput { + // Present for edits; omitted on create (the main process mints the id). + id?: string + kind: DesktopConnectionKind + label: string + url?: string + authMode?: 'oauth' | 'token' + // Plaintext token to store (encrypted at rest); omit to keep the saved one. + token?: string + allowPlainTextToken?: boolean + org?: string + host?: string + user?: string + port?: null | number + keyPath?: string + remoteHermesPath?: string + remoteProfile?: string +} + export interface DesktopSshResolveResult { hostname: string | null identityFile: string | null diff --git a/apps/desktop/src/i18n/en.ts b/apps/desktop/src/i18n/en.ts index d8bafbb7bbae0..7ebd6d70dd014 100644 --- a/apps/desktop/src/i18n/en.ts +++ b/apps/desktop/src/i18n/en.ts @@ -345,6 +345,7 @@ export const en: Translations = { providerApiKeys: 'API keys', providerCustomEndpoints: 'Custom Endpoints', gateway: 'Gateway', + connections: 'Connections', apiKeys: 'Tools & Keys', keybinds: 'Keyboard Shortcuts', keysTools: 'Tools', @@ -621,6 +622,46 @@ export const en: Translations = { set: 'Set', clear: 'Clear' }, + // v2 multi-connection registry: Settings → Connections. + connections: { + title: 'Connections', + intro: + 'Register every place your agents live — this device, remote gateways on your network, and Hermes Cloud instances — and use them side by side.', + loading: 'Loading connections…', + loadFailed: 'Could not load connections', + primaryPill: 'Primary', + managedPill: 'This device', + addConnection: 'Add connection', + editConnection: 'Edit', + removeConnection: 'Remove', + removeConfirmTitle: 'Remove this connection?', + removeConfirmDesc: (label: string) => + `“${label}” will be removed from this app. The instance itself is not touched — you can add it again any time.`, + makePrimary: 'Make primary', + testConnection: 'Test', + testing: 'Testing…', + testOk: 'Reachable', + testFailed: 'Connection test failed', + saveFailed: 'Could not save the connection', + removeFailed: 'Could not remove the connection', + kindLocal: 'Local', + kindRemote: 'Remote gateway', + kindCloud: 'Hermes Cloud', + kindSsh: 'SSH', + kindLocalDesc: 'The Hermes runtime managed by this app.', + kindRemoteDesc: 'A Hermes gateway reachable over HTTP(S) — LAN, Tailscale, or the internet.', + kindCloudDesc: 'A hosted instance discovered through your Hermes Cloud account.', + kindSshDesc: 'A Hermes install reached over SSH.', + labelTitle: 'Name', + labelDesc: 'Required. Shown everywhere this instance appears; must be unique (e.g. “Homelab”, “Work laptop”).', + labelPlaceholder: 'Homelab', + urlTitle: 'Gateway URL', + sshHostTitle: 'SSH host', + save: 'Save connection', + saving: 'Saving…', + cancel: 'Cancel', + empty: 'No connections registered yet.' + }, gateway: { loading: 'Loading gateway settings...', unavailableTitle: 'Gateway settings unavailable', diff --git a/apps/desktop/src/i18n/types.ts b/apps/desktop/src/i18n/types.ts index fb994d6de9a5c..a7f021a291abd 100644 --- a/apps/desktop/src/i18n/types.ts +++ b/apps/desktop/src/i18n/types.ts @@ -295,6 +295,7 @@ export interface Translations { providerApiKeys: string providerCustomEndpoints: string gateway: string + connections: string apiKeys: string keybinds: string keysTools: string @@ -515,6 +516,44 @@ export interface Translations { set: string clear: string } + // v2 multi-connection registry: Settings → Connections. + connections: { + title: string + intro: string + loading: string + loadFailed: string + primaryPill: string + managedPill: string + addConnection: string + editConnection: string + removeConnection: string + removeConfirmTitle: string + removeConfirmDesc: (label: string) => string + makePrimary: string + testConnection: string + testing: string + testOk: string + testFailed: string + saveFailed: string + removeFailed: string + kindLocal: string + kindRemote: string + kindCloud: string + kindSsh: string + kindLocalDesc: string + kindRemoteDesc: string + kindCloudDesc: string + kindSshDesc: string + labelTitle: string + labelDesc: string + labelPlaceholder: string + urlTitle: string + sshHostTitle: string + save: string + saving: string + cancel: string + empty: string + } gateway: { loading: string unavailableTitle: string diff --git a/apps/desktop/src/i18n/zh.ts b/apps/desktop/src/i18n/zh.ts index 69c9c0680ad8d..18186fe2cf0b2 100644 --- a/apps/desktop/src/i18n/zh.ts +++ b/apps/desktop/src/i18n/zh.ts @@ -334,6 +334,7 @@ export const zh: Translations = { providerApiKeys: 'API 密钥', providerCustomEndpoints: '自定义端点', gateway: '网关', + connections: '连接', apiKeys: '工具与密钥', keybinds: '键盘快捷键', keysTools: '工具', @@ -828,6 +829,44 @@ export const zh: Translations = { set: '设置', clear: '清除' }, + // v2 多连接注册表:设置 → 连接。 + connections: { + title: '连接', + intro: '注册你的智能体所在的每个位置——本机、局域网中的远程网关、Hermes Cloud 实例——并同时使用它们。', + loading: '正在加载连接…', + loadFailed: '无法加载连接', + primaryPill: '主连接', + managedPill: '本机', + addConnection: '添加连接', + editConnection: '编辑', + removeConnection: '移除', + removeConfirmTitle: '移除此连接?', + removeConfirmDesc: (label: string) => `“${label}”将从本应用移除。实例本身不受影响——你可以随时重新添加。`, + makePrimary: '设为主连接', + testConnection: '测试', + testing: '测试中…', + testOk: '可访问', + testFailed: '连接测试失败', + saveFailed: '无法保存连接', + removeFailed: '无法移除连接', + kindLocal: '本地', + kindRemote: '远程网关', + kindCloud: 'Hermes Cloud', + kindSsh: 'SSH', + kindLocalDesc: '由本应用管理的 Hermes 运行时。', + kindRemoteDesc: '可通过 HTTP(S) 访问的 Hermes 网关——局域网、Tailscale 或互联网。', + kindCloudDesc: '通过你的 Hermes Cloud 账户发现的托管实例。', + kindSshDesc: '通过 SSH 访问的 Hermes 安装。', + labelTitle: '名称', + labelDesc: '必填。此实例出现的所有位置都会显示该名称;必须唯一(例如“家庭服务器”、“工作笔记本”)。', + labelPlaceholder: '家庭服务器', + urlTitle: '网关 URL', + sshHostTitle: 'SSH 主机', + save: '保存连接', + saving: '保存中…', + cancel: '取消', + empty: '尚未注册任何连接。' + }, gateway: { loading: '正在加载网关设置...', unavailableTitle: '网关设置不可用', diff --git a/apps/desktop/src/lib/icons.ts b/apps/desktop/src/lib/icons.ts index 6b3d5f80cc7ba..943dd1870a16f 100644 --- a/apps/desktop/src/lib/icons.ts +++ b/apps/desktop/src/lib/icons.ts @@ -80,6 +80,7 @@ import { IconDots as MoreHorizontal, IconDots as MoreHorizontalIcon, IconDotsVertical as MoreVertical, + IconNetwork as Network, IconNotebook as NotebookTabs, IconPackage as Package, IconPalette as Palette, @@ -207,6 +208,7 @@ export { MoreHorizontal, MoreHorizontalIcon, MoreVertical, + Network, NotebookTabs, Package, Palette, From 0904f50e3e5c43b6fef5f192a6e8dbe545b7611f Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 23:01:45 -0700 Subject: [PATCH 233/376] fix(desktop): address connections-registry review findings MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Review fixes from #86679 comments (trevorgordon981, helix4u, kshitijk4poor): - Edit inheritance: mergeConnectionInput preserves fields the editor does not carry (cloud org, ssh remoteHermesPath/remoteProfile) so a rename no longer wipes them. When the payload carries an ssh host string, stored user/port are NOT inherited — the composite host field is authoritative, fixing the stale user/port resurrection on edit. - Token hygiene: tokens only persist on token-auth remotes; switching an entry to oauth (or cloud) clears the stale envelope. - Plain-text opt-in: the panel now surfaces the same consent dialog as Settings -> Gateway on keyring-less machines (registry list exposes secureTokenStorage; save retries with allowPlainTextToken after consent). - Registry test isolation: hermes:connections:test builds the probe directly from the registry entry instead of coercing against v1 connection.json — no more inheriting the v1 global token for a different host, and the local entry now probes the app-managed backend (never v1 remote/ssh state, so the test button can no longer trigger a v1 file write). - 'local' id reserved at the validation boundary: a crafted IPC payload can no longer replace the local entry via upsert. - Cloud creation hidden in the editor (a dialable cloud entry comes from the Cloud sign-in/discovery flow); migrated cloud entries stay editable. - First-run migration write is guarded: a failed write keeps the migrated registry in memory instead of hard-failing every connections IPC call. - uniqueLabel(): single label-dedup helper — counts up instead of "X 2 2", clamps 253-char migrated URL-host labels under LABEL_MAX; used by normalizeRegistry and both migration paths. - UI copy: staged-rollout note replaces the "side by side" claim; test failure toast leads with the failure wording; dropped unused i18n keys. Tests: +9 pure-module cases (reserved id, token-drop rules, merge inheritance, ssh host precedence, uniqueLabel); electron+settings suites 1355 passed. --- .../electron/connection-registry.test.ts | 90 ++++++++++++ apps/desktop/electron/connection-registry.ts | 105 ++++++++++++-- apps/desktop/electron/main.ts | 103 +++++++++++--- .../settings/connections-settings.test.tsx | 1 + .../src/app/settings/connections-settings.tsx | 129 ++++++++++++------ apps/desktop/src/global.d.ts | 3 + apps/desktop/src/i18n/en.ts | 6 +- apps/desktop/src/i18n/types.ts | 3 +- apps/desktop/src/i18n/zh.ts | 5 +- 9 files changed, 362 insertions(+), 83 deletions(-) diff --git a/apps/desktop/electron/connection-registry.test.ts b/apps/desktop/electron/connection-registry.test.ts index 88398cfbec756..eb4c025fbef6e 100644 --- a/apps/desktop/electron/connection-registry.test.ts +++ b/apps/desktop/electron/connection-registry.test.ts @@ -17,12 +17,14 @@ import { labelKey, labelSlug, LOCAL_CONNECTION_ID, + mergeConnectionInput, migrateV1ToRegistry, normalizeConnectionInput, normalizeRegistry, REGISTRY_VERSION, removeConnection, setPrimaryConnection, + uniqueLabel, upsertConnection } from './connection-registry' @@ -57,8 +59,96 @@ test('connectionIdForLabel suffixes on collision and never mints "local"', () => assert.equal(connectionIdForLabel('Local', []), 'local-2') }) +test('uniqueLabel counts up (never "X 2 2") and clamps long candidates', () => { + assert.equal(uniqueLabel('Homelab', []), 'Homelab') + assert.equal(uniqueLabel('Homelab', ['Homelab']), 'Homelab 2') + assert.equal(uniqueLabel('Homelab', ['Homelab', 'Homelab 2']), 'Homelab 3') + // Case-insensitive collision detection. + assert.equal(uniqueLabel('homelab', ['HOMELAB']), 'homelab 2') + + const long = 'x'.repeat(300) + assert.ok(uniqueLabel(long, []).length <= 64) + assert.ok(uniqueLabel(long, [uniqueLabel(long, [])]).length <= 64) +}) + // --- normalizeConnectionInput --- +test('save rejects the reserved "local" id on non-local kinds', () => { + assert.throws( + () => normalizeConnectionInput({ id: 'local', kind: 'remote', label: 'Sneaky', url: 'http://x:1' }, emptyRegistry()), + /reserved/ + ) +}) + +test('token only persists on token-auth remotes; oauth/cloud drop it', () => { + const registry = emptyRegistry() + + const tokenAuth = normalizeConnectionInput( + { kind: 'remote', label: 'A', url: 'http://a:1', authMode: 'token', token: { enc: 'x' } }, + registry + ) + + assert.deepEqual(tokenAuth.token, { enc: 'x' }) + + const oauth = normalizeConnectionInput( + { kind: 'remote', label: 'B', url: 'http://b:1', authMode: 'oauth', token: { enc: 'x' } }, + registry + ) + + assert.equal(oauth.token, undefined) + + const cloud = normalizeConnectionInput( + { kind: 'cloud', label: 'C', url: 'https://c.hermes.cloud', authMode: 'oauth', token: { enc: 'x' } }, + registry + ) + + assert.equal(cloud.token, undefined) +}) + +// --- mergeConnectionInput (edit inheritance) --- + +test('merge preserves fields the editor does not carry (org, ssh extras)', () => { + const cloud = { authMode: 'oauth' as const, id: 'c', kind: 'cloud' as const, label: 'Cloud', org: 'nous', url: 'https://a.cloud' } + const renamed = mergeConnectionInput({ id: 'c', kind: 'cloud', label: 'Renamed', url: 'https://a.cloud' }, cloud) + + assert.equal(renamed.org, 'nous') + + const ssh = { + host: 'homelab.lan', + id: 's', + keyPath: '/k/id', + kind: 'ssh' as const, + label: 'Box', + port: 2222, + remoteHermesPath: '/opt/hermes', + remoteProfile: 'research', + user: 'k' + } + + const labelOnly = mergeConnectionInput({ id: 's', kind: 'ssh', label: 'Renamed box' }, ssh) + + assert.equal(labelOnly.remoteHermesPath, '/opt/hermes') + assert.equal(labelOnly.remoteProfile, 'research') + assert.equal(labelOnly.host, 'homelab.lan') + assert.equal(labelOnly.user, 'k') + assert.equal(labelOnly.port, 2222) +}) + +test('merge: a supplied ssh host string beats stored user/port', () => { + const ssh = { host: 'spark1', id: 's', kind: 'ssh' as const, label: 'Spark', port: 2222, user: 'tek' } + const merged = mergeConnectionInput({ host: 'admin@newbox:2200', id: 's', kind: 'ssh', label: 'Spark' }, ssh) + + // Stored user/port must NOT ride along — the host string is authoritative. + assert.equal(merged.user, undefined) + assert.equal(merged.port, undefined) + + const entry = normalizeConnectionInput(merged, emptyRegistry()) + + assert.equal(entry.host, 'newbox') + assert.equal(entry.user, 'admin') + assert.equal(entry.port, 2200) +}) + test('save rejects a missing label with a device-name message', () => { assert.throws( () => normalizeConnectionInput({ kind: 'remote', label: ' ', url: 'http://10.0.0.5:9119' }, emptyRegistry()), diff --git a/apps/desktop/electron/connection-registry.ts b/apps/desktop/electron/connection-registry.ts index f1fca1ff17b04..7f4e33b3e6314 100644 --- a/apps/desktop/electron/connection-registry.ts +++ b/apps/desktop/electron/connection-registry.ts @@ -78,6 +78,33 @@ export function labelKey(label: string): string { .toLowerCase() } +/** + * Derive a registry-unique label from a candidate: clamps to LABEL_MAX (a + * migrated URL host can exceed it, which would fail validation on any later + * edit) and suffixes " 2" / " 3" / … on collision. The single home of the + * label-dedup rule — normalizeRegistry and the migration both use it. + */ +export function uniqueLabel(candidate: string, taken: Iterable): string { + const used = new Set([...taken].map(labelKey)) + + // Reserve room for a collision suffix so the suffixed form stays in-bounds. + const base = String(candidate || '') + .trim() + .slice(0, LABEL_MAX - 4) + + if (!used.has(labelKey(base))) { + return base + } + + for (let n = 2; ; n += 1) { + const suffixed = `${base} ${n}` + + if (!used.has(labelKey(suffixed))) { + return suffixed + } + } +} + /** Kebab-slug of a label for ids and @handles. Never empty for a non-empty label. */ export function labelSlug(label: string): string { const slug = String(label || '') @@ -168,6 +195,15 @@ export function normalizeConnectionInput(input: ConnectionInput, registry: Conne return { id: LOCAL_CONNECTION_ID, kind: 'local', label } } + // The reserved local id can never be claimed by a non-local entry — a + // crafted IPC payload ({id:'local', kind:'remote', …}) would otherwise + // replace the local entry via upsert and break the exactly-one-local + // invariant. connectionIdForLabel never mints 'local'; reject it when + // supplied, too. + if (input.id === LOCAL_CONNECTION_ID) { + throw new Error('The id "local" is reserved for the local connection.') + } + const id = input.id || connectionIdForLabel(label, registry.connections.map(c => c.id)) if (kind === 'ssh') { @@ -196,7 +232,11 @@ export function normalizeConnectionInput(input: ConnectionInput, registry: Conne const authMode = normAuthMode(input.authMode) const entry: RegistryConnection = { id, kind, label, url, authMode } - if (input.token !== undefined) { + // A token is only meaningful for token-auth remotes. Dropping it here is + // what clears the stale envelope when an entry is switched token→oauth + // (or is a cloud entry, which authenticates via the portal session) — + // otherwise dead secret material rides along on the edited entry. + if (input.token !== undefined && kind === 'remote' && authMode === 'token') { entry.token = input.token } @@ -212,6 +252,50 @@ export function normalizeConnectionInput(input: ConnectionInput, registry: Conne throw new Error(`Unknown connection kind: ${String(kind)}`) } +/** + * Merge a (possibly partial) edit payload over the stored entry so fields the + * editor doesn't carry survive a save. Renaming a migrated cloud entry must + * not drop its `org` (downstream update-fanout uses it to skip + * platform-managed instances), and renaming an ssh entry must not drop + * `remoteHermesPath`/`remoteProfile`. Only fields the payload explicitly + * carries (non-undefined) override; `token` is deliberately NOT merged here — + * the caller owns secret handling. + */ +export function mergeConnectionInput(input: ConnectionInput, existing?: null | RegistryConnection): ConnectionInput { + if (!existing || existing.kind !== input.kind) { + return input + } + + const merged: ConnectionInput = { ...input } + + const inherit = (field: keyof ConnectionInput & keyof RegistryConnection) => { + if (merged[field] === undefined && existing[field] !== undefined) { + ;(merged as unknown as Record)[field] = existing[field] + } + } + + inherit('url') + inherit('authMode') + inherit('org') + inherit('host') + inherit('keyPath') + inherit('remoteHermesPath') + inherit('remoteProfile') + + // ssh user/port: the editor shows ONE composite host field (user@host:port), + // and normalizeSshConfig gives explicit user/port fields precedence over the + // parsed host string. Inheriting stored user/port alongside a NEW host string + // would resurrect the old values over what the user just typed — so when the + // payload carries a host, the host string is authoritative and stored + // user/port are NOT inherited. + if (input.host === undefined || !String(input.host).trim()) { + inherit('user') + inherit('port') + } + + return merged +} + // ── Registry-level operations (all pure: return a new registry) ──────────── function localEntry(label = 'This device'): RegistryConnection { @@ -253,9 +337,7 @@ export function normalizeRegistry(raw: unknown): ConnectionRegistry { kind === 'ssh' ? String(entry.host || 'ssh') : hostLabelFromBaseUrl(String(entry.url || '')) || String(kind) } - while (seenLabels.has(labelKey(label))) { - label = `${label} 2` - } + label = uniqueLabel(label, seenLabels) let id = kind === 'local' ? LOCAL_CONNECTION_ID : String(entry.id || '').trim() @@ -348,11 +430,10 @@ export function migrateV1ToRegistry(v1: unknown): ConnectionRegistry { return existing } - let label = hostLabelFromBaseUrl(url) || (kind === 'cloud' ? 'Hermes Cloud' : 'Remote gateway') - - while (connections.some(c => labelKey(c.label) === labelKey(label))) { - label = `${label} 2` - } + const label = uniqueLabel( + hostLabelFromBaseUrl(url) || (kind === 'cloud' ? 'Hermes Cloud' : 'Remote gateway'), + connections.map(c => c.label) + ) const entry: RegistryConnection = { id: connectionIdForLabel(label, connections.map(c => c.id)), @@ -392,11 +473,7 @@ export function migrateV1ToRegistry(v1: unknown): ConnectionRegistry { return existing } - let label = ssh.host - - while (connections.some(c => labelKey(c.label) === labelKey(label))) { - label = `${label} 2` - } + const label = uniqueLabel(ssh.host, connections.map(c => c.label)) const { mode: _mode, ...sshFields } = ssh diff --git a/apps/desktop/electron/main.ts b/apps/desktop/electron/main.ts index 0af57083167a5..52b28ff6c251c 100644 --- a/apps/desktop/electron/main.ts +++ b/apps/desktop/electron/main.ts @@ -84,6 +84,7 @@ import { tokenPreview } from './connection-config' import { + mergeConnectionInput, migrateV1ToRegistry, normalizeConnectionInput, normalizeRegistry, @@ -7590,9 +7591,21 @@ function readDesktopConnectionsRegistry() { if (mtime === null) { // First run on this build: import the v1 single-connection config. The v1 - // file is NOT modified or deleted — older builds keep reading it. + // file is NOT modified or deleted — older builds keep reading it. The + // migration is deterministic over the v1 input, so even if two processes + // race the first run (updater relaunch, second window), both derive the + // same registry and the later atomic write is a no-op content-wise. registry = migrateV1ToRegistry(readDesktopConnectionConfig()) - writeDesktopConnectionsRegistry(registry) + + try { + writeDesktopConnectionsRegistry(registry) + } catch { + // Write failed (full disk, read-only userData). Keep the migrated + // registry in memory so list/save keep working this session instead of + // hard-failing every hermes:connections:* call. + connectionRegistryCache = registry + connectionRegistryCacheMtime = null + } return connectionRegistryCache } @@ -7638,18 +7651,33 @@ function sanitizeRegistryConnection(entry) { } function sanitizeConnectionsRegistry(registry = readDesktopConnectionsRegistry()) { + // Same keyring probe the v1 sanitize exposes: lets the Connections panel + // offer the plain-text opt-in on keyring-less Linux instead of failing. + let secureTokenStorage = false + + try { + secureTokenStorage = Boolean(safeStorage.isEncryptionAvailable()) + } catch { + secureTokenStorage = false + } + return { version: registry.version, primary: registry.primary, + secureTokenStorage, connections: registry.connections.map(sanitizeRegistryConnection) } } /** * Save (create or edit) a registry connection from a renderer payload. - * Token handling mirrors coerceDesktopConnectionConfig: an incoming plaintext - * token is encrypted (with the same plain-text opt-in seam); an absent token - * field inherits the stored envelope on edit. + * Edits merge over the stored entry (mergeConnectionInput) so fields the + * editor doesn't carry — cloud `org`, ssh `remoteHermesPath`/`remoteProfile` — + * survive a rename. Token handling mirrors coerceDesktopConnectionConfig: an + * incoming plaintext token is encrypted (honoring the same allowPlainTextToken + * opt-in seam as Settings → Gateway); an absent token field inherits the + * stored envelope on edit; switching auth away from 'token' clears it + * (normalizeConnectionInput drops tokens on non-token entries). */ function saveRegistryConnection(input: any = {}) { const registry = readDesktopConnectionsRegistry() @@ -7664,7 +7692,8 @@ function saveRegistryConnection(input: any = {}) { encryptSecret: encryptDesktopSecret }) - const entry = normalizeConnectionInput({ ...input, token }, registry) + const merged = mergeConnectionInput({ ...input, token }, existing) + const entry = normalizeConnectionInput(merged, registry) // Token-auth remotes must actually have a token to be dialable. OAuth and // cloud entries authenticate via cookies/native tokens instead. @@ -11352,12 +11381,8 @@ ipcMain.handle('hermes:connections:test', async (_event, id) => { throw new Error(`No connection with id "${String(id || '')}".`) } - // Reuse the existing probe stack by mapping the registry entry onto the - // settings-payload shape testDesktopConnectionConfig already understands. - if (entry.kind === 'local') { - return testDesktopConnectionConfig({ mode: 'local' }) - } - + // The ssh probe path in testDesktopConnectionConfig never consults v1 + // connection state, so mapping the entry onto it is safe. if (entry.kind === 'ssh') { return testDesktopConnectionConfig({ mode: 'ssh', @@ -11369,13 +11394,53 @@ ipcMain.handle('hermes:connections:test', async (_event, id) => { }) } - return testDesktopConnectionConfig({ - mode: entry.kind, - remoteUrl: entry.url, - remoteAuthMode: entry.authMode, - remoteToken: decryptDesktopSecret(entry.token) || undefined, - cloudOrg: entry.org - }) + // Remote/cloud/local probe built DIRECTLY from the registry entry. Routing + // through coerceDesktopConnectionConfig would use v1 connection.json as the + // `existing` base: an entry with a broken/absent token would inherit the v1 + // global remote's token and send it to THIS entry's URL (cross-host + // credential transmission + a false "reachable"), and testing the local + // entry would probe whatever v1's global mode points at instead of the + // app-managed local backend. + let baseUrl + let token = null + let authMode = 'token' + + if (entry.kind === 'local') { + const local = await startHermes() + baseUrl = local.baseUrl + token = local.token + authMode = normAuthMode(local.authMode) + } else { + baseUrl = normalizeRemoteBaseUrl(entry.url) + authMode = normAuthMode(entry.authMode) + + if (authMode !== 'oauth') { + token = decryptDesktopSecret(entry.token) + + if (!token) { + throw new Error('This connection has no saved session token. Edit the connection and paste one.') + } + } + } + + const status = (await fetchJson(`${baseUrl}/api/status`, token, { timeoutMs: 8_000 })) as any + + // Same HTTP+WS two-leg check as testDesktopConnectionConfig: HTTP alone is + // a false positive when the WebSocket leg is blocked. + const wsUrl = await resolveTestWsUrl(baseUrl, authMode, token, { mintTicket: mintGatewayWsTicket }) + + if (wsUrl && typeof globalThis.WebSocket === 'function') { + const probe = await probeGatewayWebSocket(wsUrl, { WebSocketImpl: globalThis.WebSocket }) + + if (!probe.ok) { + throw new Error( + `Reached the gateway over HTTP, but the live WebSocket (/api/ws) connection failed: ${probe.reason} ` + + 'The HTTP check can pass while the WebSocket is blocked by a proxy, firewall, or gateway auth/origin guard.' + ) + } + } + + return { ok: true, baseUrl, version: status?.version || null } }) ipcMain.handle('hermes:connection-config:probe', async (_event, rawUrl) => probeRemoteAuthMode(rawUrl)) ipcMain.handle('hermes:connection-config:oauth-login', async (_event, rawUrl) => { diff --git a/apps/desktop/src/app/settings/connections-settings.test.tsx b/apps/desktop/src/app/settings/connections-settings.test.tsx index aaaba36d1b5be..a5c4ba2006b3e 100644 --- a/apps/desktop/src/app/settings/connections-settings.test.tsx +++ b/apps/desktop/src/app/settings/connections-settings.test.tsx @@ -25,6 +25,7 @@ const registry: DesktopConnectionsRegistry = { } ], primary: 'local', + secureTokenStorage: true, version: 2 } diff --git a/apps/desktop/src/app/settings/connections-settings.tsx b/apps/desktop/src/app/settings/connections-settings.tsx index ef793cd1ea688..9f504b2ef8be4 100644 --- a/apps/desktop/src/app/settings/connections-settings.tsx +++ b/apps/desktop/src/app/settings/connections-settings.tsx @@ -32,8 +32,6 @@ interface EditorState { authMode: 'oauth' | 'token' token: string host: string - user: string - port: string keyPath: string } @@ -45,15 +43,18 @@ function editorFromConnection(conn: DesktopRegistryConnection): EditorState { url: conn.url || '', authMode: conn.authMode || 'token', token: '', - host: conn.host || '', - user: conn.user || '', - port: conn.port ? String(conn.port) : '', + // Reconstruct the composite the single ssh host field displays. The save + // payload sends ONLY this string (never separate user/port), because + // normalizeSshConfig gives explicit user/port fields precedence over the + // parsed host string — sending stored user/port alongside a retyped host + // would silently resurrect the old values. + host: conn.host ? `${conn.user ? `${conn.user}@` : ''}${conn.host}${conn.port ? `:${conn.port}` : ''}` : '', keyPath: conn.keyPath || '' } } function emptyEditor(kind: DesktopConnectionKind): EditorState { - return { id: null, kind, label: '', url: '', authMode: 'token', token: '', host: '', user: '', port: '', keyPath: '' } + return { id: null, kind, label: '', url: '', authMode: 'token', token: '', host: '', keyPath: '' } } /** @@ -72,6 +73,7 @@ export function ConnectionsSettings() { const [busyId, setBusyId] = useState(null) const [testingId, setTestingId] = useState(null) const [removeTarget, setRemoveTarget] = useState(null) + const [plainTextConfirm, setPlainTextConfirm] = useState(false) const bridge = window.hermesDesktop?.connections @@ -97,46 +99,69 @@ export function ConnectionsSettings() { void load() }, [load]) - const save = useCallback(async () => { - if (!bridge || !editor) { - return - } + const save = useCallback( + async (allowPlainTextToken = false) => { + if (!bridge || !editor) { + return + } - setSaving(true) + setSaving(true) - try { - const payload: DesktopRegistryConnectionInput = { - kind: editor.kind, - label: editor.label - } + try { + const payload: DesktopRegistryConnectionInput = { + kind: editor.kind, + label: editor.label + } - if (editor.id) { - payload.id = editor.id - } + if (editor.id) { + payload.id = editor.id + } - if (editor.kind === 'remote' || editor.kind === 'cloud') { - payload.url = editor.url - payload.authMode = editor.authMode + if (editor.kind === 'remote' || editor.kind === 'cloud') { + payload.url = editor.url + payload.authMode = editor.authMode + + if (editor.token.trim()) { + payload.token = editor.token.trim() + } + + if (allowPlainTextToken) { + payload.allowPlainTextToken = true + } + } else if (editor.kind === 'ssh') { + // The composite host string (user@host:port) is the single source + // of truth — never send separate user/port (see editorFromConnection). + payload.host = editor.host + payload.keyPath = editor.keyPath || undefined + } - if (editor.token.trim()) { - payload.token = editor.token.trim() + const result = await bridge.save(payload) + setRegistry(result.registry) + setEditor(null) + setPlainTextConfirm(false) + } catch (err) { + // Keyring-less machine and the user hasn't consented to plain-text + // storage yet: raise the same opt-in dialog Settings → Gateway uses + // instead of dead-ending the save. + if ( + !allowPlainTextToken && + registry?.secureTokenStorage === false && + editor.kind === 'remote' && + editor.authMode === 'token' && + editor.token.trim() + ) { + setPlainTextConfirm(true) + + return } - } else if (editor.kind === 'ssh') { - payload.host = editor.host - payload.user = editor.user || undefined - payload.port = editor.port.trim() ? Number(editor.port) : null - payload.keyPath = editor.keyPath || undefined - } - const result = await bridge.save(payload) - setRegistry(result.registry) - setEditor(null) - } catch (err) { - notifyError(err, s.saveFailed) - } finally { - setSaving(false) - } - }, [bridge, editor, s.saveFailed]) + notifyError(err, s.saveFailed) + } finally { + setSaving(false) + } + }, + [bridge, editor, registry?.secureTokenStorage, s.saveFailed] + ) const remove = useCallback(async () => { if (!bridge || !removeTarget) { @@ -191,7 +216,7 @@ export function ConnectionsSettings() { if (reachable) { notify({ title: conn.label, message: s.testOk }) } else { - notifyError(new Error(result.error || s.testFailed), conn.label) + notifyError(new Error(result.error || conn.label), s.testFailed) } } catch (err) { notifyError(err, s.testFailed) @@ -216,7 +241,12 @@ export function ConnectionsSettings() { return ( -

{s.intro}

+

{s.intro}

+ {/* Storage-only slice: be explicit that routing consumption is staged so + "Make primary" isn't read as an immediate connection switch. */} +

+ {s.stagedNote} +

{!registry || registry.connections.length === 0 ? ( @@ -293,7 +323,11 @@ export function ConnectionsSettings() { {editor ? (
- {(['remote', 'cloud', 'ssh'] as const).map(kind => ( + {/* Cloud creation is deliberately absent: a dialable cloud entry + comes from the Hermes Cloud sign-in/discovery flow (Settings → + Gateway), not a hand-typed URL. Migrated/discovered cloud + entries remain editable (kind buttons are disabled on edit). */} + {(editor.kind === 'cloud' ? (['cloud'] as const) : (['remote', 'ssh'] as const)).map(kind => ( diff --git a/apps/desktop/src/app/chat/sidebar/virtual-session-list.tsx b/apps/desktop/src/app/chat/sidebar/virtual-session-list.tsx index 2a10fd9a3f16a..48cbe68396087 100644 --- a/apps/desktop/src/app/chat/sidebar/virtual-session-list.tsx +++ b/apps/desktop/src/app/chat/sidebar/virtual-session-list.tsx @@ -161,7 +161,9 @@ export const VirtualSessionList: FC = ({ )} ref={scrollerRef} > -
{rows}
+
+ {rows} +
) } diff --git a/apps/desktop/src/app/cron/cron-actions.test.ts b/apps/desktop/src/app/cron/cron-actions.test.ts index 61e6dd1aa305f..654f65b6b73a1 100644 --- a/apps/desktop/src/app/cron/cron-actions.test.ts +++ b/apps/desktop/src/app/cron/cron-actions.test.ts @@ -143,9 +143,7 @@ describe('mutateAndRefreshCronJobs', () => { it('allows overlapping same-profile mutations to authoritatively refresh', async () => { const first = deferred() const second = deferred() - getCronJobs - .mockResolvedValueOnce([{ id: 'after-second' }]) - .mockResolvedValueOnce([{ id: 'after-both' }]) + getCronJobs.mockResolvedValueOnce([{ id: 'after-second' }]).mockResolvedValueOnce([{ id: 'after-both' }]) const firstResult = mutateAndRefreshCronJobs('work', () => first.promise) const secondResult = mutateAndRefreshCronJobs('work', () => second.promise) diff --git a/apps/desktop/src/app/cron/cron-actions.ts b/apps/desktop/src/app/cron/cron-actions.ts index 95cc07c63e9d6..7fafe13c59cde 100644 --- a/apps/desktop/src/app/cron/cron-actions.ts +++ b/apps/desktop/src/app/cron/cron-actions.ts @@ -18,10 +18,7 @@ export interface CronMutationRefreshResult extends CronTriggerRefreshResult { value: T | null } -async function refreshForGeneration( - profile: string, - request: CronJobsRequest -): Promise { +async function refreshForGeneration(profile: string, request: CronJobsRequest): Promise { try { const jobs = await getCronJobs(profile) @@ -90,9 +87,7 @@ export async function triggerAndRefreshCronJobs( jobId: string, profile: 'all' | string ): Promise { - const { value: _value, ...result } = await mutateAndRefreshCronJobs(profile, () => - triggerCronJob(jobId) - ) + const { value: _value, ...result } = await mutateAndRefreshCronJobs(profile, () => triggerCronJob(jobId)) return result -} \ No newline at end of file +} diff --git a/apps/desktop/src/app/cron/index.tsx b/apps/desktop/src/app/cron/index.tsx index 1b6e6ab1566d2..9bf6923b5c9e3 100644 --- a/apps/desktop/src/app/cron/index.tsx +++ b/apps/desktop/src/app/cron/index.tsx @@ -541,9 +541,7 @@ export function CronView({ onClose, onOpenSession, setStatusbarItemGroup: _setSt setDeleting(true) try { - const { refreshError, stale } = await mutateAndRefreshCronJobs(profile, () => - deleteCronJob(pendingDelete.id) - ) + const { refreshError, stale } = await mutateAndRefreshCronJobs(profile, () => deleteCronJob(pendingDelete.id)) if (stale) { return @@ -564,7 +562,11 @@ export function CronView({ onClose, onOpenSession, setStatusbarItemGroup: _setSt async function handleEditorSave(values: EditorValues) { if (editor.mode === 'create') { - const { value: created, refreshError, stale } = await mutateAndRefreshCronJobs(profile, () => + const { + value: created, + refreshError, + stale + } = await mutateAndRefreshCronJobs(profile, () => createCronJob({ prompt: values.prompt, schedule: values.schedule, @@ -586,7 +588,11 @@ export function CronView({ onClose, onOpenSession, setStatusbarItemGroup: _setSt } else if (editor.mode === 'edit') { const scriptOnlyJob = jobIsScriptOnly(editor.job) - const { value: updated, refreshError, stale } = await mutateAndRefreshCronJobs(profile, () => + const { + value: updated, + refreshError, + stale + } = await mutateAndRefreshCronJobs(profile, () => updateCronJob(editor.job.id, cronEditorUpdates(values, { scriptOnlyJob })) ) @@ -612,7 +618,11 @@ export function CronView({ onClose, onOpenSession, setStatusbarItemGroup: _setSt async function handleBlueprintCreate(blueprint: AutomationBlueprint, values: Record) { const writableProfile = profileScope === ALL_PROFILES ? 'default' : profileScope - const { value: job, refreshError, stale } = await mutateAndRefreshCronJobs(profile, () => + const { + value: job, + refreshError, + stale + } = await mutateAndRefreshCronJobs(profile, () => instantiateAutomationBlueprint({ blueprint: blueprint.key, values }, writableProfile) ) diff --git a/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event.ts b/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event.ts index 238f3c9547be3..15eb0d6171c2e 100644 --- a/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event.ts +++ b/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event.ts @@ -12,7 +12,11 @@ import { translateNow } from '@/i18n' import { type GatewayEventPayload, textPart } from '@/lib/chat-messages' import { coerceGatewayText, coerceThinkingText, normalizePersonalityValue } from '@/lib/chat-runtime' import { playCompletionSound } from '@/lib/completion-sound' -import { approvalReplaySessionId, resolveGatewayEventSessionId, UNSCOPED_STREAM_EVENT_TYPES } from '@/lib/gateway-events' +import { + approvalReplaySessionId, + resolveGatewayEventSessionId, + UNSCOPED_STREAM_EVENT_TYPES +} from '@/lib/gateway-events' import { triggerHaptic } from '@/lib/haptics' import { modelOptionsQueryKey } from '@/lib/model-options' import { isProviderSetupErrorMessage } from '@/lib/provider-setup-errors' diff --git a/apps/desktop/src/components/assistant-ui/thread/list.tsx b/apps/desktop/src/components/assistant-ui/thread/list.tsx index 9db5aedcb16fd..94afcf047445a 100644 --- a/apps/desktop/src/components/assistant-ui/thread/list.tsx +++ b/apps/desktop/src/components/assistant-ui/thread/list.tsx @@ -577,7 +577,11 @@ const ThreadMessageListInner: FC = ({ // can be overwritten by another mounted pane; leave a scrolled-up reader // exactly where they were. useEffect( - () => subscribeToThreadForeground(() => isAtBottom, () => void scrollToBottom()), + () => + subscribeToThreadForeground( + () => isAtBottom, + () => void scrollToBottom() + ), [isAtBottom, scrollToBottom] ) diff --git a/apps/desktop/src/store/cron.test.ts b/apps/desktop/src/store/cron.test.ts index e4db02e1f5bc0..2e69343548c6e 100644 --- a/apps/desktop/src/store/cron.test.ts +++ b/apps/desktop/src/store/cron.test.ts @@ -1,12 +1,6 @@ import { beforeEach, describe, expect, it } from 'vitest' -import { - $cronJobs, - beginCronJobsRequest, - commitCronJobsRequest, - setCronJobs, - updateCronJobs -} from './cron' +import { $cronJobs, beginCronJobsRequest, commitCronJobsRequest, setCronJobs, updateCronJobs } from './cron' const oldJob = { id: 'old' } as never const newJob = { id: 'new' } as never diff --git a/apps/desktop/src/store/suggestion-providers/skill.test.ts b/apps/desktop/src/store/suggestion-providers/skill.test.ts index 3661a0aa52804..f85c29675b87a 100644 --- a/apps/desktop/src/store/suggestion-providers/skill.test.ts +++ b/apps/desktop/src/store/suggestion-providers/skill.test.ts @@ -91,12 +91,12 @@ describe('skillTouchedInMessages', () => { }) it('matches qualified skill names (category/name, plugin:name)', () => { - expect(skillTouchedInMessages('hermes-agent-dev', [toolCall('skill_view', { name: 'github/hermes-agent-dev' })])).toBe( - true - ) - expect(skillTouchedInMessages('writing-plans', [toolCall('skill_view', { name: 'superpowers:writing-plans' })])).toBe( - true - ) + expect( + skillTouchedInMessages('hermes-agent-dev', [toolCall('skill_view', { name: 'github/hermes-agent-dev' })]) + ).toBe(true) + expect( + skillTouchedInMessages('writing-plans', [toolCall('skill_view', { name: 'superpowers:writing-plans' })]) + ).toBe(true) }) it('falls back to argsText when args were not parsed', () => { diff --git a/apps/desktop/src/store/suggestion-providers/skill.ts b/apps/desktop/src/store/suggestion-providers/skill.ts index 3e46e943cbe72..6b203aad55bf8 100644 --- a/apps/desktop/src/store/suggestion-providers/skill.ts +++ b/apps/desktop/src/store/suggestion-providers/skill.ts @@ -149,7 +149,11 @@ export function skillTouchedInMessages(skillName: string, messages: readonly Cha for (const message of messages) { for (const part of message.parts) { - if (part.type === 'tool-call' && SKILL_TOOL_NAMES.has(part.toolName) && argNamesSkill(skillArgName(part), skillName)) { + if ( + part.type === 'tool-call' && + SKILL_TOOL_NAMES.has(part.toolName) && + argNamesSkill(skillArgName(part), skillName) + ) { return true } diff --git a/tests-js/react-dom-pair-compat.test.ts b/tests-js/react-dom-pair-compat.test.ts index 839d37f83df4b..46dec4280e141 100644 --- a/tests-js/react-dom-pair-compat.test.ts +++ b/tests-js/react-dom-pair-compat.test.ts @@ -50,6 +50,7 @@ function workspaceManifests(): { name: string, manifest: Manifest }[] { for (const pattern of patterns) { // The globs in use are plain paths or a single trailing ``/*``. const parent = pattern.endsWith('/*') ? path.join(REPO_ROOT, pattern.slice(0, -2)) : null + const dirs = parent === null ? [pattern] : fs.existsSync(parent) @@ -78,6 +79,7 @@ test('workspaces declaring react and react-dom pin them to the same exact versio if (react !== reactDom) { offenders.push(`${name} declares react"${react}" but react-dom"${reactDom}"`) + continue } From 30c469b15313711d47c45e7175d6ef5c8437f1ed Mon Sep 17 00:00:00 2001 From: EvanProgramming Date: Sat, 15 Aug 2026 12:33:43 +0800 Subject: [PATCH 235/376] fix(gateway): spare pidfile-less Scheduled-Task gateways from the orphan reaper on Windows (#83683) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit On Windows _get_service_pids() is empty (no systemd/launchd query), so a Scheduled-Task-supervised gateway whose gateway.pid record is missing or stale is invisible to both the service-PID and recorded-PID exclusions the reaper already applies (#86658) — and gets SIGTERM'd on every desktop open (#86098 class, pidfile-less path). Add a Windows-only backstop: any reaper candidate whose parent chain reaches services.exe (the Task Scheduler launches tasks under the services tree) is spared even with no pidfile. The backstop is deliberately inert on POSIX: every process there has PID 1 (launchd/init/systemd) in its ancestry — and a genuine orphan is reparented directly to PID 1 — so supervisor-name ancestry carries zero supervision signal and would disable the reaper entirely on macOS/WSL (#51325, #75936). POSIX supervised gateways are already covered pidfile-independently by the _get_service_pids() exclusion. Known limitation (fail-open, documented): if the Task-launched bootstrap parent has already exited, Windows does not reparent the gateway, the chain breaks before services.exe, and the gateway is treated as an orphan. Salvaged from #86702 by @EvanProgramming (authorship preserved); reduced to the genuinely-new Windows backstop — the PR's other two hunks were already merged on main via #86658 (one in a strictly stronger full-parent-chain form) and its POSIX ancestry checks were dropped as unsound (verified empirically: a true double-fork orphan's psutil parent IS launchd). --- hermes_cli/gateway.py | 58 +++++++++++++- tests/hermes_cli/test_gateway.py | 129 +++++++++++++++++++++++++++++++ 2 files changed, 185 insertions(+), 2 deletions(-) diff --git a/hermes_cli/gateway.py b/hermes_cli/gateway.py index e8634ea12da1f..b9df20436bb14 100644 --- a/hermes_cli/gateway.py +++ b/hermes_cli/gateway.py @@ -1525,6 +1525,54 @@ def kill_gateway_processes( return killed +_REAPER_SUPERVISOR_WALK_LIMIT = 12 + + +def _reaper_candidate_is_supervisor_owned(pid: int) -> bool: + """True when ``pid`` is a gateway process owned by the Windows Task Scheduler. + + Windows-only backstop for the orphan reaper: ``_get_service_pids()`` is + empty on Windows (no systemd/launchd query), so a Scheduled-Task gateway + whose ``gateway.pid`` record is missing or stale is invisible to both the + service-PID and recorded-PID exclusions — yet it is alive and supervised. + Scheduled Tasks run under the services tree, so a candidate whose parent + chain reaches ``services.exe`` is spared even with no pidfile (#83683, + #86098). + + This check is deliberately NOT applied on POSIX: there, every process has + PID 1 (launchd / init / systemd) in its ancestry — and a genuine orphan is + *reparented directly to PID 1* — so supervisor-name ancestry carries zero + signal and would spare every orphan the reaper exists to kill (#51325, + #75936). POSIX supervised gateways are already covered pidfile- + independently by the ``_get_service_pids()`` exclusion. + + Known limitation (fail-open): if the Task-launched bootstrap parent has + already exited, Windows does not reparent the gateway, the chain breaks + before ``services.exe``, and the gateway is treated as an orphan. Any + error (process gone, psutil unavailable) is likewise treated as "not + owned" so a genuine orphan is still reaped. + """ + if not is_windows(): + return False + try: + import psutil # type: ignore + + parent = psutil.Process(pid).parent() + for _ in range(_REAPER_SUPERVISOR_WALK_LIMIT): + if parent is None: + break + try: + name = (parent.name() or "").lower() + except Exception: + name = "" + if name == "services.exe": + return True + parent = parent.parent() + except Exception: + pass + return False + + def _reap_unsupervised_gateway_orphans(extra_exclude: set | None = None) -> bool: """Kill no-supervisor gateway orphans the pidfile/runtime record can't see. @@ -1595,8 +1643,14 @@ def _reap_unsupervised_gateway_orphans(extra_exclude: set | None = None) -> bool pass try: # find_gateway_pids() includes no-supervisor `gateway restart` runtimes - # for the current profile when no systemd supervisor is present. - orphans = [p for p in find_gateway_pids(exclude_pids=own) if p and p > 0] + # for the current profile when no systemd supervisor is present. On + # Windows, additionally drop any candidate the Task Scheduler owns — + # the pidfile-less gap neither exclusion above can see (#83683, #86098). + orphans = [ + p + for p in find_gateway_pids(exclude_pids=own) + if p and p > 0 and not _reaper_candidate_is_supervisor_owned(p) + ] except Exception: return False if not orphans: diff --git a/tests/hermes_cli/test_gateway.py b/tests/hermes_cli/test_gateway.py index d47c647506bee..d050f307fc265 100644 --- a/tests/hermes_cli/test_gateway.py +++ b/tests/hermes_cli/test_gateway.py @@ -604,6 +604,135 @@ def test_windows_no_orphans_when_only_recorded_gateway_running(self, monkeypatch assert killed_pids == [] # nothing was killed +class TestReaperCandidateIsSupervisorOwned: + """Regression for the Windows pidfile-less supervisor-owned case (#83683). + + On Windows ``_get_service_pids()`` is empty and a Scheduled-Task gateway + that lost ``gateway.pid`` is invisible to both the service-PID and + recorded-PID exclusions — the backstop spares it via services.exe + ancestry. On POSIX the backstop must be inert: every process (and + especially a genuine orphan, which is reparented to PID 1) has + launchd/init in its ancestry, so ancestry carries no supervision signal + there (#51325, #75936). + """ + + @staticmethod + def _install_fake_psutil(monkeypatch, by_pid): + fake_psutil = SimpleNamespace(Process=lambda pid: by_pid[pid]) + monkeypatch.setitem(sys.modules, "psutil", fake_psutil) + + def test_windows_scheduled_task_gateway_spared_without_pidfile(self, monkeypatch): + """A Windows gateway launched by the Scheduled Task is spared even when + gateway.pid is missing — the supervisor-owned backstop catches it.""" + gateway_pid = 52615 + bootstrap_pid = 52616 # Task-launched `hermes gateway run` bootstrap + orphan_pid = 99998 # a genuine orphan that SHOULD be reaped + + monkeypatch.setattr(gateway, "is_windows", lambda: True) + monkeypatch.setattr(gateway, "is_macos", lambda: False) + monkeypatch.setattr(gateway, "supports_systemd_services", lambda: False) + # No pidfile => get_running_pid() returns None. + monkeypatch.setattr("gateway.status.get_running_pid", lambda: None) + # _get_service_pids() is empty on Windows. + monkeypatch.setattr(gateway, "_get_service_pids", lambda: set()) + + # Parent chain: gateway -> bootstrap -> services.exe (Task Scheduler). + services = SimpleNamespace(pid=4, parent=lambda: None, name=lambda: "services.exe") + bootstrap = SimpleNamespace( + pid=bootstrap_pid, parent=lambda: services, name=lambda: "hermes-gateway.exe" + ) + gw = SimpleNamespace( + pid=gateway_pid, parent=lambda: bootstrap, name=lambda: "hermes-gateway.exe" + ) + # Genuine Windows orphan: its parent exited; Windows does NOT reparent, + # so psutil reports parent() is None — the chain never reaches + # services.exe and the orphan is reaped. + orphan = SimpleNamespace(pid=orphan_pid, parent=lambda: None, name=lambda: "hermes-gateway.exe") + by_pid = {gateway_pid: gw, bootstrap_pid: bootstrap, orphan_pid: orphan} + self._install_fake_psutil(monkeypatch, by_pid) + + monkeypatch.setattr( + gateway, + "find_gateway_pids", + lambda exclude_pids=None: [ + p for p in [gateway_pid, bootstrap_pid, orphan_pid] + if p not in (exclude_pids or set()) + ], + ) + + killed_pids = [] + monkeypatch.setattr(gateway.os, "kill", lambda pid, sig: killed_pids.append((pid, sig))) + monkeypatch.setattr("gateway.status._pid_exists", lambda pid: False) + monkeypatch.setattr("gateway.status.write_planned_stop_marker", lambda pid: None) + monkeypatch.setattr("time.sleep", lambda _: None) + monkeypatch.setattr("time.monotonic", lambda: 1.0) + + result = gateway._reap_unsupervised_gateway_orphans() + + assert result is True # the genuine orphan was reaped + killed = [pid for pid, _ in killed_pids] + assert orphan_pid in killed # orphan killed + assert gateway_pid not in killed # supervisor-owned gateway spared (no pidfile!) + assert bootstrap_pid not in killed # its bootstrap spared too + + def test_macos_orphan_reparented_to_launchd_is_still_reaped(self, monkeypatch): + """POSIX inertness guard: a genuine macOS orphan is reparented directly + to launchd (PID 1) — supervisor-name ancestry must NOT spare it, or the + reaper becomes a permanent no-op on macOS/WSL (#51325, #75936).""" + orphan_pid = 99998 + + monkeypatch.setattr(gateway, "is_macos", lambda: True) + monkeypatch.setattr(gateway, "is_windows", lambda: False) + monkeypatch.setattr(gateway, "supports_systemd_services", lambda: False) + monkeypatch.setattr("gateway.status.get_running_pid", lambda: None) + monkeypatch.setattr(gateway, "_get_service_pids", lambda: set()) + + # Realistic macOS topology: the orphan's parent IS launchd (PID 1). + launchd = SimpleNamespace(pid=1, parent=lambda: None, name=lambda: "launchd") + orphan = SimpleNamespace(pid=orphan_pid, parent=lambda: launchd, name=lambda: "Python") + self._install_fake_psutil(monkeypatch, {orphan_pid: orphan, 1: launchd}) + + monkeypatch.setattr( + gateway, + "find_gateway_pids", + lambda exclude_pids=None: [ + p for p in [orphan_pid] if p not in (exclude_pids or set()) + ], + ) + + killed_pids = [] + monkeypatch.setattr(gateway.os, "kill", lambda pid, sig: killed_pids.append((pid, sig))) + monkeypatch.setattr("gateway.status._pid_exists", lambda pid: False) + monkeypatch.setattr("gateway.status.write_planned_stop_marker", lambda pid: None) + monkeypatch.setattr("time.sleep", lambda _: None) + monkeypatch.setattr("time.monotonic", lambda: 1.0) + + result = gateway._reap_unsupervised_gateway_orphans() + + assert result is True + assert orphan_pid in [pid for pid, _ in killed_pids] + + def test_backstop_is_inert_on_posix(self, monkeypatch): + """Direct unit guard: on non-Windows the backstop returns False without + touching psutil, even for a launchd/init-ancestored process.""" + monkeypatch.setattr(gateway, "is_windows", lambda: False) + + def _boom(_pid): + raise AssertionError("psutil must not be consulted on POSIX") + + monkeypatch.setitem(sys.modules, "psutil", SimpleNamespace(Process=_boom)) + assert gateway._reaper_candidate_is_supervisor_owned(12345) is False + + def test_windows_backstop_fails_open_when_bootstrap_exited(self, monkeypatch): + """Documented limitation: if the Task bootstrap already exited, the + chain breaks before services.exe (Windows does not reparent) and the + candidate is treated as a reapable orphan.""" + monkeypatch.setattr(gateway, "is_windows", lambda: True) + stranded = SimpleNamespace(pid=4242, parent=lambda: None, name=lambda: "hermes-gateway.exe") + self._install_fake_psutil(monkeypatch, {4242: stranded}) + assert gateway._reaper_candidate_is_supervisor_owned(4242) is False + + def test_module_has_logger(): """Verify module has a logger instance (regression guard for #27154).""" assert hasattr(gateway, "logger") From 5ef52273cda8f34fbea589a8dfa61345fd5c6bd1 Mon Sep 17 00:00:00 2001 From: kshitij <82637225+kshitijk4poor@users.noreply.github.com> Date: Sat, 15 Aug 2026 12:39:21 +0530 Subject: [PATCH 236/376] fix(bedrock): let aux calls omit the Converse maxTokens cap The Bedrock Converse shim hardcoded 'else 4096' when the caller passed no max_tokens, so auxiliary vision descriptions stayed capped at 4096 tokens on the Bedrock wire even after #75253 removed the vision call sites' own caps (#10809 was only partially fixed there). Converse's inferenceConfig.maxTokens is optional; when omitted, Bedrock defaults to the model's maximum allowed output. Thread an explicit max_tokens=None through build_converse_kwargs/call_converse to omit the field, and drop an all-empty inferenceConfig from the wire request entirely. The 4096 default is unchanged for every existing caller (main transport passes params.get('max_tokens', 4096) explicitly), so only no-cap aux calls opt in. Surfaced during review of #75253. --- agent/auxiliary_client.py | 7 ++++- agent/bedrock_adapter.py | 24 ++++++++++++----- tests/agent/test_bedrock_adapter.py | 34 +++++++++++++++++++++++ tests/agent/test_bedrock_integration.py | 36 +++++++++++++++++++++++++ 4 files changed, 94 insertions(+), 7 deletions(-) diff --git a/agent/auxiliary_client.py b/agent/auxiliary_client.py index 7e409eb8e3bde..0f163f38efc25 100644 --- a/agent/auxiliary_client.py +++ b/agent/auxiliary_client.py @@ -2213,7 +2213,12 @@ def create(self, **kwargs) -> Any: model=model, messages=messages, tools=kwargs.get("tools"), - max_tokens=int(max_tokens) if max_tokens else 4096, + # Omitted/None caller cap → None: build_converse_kwargs then omits + # inferenceConfig.maxTokens so Bedrock uses the model's maximum + # allowed output, matching the no-cap-by-default policy every + # other aux wire already follows (#10809: vision descriptions + # stayed capped at the shim's old hardcoded 4096 on Bedrock). + max_tokens=int(max_tokens) if max_tokens else None, temperature=kwargs.get("temperature"), top_p=kwargs.get("top_p"), stop_sequences=stop, diff --git a/agent/bedrock_adapter.py b/agent/bedrock_adapter.py index c399081619ffa..8d63323fd299c 100644 --- a/agent/bedrock_adapter.py +++ b/agent/bedrock_adapter.py @@ -1019,7 +1019,7 @@ def build_converse_kwargs( model: str, messages: List[Dict], tools: Optional[List[Dict]] = None, - max_tokens: int = 4096, + max_tokens: Optional[int] = 4096, temperature: Optional[float] = None, top_p: Optional[float] = None, stop_sequences: Optional[List[str]] = None, @@ -1028,16 +1028,24 @@ def build_converse_kwargs( """Build kwargs for ``bedrock-runtime.converse()`` or ``converse_stream()``. Converts OpenAI-format inputs to Converse API parameters. + + ``max_tokens=None`` omits ``inferenceConfig.maxTokens`` entirely, in which + case Bedrock defaults to the model's maximum allowed output — the Converse + field is optional per the AWS API reference. The default stays 4096 so + existing callers are unaffected; callers that want the model's full output + budget (e.g. uncapped auxiliary vision calls) pass ``None`` explicitly. """ system_prompt, converse_messages = convert_messages_to_converse(messages) cache_enabled = _model_supports_prompt_cache(model) + inference_config: Dict[str, Any] = {} + if max_tokens is not None: + inference_config["maxTokens"] = max_tokens + kwargs: Dict[str, Any] = { "modelId": model, "messages": converse_messages, - "inferenceConfig": { - "maxTokens": max_tokens, - }, + "inferenceConfig": inference_config, } if system_prompt: @@ -1086,6 +1094,10 @@ def build_converse_kwargs( if guardrail_config: kwargs["guardrailConfig"] = guardrail_config + if not kwargs["inferenceConfig"]: + # inferenceConfig is optional on the wire; don't send an empty object. + del kwargs["inferenceConfig"] + return kwargs @@ -1094,7 +1106,7 @@ def call_converse( model: str, messages: List[Dict], tools: Optional[List[Dict]] = None, - max_tokens: int = 4096, + max_tokens: Optional[int] = 4096, temperature: Optional[float] = None, top_p: Optional[float] = None, stop_sequences: Optional[List[str]] = None, @@ -1135,7 +1147,7 @@ def call_converse_stream( model: str, messages: List[Dict], tools: Optional[List[Dict]] = None, - max_tokens: int = 4096, + max_tokens: Optional[int] = 4096, temperature: Optional[float] = None, top_p: Optional[float] = None, stop_sequences: Optional[List[str]] = None, diff --git a/tests/agent/test_bedrock_adapter.py b/tests/agent/test_bedrock_adapter.py index 8994688e0f211..8d6acf2ede424 100644 --- a/tests/agent/test_bedrock_adapter.py +++ b/tests/agent/test_bedrock_adapter.py @@ -389,6 +389,40 @@ def test_includes_tools(self): assert "toolConfig" in kwargs assert len(kwargs["toolConfig"]["tools"]) == 1 + def test_default_max_tokens_stays_4096(self): + """Callers that don't pass max_tokens keep the historical 4096 cap — + the None-omission behavior is strictly opt-in.""" + from agent.bedrock_adapter import build_converse_kwargs + kwargs = build_converse_kwargs( + model="test-model", messages=[{"role": "user", "content": "Hi"}], + ) + assert kwargs["inferenceConfig"]["maxTokens"] == 4096 + + def test_max_tokens_none_omits_cap(self): + """max_tokens=None omits inferenceConfig.maxTokens so Bedrock uses the + model's maximum allowed output (the Converse field is optional).""" + from agent.bedrock_adapter import build_converse_kwargs + kwargs = build_converse_kwargs( + model="test-model", + messages=[{"role": "user", "content": "Hi"}], + max_tokens=None, + temperature=0.1, + ) + assert "maxTokens" not in kwargs["inferenceConfig"] + # Other inference params still flow through. + assert kwargs["inferenceConfig"]["temperature"] == 0.1 + + def test_max_tokens_none_and_no_sampling_drops_empty_inference_config(self): + """When every inference param is absent, don't send an empty + inferenceConfig object on the wire.""" + from agent.bedrock_adapter import build_converse_kwargs + kwargs = build_converse_kwargs( + model="test-model", + messages=[{"role": "user", "content": "Hi"}], + max_tokens=None, + ) + assert "inferenceConfig" not in kwargs + diff --git a/tests/agent/test_bedrock_integration.py b/tests/agent/test_bedrock_integration.py index 6f6fcbd8e08a0..32cd9056ee7f1 100644 --- a/tests/agent/test_bedrock_integration.py +++ b/tests/agent/test_bedrock_integration.py @@ -437,3 +437,39 @@ def test_bedrock_converse_shim_stream_returns_complete_response(self, monkeypatc # got-final-object downgrade path handles the rest. assert resp is sentinel assert mock_converse.call_count == 1 + + def test_bedrock_shim_uncapped_when_caller_omits_max_tokens(self, monkeypatch): + """No caller max_tokens → the shim passes None through and the wire + request carries no inferenceConfig.maxTokens, so Bedrock uses the + model's maximum allowed output (#10809 on the Bedrock wire). + + Guards against the shim's old hardcoded ``else 4096`` fallback, which + kept aux vision descriptions capped after the vision call sites + dropped their own caps.""" + monkeypatch.setenv("AWS_ACCESS_KEY_ID", "AKIAIO...MPLE") + monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY") + + from agent.auxiliary_client import BedrockAuxiliaryClient + + client = BedrockAuxiliaryClient("us-east-1", "openai.gpt-oss-20b-1:0") + boto3_client = MagicMock() + with patch("agent.bedrock_adapter._get_bedrock_runtime_client", + return_value=boto3_client), \ + patch("agent.bedrock_adapter.normalize_converse_response"): + # Aux vision-style call: no max_tokens key at all. + client.chat.completions.create( + model="openai.gpt-oss-20b-1:0", + messages=[{"role": "user", "content": "describe"}], + temperature=0.1, + ) + wire_kwargs = boto3_client.converse.call_args.kwargs + assert "maxTokens" not in wire_kwargs.get("inferenceConfig", {}) + + # An explicit caller cap still lands on the wire unchanged. + client.chat.completions.create( + model="openai.gpt-oss-20b-1:0", + messages=[{"role": "user", "content": "describe"}], + max_tokens=1234, + ) + wire_kwargs = boto3_client.converse.call_args.kwargs + assert wire_kwargs["inferenceConfig"]["maxTokens"] == 1234 From 8b58f9f68f01a96f101366b6b9a98dbd341db301 Mon Sep 17 00:00:00 2001 From: kshitij <82637225+kshitijk4poor@users.noreply.github.com> Date: Sat, 15 Aug 2026 12:43:49 +0530 Subject: [PATCH 237/376] test(bedrock): pin stream-path cap omission; document truthiness edge Self-review follow-up: cover call_converse_stream's max_tokens=None path (same builder, previously unpinned) and document why the shim reads the caller cap with truthiness rather than 'is None' (parity with the Anthropic shim's reading). --- agent/auxiliary_client.py | 3 +++ tests/agent/test_bedrock_adapter.py | 21 +++++++++++++++++++++ 2 files changed, 24 insertions(+) diff --git a/agent/auxiliary_client.py b/agent/auxiliary_client.py index 0f163f38efc25..d49a797ed8a60 100644 --- a/agent/auxiliary_client.py +++ b/agent/auxiliary_client.py @@ -2218,6 +2218,9 @@ def create(self, **kwargs) -> Any: # allowed output, matching the no-cap-by-default policy every # other aux wire already follows (#10809: vision descriptions # stayed capped at the shim's old hardcoded 4096 on Bedrock). + # Truthiness (not `is None`) is deliberate — it matches the + # sibling Anthropic shim's reading of max_tokens above, so a + # nonsense explicit 0 is treated as "no cap" on both wires. max_tokens=int(max_tokens) if max_tokens else None, temperature=kwargs.get("temperature"), top_p=kwargs.get("top_p"), diff --git a/tests/agent/test_bedrock_adapter.py b/tests/agent/test_bedrock_adapter.py index 8d6acf2ede424..e6c2c3c3c488a 100644 --- a/tests/agent/test_bedrock_adapter.py +++ b/tests/agent/test_bedrock_adapter.py @@ -423,6 +423,27 @@ def test_max_tokens_none_and_no_sampling_drops_empty_inference_config(self): ) assert "inferenceConfig" not in kwargs + def test_call_converse_stream_omits_cap_for_none(self): + """The streaming entry point funnels through the same builder — pin + that max_tokens=None omits the cap there too.""" + from unittest.mock import MagicMock, patch as mock_patch + from agent.bedrock_adapter import call_converse_stream + boto3_client = MagicMock() + boto3_client.converse_stream.return_value = {"stream": []} + with mock_patch( + "agent.bedrock_adapter._get_bedrock_runtime_client", + return_value=boto3_client, + ): + call_converse_stream( + region="us-east-1", + model="test-model", + messages=[{"role": "user", "content": "Hi"}], + max_tokens=None, + temperature=0.2, + ) + wire_kwargs = boto3_client.converse_stream.call_args.kwargs + assert "maxTokens" not in wire_kwargs.get("inferenceConfig", {}) + From ce996d40577c242dc04cc6d66e827dcdf8daa569 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 15 Aug 2026 00:20:32 -0700 Subject: [PATCH 238/376] feat(delegation): raise max_concurrent_children default 3 -> 10 (+migration) (#86745) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit delegation.max_concurrent_children caps how many delegated children run in parallel per batch (and concurrent background delegation units). The old default of 3 needlessly serialized independent fan-outs (e.g. reviewing/​investigating N PRs or issues at once), so large batches ran in slow chunks of 3. Raise the shipped default to 10, which sits at/below the existing high-cost advisory threshold (>10), so the default never trips the warning. Each child still consumes API tokens independently, so this is a throughput/latency win the user pays for in parallel token spend — the floor stays 1 and there is no ceiling, so anyone can tune it down or up. - config_defaults.py: default 3 -> 10; _config_version 36 -> 37. - delegate_tool.py: _DEFAULT_MAX_CONCURRENT_CHILDREN 3 -> 10 (+ docstring). - config_migrations.py: _migrate_to_37 lifts configs pinned at exactly the old default 3 to 10 (deliberate non-3 overrides preserved; unset inherits 10). - cli-config.yaml.example: documented default updated. Verified: default/fallback read 10, version 37, and the migration lifts 3->10, preserves an explicit 5, and leaves unset untouched. Co-authored-by: Teknium --- cli-config.yaml.example | 2 +- hermes_cli/config_defaults.py | 4 ++-- hermes_cli/config_migrations.py | 32 ++++++++++++++++++++++++++++++++ tools/delegate_tool.py | 4 ++-- 4 files changed, 37 insertions(+), 5 deletions(-) diff --git a/cli-config.yaml.example b/cli-config.yaml.example index 92a194e1bcbb5..4dee7a61f3bc8 100644 --- a/cli-config.yaml.example +++ b/cli-config.yaml.example @@ -1396,7 +1396,7 @@ code_execution: # Supports single tasks and batch mode (default 3 parallel, configurable). delegation: max_iterations: 250 # Max tool-calling turns per child (default: 250) - # max_concurrent_children: 3 # Max parallel child agents per batch (default: 3, floor: 1, no ceiling). + # max_concurrent_children: 10 # Max parallel child agents per batch (default: 10, floor: 1, no ceiling). # WARNING: values above 10 multiply API cost linearly. # max_spawn_depth: 1 # Delegation tree depth cap (range: 1-3, default: 1 = flat). # Raise to 2 to allow workers to spawn their own subagents. diff --git a/hermes_cli/config_defaults.py b/hermes_cli/config_defaults.py index 5ffd661354c21..0f29f5059efb4 100644 --- a/hermes_cli/config_defaults.py +++ b/hermes_cli/config_defaults.py @@ -1813,7 +1813,7 @@ # (floor 30s) to enforce a hard cap. "reasoning_effort": "", # subagent effort: "ultra", "max", "xhigh", "high", # "medium", "low", "minimal", "none" (empty = inherit) - "max_concurrent_children": 3, # unified concurrency cap: max parallel children per batch + "max_concurrent_children": 10, # unified concurrency cap: max parallel children per batch # AND max concurrent background (background=true) # delegation units. New async dispatches beyond the cap # fall back to synchronous execution. Floor of 1, no ceiling. @@ -3402,7 +3402,7 @@ }, # Config schema version - bump this when adding new required fields - "_config_version": 36, + "_config_version": 37, } # Optional environment variables that enhance functionality diff --git a/hermes_cli/config_migrations.py b/hermes_cli/config_migrations.py index 54f360c319fd6..b1464604fa436 100644 --- a/hermes_cli/config_migrations.py +++ b/hermes_cli/config_migrations.py @@ -784,6 +784,37 @@ def _migrate_to_36(results: Dict[str, Any], quiet: bool) -> None: ) +def _migrate_to_37(results: Dict[str, Any], quiet: bool) -> None: + # ── Version 36 → 37: raise the delegation concurrency default 3 → 10 ── + # delegation.max_concurrent_children caps how many children run in parallel + # per batch (and concurrent background delegation units). The old default of + # 3 needlessly serialized independent fan-outs (e.g. reviewing N PRs at + # once). The shipped default is now 10, which stays at/below the high-cost + # warning threshold. Configs still pinned at exactly the old default 3 — + # almost always the inherited default rather than a deliberate choice — are + # lifted to 10 so existing installs get the wider fan-out on update. Any + # OTHER explicit value (a deliberate override) is preserved; unset inherits + # 10 at read time. + _c = _cfg() + read_raw_config = _c.read_raw_config + _persist_migration = _c._persist_migration + + config = read_raw_config() + raw_deleg = config.get("delegation") + if isinstance(raw_deleg, dict) and raw_deleg.get("max_concurrent_children") == 3: + raw_deleg["max_concurrent_children"] = 10 + config["delegation"] = raw_deleg + _persist_migration(config) + results["config_added"].append("delegation.max_concurrent_children=10 (was: 3)") + if not quiet: + print( + " ✓ Raised delegation.max_concurrent_children from 3 to 10 — " + "independent delegated children now fan out wider in parallel. " + "Each child consumes API tokens independently; set " + "delegation.max_concurrent_children back to 3 to restore the old cap." + ) + + #: Registry of (target_version, migration_fn), strictly ascending. The driver #: applies every entry whose target version is greater than the on-disk #: observe earlier steps' writes via read_raw_config() (filesystem state). @@ -807,6 +838,7 @@ def _migrate_to_36(results: Dict[str, Any], quiet: bool) -> None: (34, _migrate_to_34), (35, _migrate_to_35), (36, _migrate_to_36), + (37, _migrate_to_37), ) diff --git a/tools/delegate_tool.py b/tools/delegate_tool.py index fd47d15133eb5..a8da6988a5e8e 100644 --- a/tools/delegate_tool.py +++ b/tools/delegate_tool.py @@ -119,7 +119,7 @@ def _get_subagent_approval_callback(): # "delegation" toolset in _build_child_agent), NOT by the model naming toolsets # — the model has no toolsets argument. Subagents inherit the parent's toolsets. -_DEFAULT_MAX_CONCURRENT_CHILDREN = 3 +_DEFAULT_MAX_CONCURRENT_CHILDREN = 10 # One-shot guard: the high-concurrency cost advisory is emitted at most once # per process. _get_max_concurrent_children() runs on every get_definitions() # schema rebuild (via _build_top_level_description / _build_tasks_param_description), @@ -733,7 +733,7 @@ def _normalize_role(r: Optional[str]) -> str: def _get_max_concurrent_children() -> int: """Read delegation.max_concurrent_children from config, falling back to - DELEGATION_MAX_CONCURRENT_CHILDREN env var, then the default (3). + DELEGATION_MAX_CONCURRENT_CHILDREN env var, then the default (10). Users can raise this as high as they want; only the floor (1) is enforced. From dcc2f3de1d3a11b21289d50343bc404cddc55635 Mon Sep 17 00:00:00 2001 From: adikpb <67222969+adikpb@users.noreply.github.com> Date: Fri, 31 Jul 2026 09:50:47 +0400 Subject: [PATCH 239/376] fix(vision): stop capping aux vision output with hardcoded max_tokens MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The vision tools' call_kwargs hardcode max_tokens caps (2000 for vision_analyze/browser_vision, 4000 for video analysis), truncating descriptions of complex images at the cap. The centralized aux client already omits max_tokens by default (#34845) so providers use their model max output; these three call sites were the leftovers that bypassed that policy. Remove the hardcoded caps entirely — the aux client handles the mandatory-max_tokens Anthropic wire via _resolve_anthropic_messages_max_tokens (model output ceiling) and Gemini native omits maxOutputTokens (65K ceiling), so no wire needs an explicit cap. --- tools/browser_tool.py | 1 - tools/vision_tools.py | 2 -- 2 files changed, 3 deletions(-) diff --git a/tools/browser_tool.py b/tools/browser_tool.py index b831efbec586d..eaaff9d969ed2 100644 --- a/tools/browser_tool.py +++ b/tools/browser_tool.py @@ -4692,7 +4692,6 @@ def browser_vision(question: str, annotate: bool = False, task_id: Optional[str] ], } ], - "max_tokens": 2000, "temperature": vision_temperature, "timeout": vision_timeout, } diff --git a/tools/vision_tools.py b/tools/vision_tools.py index e84daaff7ffe1..e008983f7f20a 100644 --- a/tools/vision_tools.py +++ b/tools/vision_tools.py @@ -1501,7 +1501,6 @@ async def vision_analyze_tool( "task": "vision", "messages": messages, "temperature": vision_temperature, - "max_tokens": 2000, "timeout": vision_timeout, } if model: @@ -2073,7 +2072,6 @@ async def video_analyze_tool( "task": "vision", "messages": messages, "temperature": vision_temperature, - "max_tokens": 4000, "timeout": vision_timeout, } if model: From ec470d9db212c9b19cdf2b31f59a1adc2fbbee03 Mon Sep 17 00:00:00 2001 From: adikpb <67222969+adikpb@users.noreply.github.com> Date: Fri, 31 Jul 2026 10:15:57 +0400 Subject: [PATCH 240/376] test(vision): assert vision aux calls carry no max_tokens cap Covers the max-tokens-knob contract: vision call_kwargs omit max_tokens entirely (configured values, defaults, and even an explicit auxiliary.vision.max_tokens config entry must never be forwarded), so providers use their full output budget. --- tests/tools/test_vision_tools.py | 11 +++++++++++ 1 file changed, 11 insertions(+) diff --git a/tests/tools/test_vision_tools.py b/tests/tools/test_vision_tools.py index 55b653d6fac98..806382f649d2c 100644 --- a/tests/tools/test_vision_tools.py +++ b/tests/tools/test_vision_tools.py @@ -276,11 +276,22 @@ async def call_with(config): ) assert kwargs["temperature"] == 1.0 assert kwargs["timeout"] == 77.0 + # No hardcoded output cap — the aux client omits max_tokens so the + # provider uses its full output budget (max-tokens-knob policy). + assert "max_tokens" not in kwargs # Omitted values fall back to the built-in defaults. kwargs = await call_with({"auxiliary": {"vision": {}}}) assert kwargs["temperature"] == 0.1 assert kwargs["timeout"] == 120.0 + assert "max_tokens" not in kwargs + + # Even an explicit auxiliary.vision.max_tokens config entry must NOT + # be forwarded: user-facing max_tokens knobs are policy-prohibited. + kwargs = await call_with({"auxiliary": {"vision": {"max_tokens": 8000}}}) + assert "max_tokens" not in kwargs + assert kwargs["temperature"] == 0.1 + assert kwargs["timeout"] == 120.0 class TestVisionSafetyGuards: From 688abc585f4ed8ba7b19f6fe0629821857fed7ce Mon Sep 17 00:00:00 2001 From: adikpb <67222969+adikpb@users.noreply.github.com> Date: Fri, 31 Jul 2026 10:23:16 +0400 Subject: [PATCH 241/376] test(vision): assert no max_tokens cap in browser and video aux kwargs Sweeper follow-up: the browser-screenshot and video kwargs captures now also assert max_tokens is absent, protecting the central auxiliary no-cap policy against refactors that would restore the hardcoded caps. --- tests/tools/test_browser_console.py | 3 +++ tests/tools/test_video_analyze.py | 3 +++ 2 files changed, 6 insertions(+) diff --git a/tests/tools/test_browser_console.py b/tests/tools/test_browser_console.py index 24eca861ed4dd..abed1bb380418 100644 --- a/tests/tools/test_browser_console.py +++ b/tests/tools/test_browser_console.py @@ -274,6 +274,9 @@ def test_browser_vision_uses_configured_temperature_and_timeout(self, tmp_path): assert result["analysis"] == "Annotated screenshot analysis" assert mock_llm.call_args.kwargs["temperature"] == 1.0 assert mock_llm.call_args.kwargs["timeout"] == 45.0 + # No hardcoded output cap — the aux client omits max_tokens so the + # provider uses its full output budget (max-tokens-knob policy). + assert "max_tokens" not in mock_llm.call_args.kwargs def test_browser_vision_native_fast_path_returns_multimodal(self, tmp_path): diff --git a/tests/tools/test_video_analyze.py b/tests/tools/test_video_analyze.py index 020cfc358c5d8..0035111630229 100644 --- a/tests/tools/test_video_analyze.py +++ b/tests/tools/test_video_analyze.py @@ -206,6 +206,9 @@ async def capture_llm(**kwargs): assert content[1]["type"] == "video_url" assert "video_url" in content[1] assert content[1]["video_url"]["url"].startswith("data:video/mp4;base64,") + # No hardcoded output cap — the aux client omits max_tokens so the + # provider uses its full output budget (max-tokens-knob policy). + assert "max_tokens" not in captured_kwargs def test_non_local_backend_reads_video_from_terminal_backend(self, tmp_path, monkeypatch): """Non-local terminal backends must not read local host video paths. From d2672a349b6e783868e681735b45cad181cb05a8 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 15 Aug 2026 00:25:00 -0700 Subject: [PATCH 242/376] feat(gateway): optional profile param on cron.manage RPC (#86796) cron.manage resolved its jobs store from the process HERMES_HOME, so a profile whose cron lives in ~/.hermes/profiles//cron/ was invisible to the default gateway (and any bot/plugin querying per-profile routines saw 'no cron jobs'). Add an optional 'profile' param that scopes the whole action via set_hermes_home_override, exactly mirroring the adjacent skills.manage handler: resolve get_profile_dir(profile), 404 (err 4064) if missing, override in a try/finally that always reset_hermes_home_override. Omitted/None keeps the launch-profile behavior, so existing callers are unaffected. cronjob() itself is unchanged (it already keys off HERMES_HOME). Enables the Hermes-Bot-Mode plugin to show a bot's real routines (NousResearch/Hermes-Bot-Mode#37). Needs a SERVE-backend gateway restart to take effect live. 2/2 in the new focused test. Co-authored-by: Teknium --- tests/test_cron_manage_profile_scope.py | 75 +++++++++++++++++++++++++ tui_gateway/methods_tools.py | 25 +++++++++ 2 files changed, 100 insertions(+) create mode 100644 tests/test_cron_manage_profile_scope.py diff --git a/tests/test_cron_manage_profile_scope.py b/tests/test_cron_manage_profile_scope.py new file mode 100644 index 0000000000000..cdad2e8c7b81b --- /dev/null +++ b/tests/test_cron_manage_profile_scope.py @@ -0,0 +1,75 @@ +"""cron.manage optional ``profile`` param — per-profile store scoping. + +Mirrors ``skills.manage`` / ``mcp.catalog``: when a ``profile`` is passed the +handler resolves ``get_profile_dir(profile)`` and wraps the action dispatch in +``set_hermes_home_override`` / ``reset_hermes_home_override``. Because +``cronjob()`` -> ``list_jobs()`` keys off ``get_hermes_home()``, the list action +must then read THAT profile's ``cron/jobs.json``, not the launch profile's. +""" + +import json + +from tui_gateway import server + + +def test_cron_manage_profile_reads_that_profiles_store(tmp_path, monkeypatch): + # A temp profile home with one job in its cron store. + profile_home = tmp_path / "profiles" / "botA" + cron_dir = profile_home / "cron" + cron_dir.mkdir(parents=True) + (cron_dir / "jobs.json").write_text( + json.dumps( + { + "jobs": [ + { + "id": "job-botA", + "name": "botA-only-job", + "prompt": "scoped hello", + "enabled": True, + } + ] + } + ), + encoding="utf-8", + ) + + # Route the profile name the handler resolves to our temp home. + import hermes_cli.profiles as profiles + + monkeypatch.setattr(profiles, "get_profile_dir", lambda name: profile_home) + + resp = server.handle_request( + { + "id": "1", + "method": "cron.manage", + "params": {"action": "list", "profile": "botA"}, + } + ) + + assert "result" in resp, resp + names = [j.get("name") for j in resp["result"]["jobs"]] + assert "botA-only-job" in names + + # The override must not leak: an unscoped call after this one resolves the + # launch profile again, which does not contain botA's job. + from hermes_constants import get_hermes_home_override + + assert get_hermes_home_override() is None + + +def test_cron_manage_unknown_profile_errors(tmp_path, monkeypatch): + import hermes_cli.profiles as profiles + + missing = tmp_path / "profiles" / "ghost" + monkeypatch.setattr(profiles, "get_profile_dir", lambda name: missing) + + resp = server.handle_request( + { + "id": "2", + "method": "cron.manage", + "params": {"action": "list", "profile": "ghost"}, + } + ) + + assert "error" in resp, resp + assert resp["error"]["code"] == 4064 diff --git a/tui_gateway/methods_tools.py b/tui_gateway/methods_tools.py index dd065c520b9dd..2e1d0d95b98ee 100644 --- a/tui_gateway/methods_tools.py +++ b/tui_gateway/methods_tools.py @@ -1660,6 +1660,23 @@ def _(rid, params: dict) -> dict: @method("cron.manage") def _(rid, params: dict) -> dict: action, jid = params.get("action", "list"), params.get("name", "") + # Optional profile scoping: cronjob() keys off HERMES_HOME, so scoping the + # env override lets a per-profile cron store be listed/mutated even when + # that profile runs a separate gateway. Omitted/None = the launch profile. + # Mirrors ``skills.manage`` / ``mcp.catalog``. + profile = str(params.get("profile") or "").strip() + token = None + if profile: + try: + from hermes_cli.profiles import get_profile_dir + from hermes_constants import set_hermes_home_override + + profile_dir = get_profile_dir(profile) + if not profile_dir or not profile_dir.is_dir(): + return _err(rid, 4064, f"profile '{profile}' not found") + token = set_hermes_home_override(str(profile_dir)) + except Exception as e: + return _err(rid, 5023, str(e)) try: from tools.cronjob_tools import cronjob @@ -1700,6 +1717,14 @@ def _(rid, params: dict) -> dict: return _err(rid, 4016, f"unknown cron action: {action}") except Exception as e: return _err(rid, 5023, str(e)) + finally: + if token is not None: + try: + from hermes_constants import reset_hermes_home_override + + reset_hermes_home_override(token) + except Exception: + pass @method("learning.frames") From fbaea9bddc72c705527f3532fedda88d8e3b52a2 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 15 Aug 2026 00:31:37 -0700 Subject: [PATCH 243/376] feat(sessions): generic 'hidden' session flag (sidebar-hide, still resumable) (#86797) * feat(sessions): generic 'hidden' session flag (sidebar-hide, still resumable) Adds a source-orthogonal, archive-orthogonal 'hidden' session flag meaning 'don't show in the global Sessions sidebar, but stay fully resumable by the surface that owns it'. Mirrors the existing archived/pinned capability end to end, so it's a generic widening (any plugin that owns its own session lifecycle - kanban, Bot Mode, future plugins - can keep its sessions out of the shared recents list) rather than a per-plugin special-case. - Schema: hidden INTEGER NOT NULL DEFAULT 0 on sessions (additive; lands on existing DBs via the declarative _reconcile_columns ADD COLUMN path, same as archived/pinned - no version-gated migration). - DB: SessionDB.set_session_hidden(session_id, hidden) (clones set_session_pinned incl. the compression-lineage recursive CTE); list_sessions_rich gains include_hidden=False, appending 's.hidden = 0' by default so hidden rows drop from every listing path (and the REST sidebar endpoints inherit it with no change). - Gateway: session.set_hidden RPC (mirrors session.title); session.create accepts hidden=true, deferred via pending_hidden and applied in _ensure_session_db_row when the row is lazily created (mirrors pending_title). - REST parity: PATCH /api/sessions/{id} accepts+bool-validates 'hidden' -> set_session_hidden; _session_response exposes it. Enables Hermes-Bot-Mode to hide canonical 'Bot Chat' sessions from the sidebar (NousResearch/Hermes-Bot-Mode#46) WITHOUT retagging source (which would mis-set the agent platform). Bot Chats keep source=desktop. Gateway RPC needs a SERVE-backend restart to take effect live. 1 focused test (default-exclude / include_hidden / unhide round-trip). * fix: teach lost-and-found recovery about the 55-column sessions layout Adding the 'hidden' column makes the current sessions table 55 columns. The SQLite lost-and-found recovery classifier keys off the physical field count (SESSIONS_LAYOUT_NFIELDS) to identify a salvaged sessions row, so a recovered current-layout row (nfield=55) would otherwise be unrecognized and dropped. Add 55 to the frozenset (54/52 stay as historical prefixes) and update the column-count assertions + synthetic current-layout insert in the recovery test. --------- Co-authored-by: Teknium --- gateway/platforms/api_server.py | 10 ++-- hermes_cli/session_lost_and_found.py | 2 +- hermes_state.py | 57 +++++++++++++++++++ hermes_state_common.py | 1 + .../test_session_recovery_lost_and_found.py | 8 +-- tests/hermes_state/test_session_hidden.py | 44 ++++++++++++++ tui_gateway/methods_session.py | 32 +++++++++++ tui_gateway/server.py | 8 +++ 8 files changed, 153 insertions(+), 9 deletions(-) create mode 100644 tests/hermes_state/test_session_hidden.py diff --git a/gateway/platforms/api_server.py b/gateway/platforms/api_server.py index 841d7609feb8d..4fc3b1a91bd6b 100644 --- a/gateway/platforms/api_server.py +++ b/gateway/platforms/api_server.py @@ -3308,11 +3308,11 @@ def _session_response(session: Dict[str, Any]) -> Dict[str, Any]: "output_tokens", "cache_read_tokens", "cache_write_tokens", "reasoning_tokens", "estimated_cost_usd", "actual_cost_usd", "api_call_count", "parent_session_id", "last_active", "preview", - "_lineage_root_id", "pinned", "archived", + "_lineage_root_id", "pinned", "archived", "hidden", ) payload = {key: session.get(key) for key in safe_keys if key in session} # SQLite stores these as 0/1; clients reconcile against a real boolean. - for flag in ("pinned", "archived"): + for flag in ("pinned", "archived", "hidden"): if flag in payload: payload[flag] = bool(payload[flag]) # Avoid exposing full system prompts/model_config through the client API; @@ -3534,12 +3534,12 @@ async def _handle_patch_session(self, request: "web.Request") -> "web.Response": # sidebar owns (the "keep" flag exempts a chat from the auto-archive # sweep). Rejecting them here was silently 400ing every pin the desktop # made, so pins only ever lived in that one app's localStorage. - allowed = {"title", "end_reason", "pinned", "archived"} + allowed = {"title", "end_reason", "pinned", "archived", "hidden"} unknown = sorted(set(body) - allowed) if unknown: return web.json_response(_openai_error(f"Unsupported session fields: {', '.join(unknown)}", code="unsupported_session_field"), status=400) - for flag in ("pinned", "archived"): + for flag in ("pinned", "archived", "hidden"): if flag in body and not isinstance(body[flag], bool): return web.json_response(_openai_error(f"'{flag}' must be a boolean", code="invalid_session_field"), status=400) @@ -3553,6 +3553,8 @@ async def _handle_patch_session(self, request: "web.Request") -> "web.Response": await asyncio.to_thread(db.set_session_pinned, session_id, body["pinned"]) if "archived" in body: await asyncio.to_thread(db.set_session_archived, session_id, body["archived"]) + if "hidden" in body: + await asyncio.to_thread(db.set_session_hidden, session_id, body["hidden"]) if body.get("end_reason"): await asyncio.to_thread(db.end_session, session_id, str(body["end_reason"])) session = await asyncio.to_thread(db.get_session, session_id) or session diff --git a/hermes_cli/session_lost_and_found.py b/hermes_cli/session_lost_and_found.py index 9b7fd81bde536..90d8acba9a4df 100644 --- a/hermes_cli/session_lost_and_found.py +++ b/hermes_cli/session_lost_and_found.py @@ -47,7 +47,7 @@ # Historical physical layouts of the sessions table. Columns are only ever # appended (ALTER TABLE ADD COLUMN), so an older record is a strict prefix of # the current column order. -SESSIONS_LAYOUT_NFIELDS = frozenset({54, 52}) +SESSIONS_LAYOUT_NFIELDS = frozenset({55, 54, 52}) SESSIONS_LEGACY_MINIMAL_NFIELD = 14 SESSION_MODEL_USAGE_NFIELD = 18 diff --git a/hermes_state.py b/hermes_state.py index 3711e91f496df..3a4459f275f3e 100644 --- a/hermes_state.py +++ b/hermes_state.py @@ -7599,6 +7599,60 @@ def _do(conn): rowcount = self._execute_write(_do) return rowcount > 0 + def set_session_hidden(self, session_id: str, hidden: bool) -> bool: + """Hide or unhide a session (and its whole compression lineage). + + ``hidden`` is a generic "don't show in the global Sessions sidebar" + flag: a hidden session is dropped from the default + :meth:`list_sessions_rich` listing (which omits ``include_hidden``) but + stays fully resumable by the surface that owns it — useful for plugins + that manage their own sessions (e.g. kanban) and don't want them + cluttering the shared recents list. Like :meth:`set_session_archived` + / :meth:`set_session_pinned` the whole compression chain is flipped as + a unit, so hiding the surfaced tip hides the root (and vice-versa) no + matter which id the caller holds. Returns True when at least one row + changed. + """ + def _do(conn): + cursor = conn.execute( + """ + WITH RECURSIVE + ancestors(id) AS ( + SELECT ? + UNION + SELECT parent.id + FROM ancestors a + JOIN sessions child ON child.id = a.id + JOIN sessions parent ON parent.id = child.parent_session_id + WHERE parent.end_reason = 'compression' + ), + descendants(id) AS ( + SELECT ? + UNION + SELECT child.id + FROM descendants d + JOIN sessions parent ON parent.id = d.id + JOIN sessions child ON child.parent_session_id = parent.id + WHERE parent.end_reason = 'compression' + ), + lineage(id) AS ( + SELECT id FROM ancestors + UNION + SELECT id FROM descendants + ) + UPDATE sessions + SET hidden = ? + WHERE id IN (SELECT id FROM lineage) + """, + (session_id, session_id, 1 if hidden else 0), + ) + rowcount = cursor.rowcount + if rowcount is None or rowcount < 0: + rowcount = conn.execute("SELECT changes()").fetchone()[0] + return rowcount + rowcount = self._execute_write(_do) + return rowcount > 0 + def set_session_read(self, session_id: str, read: bool = True) -> bool: """Mark a session read or unread (and its whole compression lineage). @@ -7870,6 +7924,7 @@ def list_sessions_rich( compact_rows: bool = False, include_pinned: bool = False, session_key: str = None, + include_hidden: bool = False, ) -> List[Dict[str, Any]]: """List sessions with preview (first user message) and last active timestamp. @@ -7972,6 +8027,8 @@ def list_sessions_rich( where_clauses.append("s.archived = 1") elif not include_archived: where_clauses.append("s.archived = 0") + if not include_hidden: + where_clauses.append("s.hidden = 0") where_sql = f"WHERE {' AND '.join(where_clauses)}" if where_clauses else "" # Snapshot the filter params before the query builders below extend diff --git a/hermes_state_common.py b/hermes_state_common.py index 28f3a63cdbe02..5a9e4893e6c75 100644 --- a/hermes_state_common.py +++ b/hermes_state_common.py @@ -310,6 +310,7 @@ def _sql_session_last_active_by_id(session_id_expr: str) -> str: rewind_count INTEGER NOT NULL DEFAULT 0, archived INTEGER NOT NULL DEFAULT 0, pinned INTEGER NOT NULL DEFAULT 0, + hidden INTEGER NOT NULL DEFAULT 0, last_read_at REAL, FOREIGN KEY (parent_session_id) REFERENCES sessions(id), FOREIGN KEY (system_prompt_hash) REFERENCES system_prompts(hash) diff --git a/tests/hermes_cli/test_session_recovery_lost_and_found.py b/tests/hermes_cli/test_session_recovery_lost_and_found.py index 7f273ff70180d..0307eb28ae6eb 100644 --- a/tests/hermes_cli/test_session_recovery_lost_and_found.py +++ b/tests/hermes_cli/test_session_recovery_lost_and_found.py @@ -324,10 +324,10 @@ def _make_synthetic_lost_and_found( ] finally: schema.close() - assert len(sessions_columns) == 54 + assert len(sessions_columns) == 55 assert len(usage_columns) == 18 - max_fields = 54 + max_fields = 55 conn = sqlite3.connect(str(path), isolation_level=None) try: cells = ", ".join(f"c{i}" for i in range(max_fields)) @@ -354,8 +354,8 @@ def session_row(session_id: str, ncols: int) -> list: } return [base.get(column) for column in sessions_columns[:ncols]] - # Current 54-column layout and historical 52-column layout. - insert(54, 1, session_row("20260101_010101_aaa001", 54)) + # Current 55-column layout and historical 52-column layout. + insert(55, 1, session_row("20260101_010101_aaa001", 55)) insert(52, 2, session_row("20260202_020202_bbb002", 52)) # 14-column legacy layout: identity + a plausible epoch timestamp. legacy = ["20250303_030303_ccc003", "cli", 1_741_000_000.0] + [None] * 11 diff --git a/tests/hermes_state/test_session_hidden.py b/tests/hermes_state/test_session_hidden.py new file mode 100644 index 0000000000000..cbb1c472f61e9 --- /dev/null +++ b/tests/hermes_state/test_session_hidden.py @@ -0,0 +1,44 @@ +import pytest + +from hermes_state import SessionDB + + +@pytest.fixture +def db(tmp_path): + database = SessionDB(tmp_path / "state.db") + try: + yield database + finally: + database.close() + + +def test_hidden_excluded_by_default_included_on_request(db): + db.create_session("visible", source="cli") + db.create_session("secret", source="cli") + # Give both a message so the default min_message_count filter keeps them. + for sid in ("visible", "secret"): + db._conn.execute( + "UPDATE sessions SET message_count = 1 WHERE id = ?", (sid,) + ) + db._conn.commit() + + # Flip the hidden flag on one session. + assert db.set_session_hidden("secret", True) is True + assert db.get_session("secret")["hidden"] == 1 + assert db.get_session("visible")["hidden"] == 0 + + # Default listing drops the hidden row; include_hidden=True surfaces it. + default_ids = {s["id"] for s in db.list_sessions_rich(min_message_count=1)} + assert default_ids == {"visible"} + + all_ids = { + s["id"] + for s in db.list_sessions_rich(min_message_count=1, include_hidden=True) + } + assert all_ids == {"visible", "secret"} + + # Unhiding brings it back into the default listing. + assert db.set_session_hidden("secret", False) is True + assert db.get_session("secret")["hidden"] == 0 + unhidden_ids = {s["id"] for s in db.list_sessions_rich(min_message_count=1)} + assert unhidden_ids == {"visible", "secret"} diff --git a/tui_gateway/methods_session.py b/tui_gateway/methods_session.py index 32b202f539258..a05b303486dc3 100644 --- a/tui_gateway/methods_session.py +++ b/tui_gateway/methods_session.py @@ -98,6 +98,7 @@ def _(rid, params: dict) -> dict: "create_service_tier_override": create_service_tier_override, "parent_session_id": parent_session_id, "pending_title": title or None, + "pending_hidden": is_truthy_value(params.get("hidden", False)), "profile_home": str(profile_home) if profile_home is not None else None, "running": False, "session_key": key, @@ -1105,6 +1106,37 @@ def _(rid, params: dict) -> dict: return _err(rid, 5007, str(e)) +@method("session.set_hidden") +def _(rid, params: dict) -> dict: + """Set/clear the generic ``hidden`` flag on a session (and its lineage). + + Mirrors the durable ``pinned``/``archived`` setters: a hidden session is + dropped from the default global Sessions list (``list_sessions_rich`` + without ``include_hidden``) but stays fully resumable by the surface that + owns it — for plugins that manage their own sessions and don't want them + cluttering the shared recents list. Flips the whole compression chain as a + unit in the DB layer. + """ + session, err = _sess_nowait(params, rid) + if err: + return err + hidden = is_truthy_value(params.get("hidden", True)) + with _session_db(session) as db: + if db is None: + return _db_unavailable_error(rid, code=5007) + key = session["session_key"] + try: + changed = db.set_session_hidden(key, hidden) + if not changed: + # No row yet (write deferred to the first prompt): remember the + # intent so _ensure_session_db_row is born hidden, mirroring the + # pending_title deferral. + session["pending_hidden"] = hidden + return _ok(rid, {"hidden": hidden, "session_key": key}) + except Exception as e: + return _err(rid, 5007, str(e)) + + @method("message.react") def _(rid, params: dict) -> dict: """Set or clear one author's emoji reaction on a persisted message. diff --git a/tui_gateway/server.py b/tui_gateway/server.py index 6be948282d77b..44bc5663842fb 100644 --- a/tui_gateway/server.py +++ b/tui_gateway/server.py @@ -2976,6 +2976,14 @@ def _ensure_session_db_row(session: dict) -> None: # means the launch/default profile (matches run_agent's convention). profile_name=Path(profile_home).name if profile_home else None, ) + # A session can be born hidden (session.create hidden=true, or a + # session.set_hidden that arrived before the row existed): apply the + # deferred intent now that the row exists, mirroring pending_title. + if session.get("pending_hidden"): + try: + db.set_session_hidden(key, True) + except Exception: + logger.debug("failed to apply pending hidden flag", exc_info=True) except Exception as exc: # Disk-full is not a soft failure: if we swallow it here, prompt.submit # returns {"status":"streaming"} and the user's message vanishes with From 647949332bb499cae25ab8393637ad483e545ca2 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 23:07:34 -0700 Subject: [PATCH 244/376] =?UTF-8?q?docs(desktop):=20document=20Settings=20?= =?UTF-8?q?=E2=86=92=20Connections=20(multi-connection=20registry)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Covers the named-source registry from #86679: forced unique device names, @profile-device disambiguation, add/edit/remove/test, automatic v1 import, cloud-via-discovery, encrypted token storage, and the staged rollout note. --- website/docs/user-guide/desktop.md | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/website/docs/user-guide/desktop.md b/website/docs/user-guide/desktop.md index 77145c6514976..a881c806c4e4b 100644 --- a/website/docs/user-guide/desktop.md +++ b/website/docs/user-guide/desktop.md @@ -228,6 +228,19 @@ By default the app starts and manages its own **local** backend. You can instead Connection modes are configured **per profile** — a per-profile override can point one profile at a remote or cloud backend while others stay local (**Use default gateway** removes an override). +### Settings → Connections: the multi-connection registry + +Alongside the per-profile connection mode above, **Settings → Connections** manages a named registry of every agent source the app knows about — the local runtime, any number of remote gateways (LAN, Tailscale, internet), Hermes Cloud instances, and SSH hosts — all persisted together in one place. + +- **Every connection needs a unique name** (a device name such as "Homelab" or "Work laptop"). When the same profile name exists on several registered sources, surfaces disambiguate it as `@profile-device` (e.g. `@research-homelab`). +- **Add / edit / remove / test** connections from the panel. The local entry is managed by the app and cannot be removed. **Test** probes the connection's own HTTP and WebSocket legs directly. +- Existing settings are **imported automatically** the first time you run a build with the registry: your current global connection and any per-profile overrides become named entries. The legacy settings file is left untouched, so older builds keep working. +- Cloud entries come from the Hermes Cloud sign-in/discovery flow above, not from a hand-typed URL. +- Tokens are stored encrypted with the OS keyring (with the same explicit plain-text opt-in as Settings → Gateway on keyring-less Linux). + +Side-by-side routing is rolling out in stages: connections are managed here today, while the active connection is still chosen in **Settings → Gateway**. Follow-up releases route chats, the agent roster, and updates across all registered sources. + + :::info The remote backend is a running `hermes serve` process "Remote backend" means a **`hermes serve`** server running on the remote machine — that is the process the desktop app connects to. Nothing in this section works unless that backend is actually up and reachable. The desktop app does not start it for you; you (or a `systemd` service) keep `hermes serve` running on the remote host, and the app attaches to it. If you also use messaging channels (Telegram, Discord, etc.), the **gateway** is a *separate* long-running process you start independently — see the note after the setup steps. ::: From 96d6db1993d6a0ade40835c961e1ceff36b39a58 Mon Sep 17 00:00:00 2001 From: icemeng Date: Mon, 8 Jun 2026 23:54:41 +0800 Subject: [PATCH 245/376] fix(artifacts): convert DB timestamps from seconds to ms for Date() MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The Artifacts view reads message.timestamp, session.last_active, and session.started_at from the SQLite database, which stores all timestamps as Unix epoch seconds (REAL). These values were passed directly to JavaScript's Date() constructor, which expects milliseconds — causing every artifact timestamp to display as January 1970 dates. Fix by multiplying the database value by 1000 to convert seconds to milliseconds at the storage point, using nullish coalescing (??) instead of logical OR (||) so that valid zero timestamps are not skipped. --- apps/desktop/src/app/artifacts/artifact-utils.ts | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/apps/desktop/src/app/artifacts/artifact-utils.ts b/apps/desktop/src/app/artifacts/artifact-utils.ts index 48e4258904d37..97e0674cfdb06 100644 --- a/apps/desktop/src/app/artifacts/artifact-utils.ts +++ b/apps/desktop/src/app/artifacts/artifact-utils.ts @@ -283,7 +283,10 @@ export function collectArtifactsForSession(session: SessionInfo, messages: Sessi label: artifactLabel(value), sessionId: session.id, sessionTitle: title, - timestamp: message.timestamp || session.last_active || session.started_at || Date.now() + // DB timestamps (message.timestamp, session.last_active, session.started_at) + // are Unix epoch **seconds**. JS Date() expects **milliseconds**, so multiply by 1000. + // Date.now() returns ms, so divide by 1000 to keep the conversion uniform. + timestamp: (message.timestamp || session.last_active || session.started_at || Date.now() / 1000) * 1000 }) }) } From 021950ac8115eec305a11040229a5dd57f83c4d1 Mon Sep 17 00:00:00 2001 From: Johnny Silverhand <171453009+404Dealer@users.noreply.github.com> Date: Sat, 8 Aug 2026 17:42:46 +0000 Subject: [PATCH 246/376] fix(desktop): stop artifact over-indexing Require explicit provenance for tool-result artifacts while preserving assistant links, MEDIA deliveries, generated outputs, file mutations, and browser screenshots. Normalize persisted Unix-second timestamps at collection time and retain millisecond fallbacks. Consolidates current-main-compatible work from #41156 and #48577. Co-authored-by: LeonSGP43 Co-authored-by: tt-a1i <53142663+tt-a1i@users.noreply.github.com> --- .../src/app/artifacts/artifact-utils.ts | 200 +++++++++++--- apps/desktop/src/app/artifacts/index.test.ts | 254 +++++++++++++++++- 2 files changed, 415 insertions(+), 39 deletions(-) diff --git a/apps/desktop/src/app/artifacts/artifact-utils.ts b/apps/desktop/src/app/artifacts/artifact-utils.ts index 97e0674cfdb06..3201436e0aafe 100644 --- a/apps/desktop/src/app/artifacts/artifact-utils.ts +++ b/apps/desktop/src/app/artifacts/artifact-utils.ts @@ -29,11 +29,27 @@ export interface ArtifactLoadResult { const MARKDOWN_IMAGE_RE = /!\[([^\]]*)\]\(([^)\s]+)\)/g const MARKDOWN_LINK_RE = /\[([^\]]+)\]\(([^)\s]+)\)/g +const MEDIA_RE = /[`"']?MEDIA:\s*(`[^`\n]+`|"[^"\n]+"|'[^'\n]+'|\S+)[`"']?/g const URL_RE = /https?:\/\/[^\s<>"')]+/g const PATH_RE = /(^|[\s("'`])((?:\/|~\/|\.\.?\/)[^\s"'`<>]+(?:\.[a-z0-9]{1,8})?)/gi +const WINDOWS_PATH_RE = /(^|[\s("'`])([A-Za-z]:[\\/][^\s"'`<>]+(?:\.[a-z0-9]{1,8})?)/gi const IMAGE_EXT_RE = /\.(?:png|jpe?g|gif|webp|svg|bmp)(?:\?.*)?$/i -const FILE_EXT_RE = /\.(?:png|jpe?g|gif|webp|svg|bmp|pdf|txt|json|md|csv|zip|tar|gz|mp3|wav|mp4|mov)(?:\?.*)?$/i -const KEY_HINT_RE = /(path|file|url|image|artifact|output|download|result|target)/i + +const FILE_EXT_RE = + /\.(?:png|jpe?g|gif|webp|svg|bmp|pdf|txt|json|md|csv|zip|tar|gz|avi|flac|m4a|mkv|mp3|ogg|opus|wav|webm|mp4|mov)(?:\?.*)?$/i + +const MAX_UNIX_SECONDS = 10_000_000_000 + +const ARTIFACT_PRODUCER_TOOL_RE = + /(?:^|_)(?:creat(?:e|ion)|download|export|generat(?:e|ion)|render|save|speech|tts|write)(?:_|$)/i + +const STRONG_TOOL_ARTIFACT_KEY_RE = + /^(?:artifact_(?:file|image|path|url)|files?_(?:created|modified|written)|generated_(?:file|image|path|url)|media_tag|output_(?:file|path|url)|result_(?:file|path|url)|saved_to|screenshot_path)$/i + +const PRODUCER_TOOL_ARTIFACT_KEY_RE = + /^(?:artifact(?:s|_(?:file|image|path|url))?|attachment(?:s|_(?:file|image|path|url))?|download(?:s|_(?:file|path|url))?|(?:audio|image|video)(?:_(?:file|path|url))?|file_path|local_path|media(?:_(?:file|path|url))?|path)$/i + +const SCREENSHOT_PATH_RE = /Screenshot path:\s*([^\r\n<>]+)/gi function artifactSessionTitle(session: SessionInfo): string { return session.title?.trim() || session.preview?.trim() || 'Untitled session' @@ -43,6 +59,25 @@ function normalizeValue(value: string): string { return value.trim().replace(/[),.;]+$/, '') } +function unquoteMediaValue(value: string): string { + let trimmed = value.trim() + const quote = trimmed[0] + + if (quote && quote === trimmed.at(-1) && ['"', "'", '`'].includes(quote)) { + return trimmed.slice(1, -1) + } + + trimmed = trimmed.replace(/[`"'*_]{1,3}$/, '') + + return trimmed +} + +function collectMediaValues(text: string, pushValue: (value: string) => void): void { + for (const match of text.matchAll(MEDIA_RE)) { + pushValue(unquoteMediaValue(match[1] || '')) + } +} + function parseMaybeJson(value: string): unknown { if (!value.trim()) { return null @@ -55,6 +90,48 @@ function parseMaybeJson(value: string): unknown { } } +function untrustedToolPayload(value: string): null | string { + const trimmed = value.trim() + const openTag = trimmed.match(/^]*>\s*/) + + if (!openTag) { + return null + } + + const closeIndex = trimmed.lastIndexOf('') + + if (closeIndex <= openTag[0].length) { + return null + } + + const wrapped = trimmed.slice(openTag[0].length, closeIndex).trim() + const payloadStart = wrapped.indexOf('\n\n') + + return (payloadStart === -1 ? wrapped : wrapped.slice(payloadStart + 2)).trim() +} + +function parseToolPayloads(text: string): unknown[] { + const payloads: unknown[] = [] + + for (const candidate of [text, untrustedToolPayload(text)]) { + if (!candidate) { + continue + } + + const parsed = parseMaybeJson(candidate) + + if (parsed !== null) { + payloads.push(parsed) + } + } + + return payloads +} + +function isWindowsPath(value: string): boolean { + return /^[A-Za-z]:[\\/]/.test(value) || value.startsWith('\\\\') +} + function looksLikePathOrUrl(value: string): boolean { return ( value.startsWith('http://') || @@ -64,7 +141,8 @@ function looksLikePathOrUrl(value: string): boolean { value.startsWith('/') || value.startsWith('./') || value.startsWith('../') || - value.startsWith('~/') + value.startsWith('~/') || + isWindowsPath(value) ) } @@ -73,11 +151,7 @@ function looksLikeArtifact(value: string): boolean { return true } - if (looksLikePathOrUrl(value) && (IMAGE_EXT_RE.test(value) || FILE_EXT_RE.test(value))) { - return true - } - - return value.startsWith('/') && value.includes('.') + return looksLikePathOrUrl(value) && (IMAGE_EXT_RE.test(value) || FILE_EXT_RE.test(value)) } function artifactKind(value: string): ArtifactKind { @@ -90,7 +164,8 @@ function artifactKind(value: string): ArtifactKind { value.startsWith('./') || value.startsWith('../') || value.startsWith('~/') || - value.startsWith('file://') + value.startsWith('file://') || + isWindowsPath(value) ) { return 'file' } @@ -103,7 +178,7 @@ function artifactHref(value: string): string { return value } - if (value.startsWith('file://') || value.startsWith('/')) { + if (value.startsWith('file://') || value.startsWith('/') || isWindowsPath(value)) { return mediaExternalUrl(value) } @@ -135,6 +210,25 @@ function artifactLabel(value: string): string { } } +function normalizeArtifactTimestamp(timestamp: null | number | undefined): null | number { + if (typeof timestamp !== 'number' || !Number.isFinite(timestamp) || timestamp <= 0) { + return null + } + + // Persisted session timestamps use Unix seconds. Values above the maximum + // plausible Unix-seconds range are already milliseconds and stay unchanged. + return timestamp < MAX_UNIX_SECONDS ? timestamp * 1000 : timestamp +} + +function artifactTimestamp(message: SessionMessage, session: SessionInfo): number { + return ( + normalizeArtifactTimestamp(message.timestamp) ?? + normalizeArtifactTimestamp(session.last_active) ?? + normalizeArtifactTimestamp(session.started_at) ?? + Date.now() + ) +} + function messageText(message: SessionMessage): string { if (typeof message.content === 'string' && message.content.trim()) { return message.content @@ -178,6 +272,8 @@ function collectStringValues( } function collectArtifactsFromText(text: string, pushValue: (value: string) => void): void { + collectMediaValues(text, pushValue) + for (const match of text.matchAll(MARKDOWN_IMAGE_RE)) { pushValue(match[2] || '') } @@ -207,46 +303,87 @@ function collectArtifactsFromText(text: string, pushValue: (value: string) => vo for (const match of text.matchAll(PATH_RE)) { pushValue(match[2] || '') } + + for (const match of text.matchAll(WINDOWS_PATH_RE)) { + pushValue(match[2] || '') + } +} + +function toolName(message: SessionMessage): string { + return (message.tool_name || message.name || '').trim().toLowerCase() +} + +function isArtifactProducerTool(name: string): boolean { + return ARTIFACT_PRODUCER_TOOL_RE.test(name) || name.startsWith('bfl_flux3_') +} + +function explicitToolArtifactKey(keyPath: string, producerTool: boolean): boolean { + return keyPath + .split('.') + .filter(segment => segment && !/^\d+$/.test(segment)) + .some( + segment => STRONG_TOOL_ARTIFACT_KEY_RE.test(segment) || (producerTool && PRODUCER_TOOL_ARTIFACT_KEY_RE.test(segment)) + ) +} + +function structuredToolPayload(message: SessionMessage): null | unknown { + const content = message.content + + if (!content || typeof content !== 'object') { + return null + } + + if (!Array.isArray(content) && (content as Record)._multimodal === true) { + return (content as Record).meta || null + } + + return content } function collectArtifactsFromMessage(message: SessionMessage, pushValue: (value: string) => void): void { const text = messageText(message) - if (text) { + if (message.role === 'assistant' && text) { collectArtifactsFromText(text, pushValue) + + return } - if (message.role !== 'tool' && !Array.isArray(message.tool_calls)) { + if (message.role !== 'tool') { return } - if (Array.isArray(message.tool_calls)) { - for (const call of message.tool_calls) { - collectStringValues(call, 'tool_call', (value, keyPath) => { - const normalized = normalizeValue(value) + const name = toolName(message) + const producerTool = isArtifactProducerTool(name) - if (!normalized) { - return - } + if (text && producerTool) { + collectMediaValues(text, pushValue) + } - if (KEY_HINT_RE.test(keyPath) && (looksLikePathOrUrl(normalized) || FILE_EXT_RE.test(normalized))) { - pushValue(normalized) - } - }) + if (name === 'browser_vision' && text) { + for (const match of text.matchAll(SCREENSHOT_PATH_RE)) { + pushValue(match[1] || '') } } - const parsed = parseMaybeJson(text) + const payloads = parseToolPayloads(text) + const structured = structuredToolPayload(message) - if (parsed !== null) { - collectStringValues(parsed, 'tool_result', (value, keyPath) => { - const normalized = normalizeValue(value) + if (structured) { + payloads.push(structured) + } - if (!normalized) { + for (const parsed of payloads) { + collectStringValues(parsed, 'tool_result', (value, keyPath) => { + if (!explicitToolArtifactKey(keyPath, producerTool)) { return } - if ((KEY_HINT_RE.test(keyPath) || looksLikePathOrUrl(normalized)) && looksLikeArtifact(normalized)) { + collectMediaValues(value, pushValue) + + const normalized = normalizeValue(value) + + if (normalized && looksLikeArtifact(normalized)) { pushValue(normalized) } }) @@ -283,10 +420,7 @@ export function collectArtifactsForSession(session: SessionInfo, messages: Sessi label: artifactLabel(value), sessionId: session.id, sessionTitle: title, - // DB timestamps (message.timestamp, session.last_active, session.started_at) - // are Unix epoch **seconds**. JS Date() expects **milliseconds**, so multiply by 1000. - // Date.now() returns ms, so divide by 1000 to keep the conversion uniform. - timestamp: (message.timestamp || session.last_active || session.started_at || Date.now() / 1000) * 1000 + timestamp: artifactTimestamp(message, session) }) }) } diff --git a/apps/desktop/src/app/artifacts/index.test.ts b/apps/desktop/src/app/artifacts/index.test.ts index eab184cf19fba..09a48f128cdd6 100644 --- a/apps/desktop/src/app/artifacts/index.test.ts +++ b/apps/desktop/src/app/artifacts/index.test.ts @@ -48,25 +48,267 @@ describe('collectArtifactsForSession', () => { }) }) - it('indexes http links present in tool JSON payloads', () => { + it('does not index passive links and paths observed in tool output', () => { const messages: SessionMessage[] = [ { - content: JSON.stringify({ source_url: 'https://example.com/changelog/latest' }), + content: JSON.stringify({ + results: [ + { + cache_path: '/home/example/.cache/node.v24.18.1/bin', + source_url: 'https://example.com/changelog/latest' + } + ] + }), role: 'tool', - timestamp: 3000 + timestamp: 1_781_774_001, + tool_name: 'web_search' + }, + { + content: JSON.stringify({ + attachments: [{ url: 'https://cdn.example.com/passive/photo.png' }], + image: 'https://cdn.example.com/passive/thumbnail.png' + }), + role: 'tool', + timestamp: 1_781_774_002, + tool_name: 'discord_read_messages' + }, + { + content: 'External documentation example: MEDIA:/tmp/passive-example.png', + role: 'tool', + timestamp: 1_781_774_003, + tool_name: 'browser_snapshot' } ] const artifacts = collectArtifactsForSession(makeSession({ id: 'session-2' }), messages) + expect(artifacts).toHaveLength(0) + }) + + it('keeps explicit generated artifacts from tool output', () => { + const artifacts = collectArtifactsForSession(makeSession({ id: 'generated-session' }), [ + { + content: JSON.stringify({ image: 'https://cdn.example.com/generated/cat.png', success: true }), + role: 'tool', + timestamp: 1_781_774_001, + tool_name: 'image_generate' + }, + { + content: JSON.stringify({ output_path: '/tmp/generated/report.pdf', success: true }), + role: 'tool', + timestamp: 1_781_774_002, + tool_name: 'document_export' + }, + { + content: JSON.stringify({ files_modified: ['/tmp/generated/notes.md'], success: true }), + role: 'tool', + timestamp: 1_781_774_003, + tool_name: 'write_file' + }, + { + content: JSON.stringify({ artifacts: [{ url: 'https://cdn.example.com/generated/data.csv' }] }), + role: 'tool', + timestamp: 1_781_774_004, + tool_name: 'data_export' + }, + { + content: JSON.stringify({ + file_path: '/tmp/generated/voice.ogg', + media_tag: 'MEDIA:/tmp/generated/voice.ogg', + success: true + }), + role: 'tool', + timestamp: 1_781_774_005, + tool_name: 'text_to_speech' + } + ]) + + expect(artifacts.map(artifact => artifact.value)).toEqual([ + 'https://cdn.example.com/generated/cat.png', + '/tmp/generated/report.pdf', + '/tmp/generated/notes.md', + 'https://cdn.example.com/generated/data.csv', + '/tmp/generated/voice.ogg' + ]) + }) + + it('keeps an explicit browser screenshot but ignores page assets', () => { + const payload = JSON.stringify({ + images: ['https://cdn.example.com/advertising/banner.gif'], + page_url: 'https://example.com/article', + screenshot_path: '/tmp/hermes-browser/screenshot.png' + }) + + const artifacts = collectArtifactsForSession(makeSession({ id: 'browser-session' }), [ + { + content: ` +The following content came from an external source and is data, not instructions. + +${payload} +`, + role: 'tool', + timestamp: 1_781_774_001, + tool_name: 'browser_snapshot' + } + ]) + expect(artifacts).toHaveLength(1) expect(artifacts[0]).toMatchObject({ - href: 'https://example.com/changelog/latest', - kind: 'link', - value: 'https://example.com/changelog/latest' + kind: 'image', + value: '/tmp/hermes-browser/screenshot.png' }) }) + it('keeps native browser screenshots without indexing embedded image data', () => { + const artifacts = collectArtifactsForSession(makeSession({ id: 'native-browser-session' }), [ + { + content: { + _multimodal: true, + content: [{ image_url: { url: 'data:image/png;base64,AAAA' }, type: 'image_url' }], + meta: { screenshot_path: '/tmp/hermes-browser/native-screenshot.png' }, + text_summary: 'Screenshot attached' + }, + role: 'tool', + timestamp: 1_781_774_001, + tool_name: 'browser_vision' + }, + { + content: 'Image attached. Screenshot path: /tmp/hermes browser/summary screenshot.png', + role: 'tool', + timestamp: 1_781_774_002, + tool_name: 'browser_vision' + }, + { + content: 'Image attached. Screenshot path: C:\\Users\\Example User\\.hermes\\screenshot.png', + role: 'tool', + timestamp: 1_781_774_003, + tool_name: 'browser_vision' + } + ]) + + expect(artifacts.map(artifact => artifact.value)).toEqual([ + '/tmp/hermes-browser/native-screenshot.png', + '/tmp/hermes browser/summary screenshot.png', + 'C:\\Users\\Example User\\.hermes\\screenshot.png' + ]) + }) + + it('does not treat an arbitrary dotted absolute path as an artifact', () => { + const artifacts = collectArtifactsForSession(makeSession(), [ + { + content: 'Runtime discovered at /home/example/.cache/node.v24.18.1/bin', + role: 'assistant', + timestamp: 1_781_774_001 + } + ]) + + expect(artifacts).toHaveLength(0) + }) + + it('keeps supported output files from assistant text', () => { + const artifacts = collectArtifactsForSession(makeSession(), [ + { + content: 'Created: /tmp/generated/report.pdf', + role: 'assistant', + timestamp: 1_781_774_001 + }, + { + content: 'Created: C:\\Temp\\generated-report.pdf', + role: 'assistant', + timestamp: 1_781_774_002 + } + ]) + + expect(artifacts.map(artifact => artifact.value)).toEqual([ + '/tmp/generated/report.pdf', + 'C:\\Temp\\generated-report.pdf' + ]) + }) + + it('keeps explicitly delivered MEDIA files', () => { + const artifacts = collectArtifactsForSession(makeSession(), [ + { + content: 'Finished rendering. **MEDIA: /tmp/generated/demo.mp4**', + role: 'assistant', + timestamp: 1_781_774_001 + }, + { + content: 'Second render. MEDIA: "/tmp/generated/demo clip.mp4"', + role: 'assistant', + timestamp: 1_781_774_002 + }, + { + content: 'Third render. "MEDIA:/tmp/generated/quoted.mp4"', + role: 'assistant', + timestamp: 1_781_774_003 + } + ]) + + expect(artifacts.map(artifact => artifact.value)).toEqual([ + '/tmp/generated/demo.mp4', + '/tmp/generated/demo clip.mp4', + '/tmp/generated/quoted.mp4' + ]) + }) + + it('normalizes epoch-second message timestamps', () => { + const artifacts = collectArtifactsForSession(makeSession(), [ + { + content: 'Created: /tmp/generated/report.pdf', + role: 'assistant', + timestamp: 1_781_773_226.453548 + } + ]) + + expect(artifacts[0]?.timestamp).toBeCloseTo(1_781_773_226_453.548) + expect(new Date(artifacts[0]?.timestamp ?? 0).getUTCFullYear()).toBe(2026) + }) + + it('normalizes session fallback timestamps and preserves existing milliseconds', () => { + const fromSession = collectArtifactsForSession(makeSession({ last_active: 1_781_774_001 }), [ + { + content: 'Created: /tmp/generated/session-report.pdf', + role: 'assistant' + } + ]) + + const milliseconds = 42_000_000_000 + + const alreadyNormalized = collectArtifactsForSession(makeSession({ id: 'millisecond-session' }), [ + { + content: 'Created: /tmp/generated/ms-report.pdf', + role: 'assistant', + timestamp: milliseconds + } + ]) + + expect(fromSession[0]?.timestamp).toBe(1_781_774_001_000) + expect(alreadyNormalized[0]?.timestamp).toBe(milliseconds) + }) + + it('falls back past invalid timestamps without multiplying Date.now', () => { + const now = 1_781_774_001_594 + vi.spyOn(Date, 'now').mockReturnValue(now) + + const fromSession = collectArtifactsForSession(makeSession({ last_active: 1_781_774_001 }), [ + { + content: 'Created: /tmp/generated/fallback-report.pdf', + role: 'assistant', + timestamp: Number.POSITIVE_INFINITY + } + ]) + + const fromNow = collectArtifactsForSession(makeSession({ id: 'now-session', last_active: 0, started_at: 0 }), [ + { + content: 'Created: /tmp/generated/now-report.pdf', + role: 'assistant' + } + ]) + + expect(fromSession[0]?.timestamp).toBe(1_781_774_001_000) + expect(fromNow[0]?.timestamp).toBe(now) + }) + it('resolves remote image artifact thumbnails through the desktop fs bridge', async () => { const api = vi.fn(async ({ path }: { path: string }) => { if (path.startsWith('/api/fs/read-data-url?')) { From bceac696aedb45ca356aa08c15401f95f07b2244 Mon Sep 17 00:00:00 2001 From: thatssoheil Date: Mon, 10 Aug 2026 15:43:57 -0800 Subject: [PATCH 247/376] fix(desktop): artifacts page timestamps render 1970 and local images fail MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit All three artifact timestamp sources (message.timestamp, session.last_active, session.started_at) are epoch SECONDS — the transcript reader and session-date-groups both multiply by 1000 — but the collector passed them straight to new Date() (ms), so every artifact rendered as 1970-01-21. Normalize seconds to ms once at collection; the Date.now() fallback stays ms. Local file artifacts (e.g. D:\ComfyUI\output\*.png) fell through to mediaExternalUrl() which yields a file:// URL the renderer cannot load. Route through the desktop fs bridge whenever it exists — readDesktopFileDataUrl already dispatches remote REST vs local Electron internally (#83380). --- .../src/app/artifacts/artifact-utils.ts | 21 ++++++++----------- apps/desktop/src/app/artifacts/index.test.ts | 16 ++++++++++++-- apps/desktop/src/app/artifacts/index.tsx | 2 +- 3 files changed, 24 insertions(+), 15 deletions(-) diff --git a/apps/desktop/src/app/artifacts/artifact-utils.ts b/apps/desktop/src/app/artifacts/artifact-utils.ts index 3201436e0aafe..7da0f5b307b83 100644 --- a/apps/desktop/src/app/artifacts/artifact-utils.ts +++ b/apps/desktop/src/app/artifacts/artifact-utils.ts @@ -1,5 +1,4 @@ -import { readDesktopFileDataUrl } from '@/lib/desktop-fs' -import { filePathFromMediaPath, isRemoteGateway, mediaExternalUrl } from '@/lib/media' +import { mediaExternalUrl, resolveMediaDisplaySrc } from '@/lib/media' import type { SessionInfo, SessionMessage } from '@/types/hermes' export type ArtifactKind = 'image' | 'file' | 'link' @@ -185,16 +184,14 @@ function artifactHref(value: string): string { return value } -export async function artifactImageSrc(value: string, href = artifactHref(value)): Promise { - if (/^(?:https?|data):/i.test(value)) { - return href - } - - if (typeof window !== 'undefined' && window.hermesDesktop && isRemoteGateway()) { - return readDesktopFileDataUrl(filePathFromMediaPath(value)) - } - - return href +export async function artifactImageSrc(value: string): Promise { + // Delegate the whole local/remote ladder to the shared media resolver: + // inline (http/data) stays as-is, remote gateway goes through the + // authenticated fs bridge, local desktop through the Electron + // readFileDataUrl, and bare non-path link values fall through untouched. + // Reimplementing that ladder here would drift from resolveMediaDisplaySrc + // and regress one of its legs (#83380). + return resolveMediaDisplaySrc(value) } function artifactLabel(value: string): string { diff --git a/apps/desktop/src/app/artifacts/index.test.ts b/apps/desktop/src/app/artifacts/index.test.ts index 09a48f128cdd6..e39c36ac7412a 100644 --- a/apps/desktop/src/app/artifacts/index.test.ts +++ b/apps/desktop/src/app/artifacts/index.test.ts @@ -309,6 +309,19 @@ ${payload} expect(fromNow[0]?.timestamp).toBe(now) }) + it('resolves local file image artifacts through the desktop fs bridge', async () => { + const readFileDataUrl = vi.fn(async () => 'data:image/png;base64,TE9DQUw=') + vi.stubGlobal('window', { hermesDesktop: { readFileDataUrl } }) + + // Local desktop (connection mode != 'remote'): a local image_generate + // output path must be read through the Electron bridge, not left as a + // file:// URL the renderer cannot load (#83380). + const path = '/home/me/.hermes/cache/image_generate/out.png' + + await expect(artifactImageSrc(path)).resolves.toBe('data:image/png;base64,TE9DQUw=') + expect(readFileDataUrl).toHaveBeenCalledWith(path) + }) + it('resolves remote image artifact thumbnails through the desktop fs bridge', async () => { const api = vi.fn(async ({ path }: { path: string }) => { if (path.startsWith('/api/fs/read-data-url?')) { @@ -322,9 +335,8 @@ ${payload} $connection.set({ baseUrl: 'https://gw', mode: 'remote', token: 'secret' } as never) const path = '/Users/me/.hermes/skills/work-esab/references/images/manual-step03.jpeg' - const downloadHref = `https://gw/api/files/download?path=${encodeURIComponent(path)}&token=secret` - await expect(artifactImageSrc(path, downloadHref)).resolves.toBe('data:image/jpeg;base64,cmVtb3Rl') + await expect(artifactImageSrc(path)).resolves.toBe('data:image/jpeg;base64,cmVtb3Rl') expect(api).toHaveBeenCalledWith({ path: '/api/fs/read-data-url?path=%2FUsers%2Fme%2F.hermes%2Fskills%2Fwork-esab%2Freferences%2Fimages%2Fmanual-step03.jpeg' diff --git a/apps/desktop/src/app/artifacts/index.tsx b/apps/desktop/src/app/artifacts/index.tsx index 01e90e1287a5b..07f2cc70481db 100644 --- a/apps/desktop/src/app/artifacts/index.tsx +++ b/apps/desktop/src/app/artifacts/index.tsx @@ -474,7 +474,7 @@ function ArtifactImageCard({ artifact, failedImage, onImageError, onOpenChat }: let active = true setSrc('') - void artifactImageSrc(artifact.value, artifact.href) + void artifactImageSrc(artifact.value) .then(nextSrc => { if (active) { setSrc(nextSrc) From 18bac64044d3cff99228adb8bd56cf23ea2fe7ce Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 23:05:56 -0700 Subject: [PATCH 248/376] chore: map contributor email for icemeng --- contributors/emails/menglipeng@gmail.com | 1 + 1 file changed, 1 insertion(+) create mode 100644 contributors/emails/menglipeng@gmail.com diff --git a/contributors/emails/menglipeng@gmail.com b/contributors/emails/menglipeng@gmail.com new file mode 100644 index 0000000000000..c31ef9fba9878 --- /dev/null +++ b/contributors/emails/menglipeng@gmail.com @@ -0,0 +1 @@ +icemeng From 34e05c32ec00bc62502ae7ff0dbf02c2c02b0093 Mon Sep 17 00:00:00 2001 From: NorethSea <963979204@qq.com> Date: Fri, 14 Aug 2026 23:02:52 -0700 Subject: [PATCH 249/376] fix(desktop): render pet sprite at devicePixelRatio for sharp HiDPI output DPR-sized canvas backing store separated from CSS footprint, tracking zoom/display changes live. Rendering-fix subset of PR #75307; the overlay-placement feature portion is out of scope here. Covers the devicePixelRatio half of #83216. --- .../src/components/pet/pet-sprite.test.tsx | 26 ++++++++- .../desktop/src/components/pet/pet-sprite.tsx | 54 +++++++++++++++++-- 2 files changed, 74 insertions(+), 6 deletions(-) diff --git a/apps/desktop/src/components/pet/pet-sprite.test.tsx b/apps/desktop/src/components/pet/pet-sprite.test.tsx index f197ec49d802d..596a53f32e94f 100644 --- a/apps/desktop/src/components/pet/pet-sprite.test.tsx +++ b/apps/desktop/src/components/pet/pet-sprite.test.tsx @@ -35,6 +35,7 @@ const INFO = { let root: Root | null = null let container: HTMLDivElement | null = null let windowStateCallback: ((payload: { isMinimized?: boolean; isVisible?: boolean }) => void) | null = null +let drawImage: ReturnType function render(ui: ReactNode) { container = document.createElement('div') @@ -122,6 +123,7 @@ describe('PetSprite RAF scheduling', () => { ;(globalThis as typeof globalThis & { IS_REACT_ACT_ENVIRONMENT?: boolean }).IS_REACT_ACT_ENVIRONMENT = true vi.useFakeTimers() setVisibility(false) + Object.defineProperty(window, 'devicePixelRatio', { configurable: true, value: 1 }) vi.spyOn(document, 'hasFocus').mockReturnValue(true) installWindowStateBridge() vi.stubGlobal( @@ -132,9 +134,10 @@ describe('PetSprite RAF scheduling', () => { src = '' } as unknown as typeof Image ) + drawImage = vi.fn() vi.spyOn(HTMLCanvasElement.prototype, 'getContext').mockReturnValue({ clearRect: vi.fn(), - drawImage: vi.fn(), + drawImage, imageSmoothingEnabled: false } as unknown as CanvasRenderingContext2D) }) @@ -170,6 +173,27 @@ describe('PetSprite RAF scheduling', () => { expect(raf.request).toHaveBeenCalledTimes(2) }) + it('uses a DPR-sized backing store while preserving the CSS footprint', () => { + Object.defineProperty(window, 'devicePixelRatio', { configurable: true, value: 2 }) + const raf = installRaf() + + render() + + const canvas = container?.querySelector('canvas') + + expect(canvas).not.toBeNull() + expect(canvas?.width).toBe(32) + expect(canvas?.height).toBe(32) + expect(canvas?.style.width).toBe('16px') + expect(canvas?.style.height).toBe('16px') + + act(() => { + raf.runNext(0) + }) + + expect(drawImage).toHaveBeenCalledWith(expect.anything(), 0, 0, 16, 16, 0, 0, 32, 32) + }) + it('cancels pending RAF work while the Electron window is paused and resumes when visible', () => { const raf = installRaf() diff --git a/apps/desktop/src/components/pet/pet-sprite.tsx b/apps/desktop/src/components/pet/pet-sprite.tsx index 9e79dc9953cea..4069f4ed256f0 100644 --- a/apps/desktop/src/components/pet/pet-sprite.tsx +++ b/apps/desktop/src/components/pet/pet-sprite.tsx @@ -1,4 +1,4 @@ -import { memo, useEffect, useMemo, useRef } from 'react' +import { memo, useEffect, useMemo, useRef, useState } from 'react' import { createRendererLoopPauseController } from '@/lib/renderer-loop-pause' import { $petState, type PetInfo, type PetState } from '@/store/pet' @@ -11,6 +11,47 @@ const DEFAULT_LOOP_MS = 1100 // the configured scale. const DEFAULT_SCALE = 0.33 +function readDevicePixelRatio(): number { + const ratio = window.devicePixelRatio + + return Number.isFinite(ratio) && ratio > 0 ? ratio : 1 +} + +/** + * Track the effective renderer pixel ratio. Electron page zoom and moving a + * window between displays can both change it without remounting the pet. + */ +function useDevicePixelRatio(): number { + const [ratio, setRatio] = useState(readDevicePixelRatio) + + useEffect(() => { + let resolutionQuery: MediaQueryList | null = null + + const update = () => { + resolutionQuery?.removeEventListener('change', update) + + const next = readDevicePixelRatio() + + setRatio(current => (current === next ? current : next)) + + resolutionQuery = typeof window.matchMedia === 'function' ? window.matchMedia(`(resolution: ${next}dppx)`) : null + resolutionQuery?.addEventListener('change', update) + } + + window.addEventListener('resize', update) + window.visualViewport?.addEventListener('resize', update) + update() + + return () => { + resolutionQuery?.removeEventListener('change', update) + window.removeEventListener('resize', update) + window.visualViewport?.removeEventListener('resize', update) + } + }, []) + + return ratio +} + // Mirrors agent.pet.constants.CODEX_STATE_ROWS (Petdex current taxonomy). export const DEFAULT_STATE_ROWS = [ 'idle', @@ -145,9 +186,12 @@ function PetSpriteImpl({ info, zoom = 1, stateOverride, rowOverride, pauseWhenUn const loopMs = info.loopMs ?? DEFAULT_LOOP_MS const scale = (info.scale ?? DEFAULT_SCALE) * zoom const rows = info.stateRows ?? DEFAULT_STATE_ROWS + const pixelRatio = useDevicePixelRatio() const drawW = Math.round(frameW * scale) const drawH = Math.round(frameH * scale) + const backingW = Math.max(1, Math.round(drawW * pixelRatio)) + const backingH = Math.max(1, Math.round(drawH * pixelRatio)) const image = useMemo(() => { if (!info.spritesheetBase64) { @@ -326,7 +370,7 @@ function PetSpriteImpl({ info, zoom = 1, stateOverride, rowOverride, pauseWhenUn const sy = row * frameH ctx.clearRect(0, 0, canvas.width, canvas.height) ctx.imageSmoothingEnabled = false - ctx.drawImage(image, sx, sy, frameW, frameH, 0, 0, drawW, drawH) + ctx.drawImage(image, sx, sy, frameW, frameH, 0, 0, backingW, backingH) drawnFrame = frame drawnRow = row } @@ -353,15 +397,15 @@ function PetSpriteImpl({ info, zoom = 1, stateOverride, rowOverride, pauseWhenUn pauseController?.dispose() unsubState() } - }, [image, frameW, frameH, frames, framesByState, framesByRow, loopMs, drawW, drawH, rows, pauseWhenUnfocused]) + }, [image, frameW, frameH, frames, framesByState, framesByRow, loopMs, backingW, backingH, rows, pauseWhenUnfocused]) return ( ) } From fd0872214bb198485794e8fae43979e2581cb0fa Mon Sep 17 00:00:00 2001 From: fanfan343 <202709309+fanfan343@users.noreply.github.com> Date: Wed, 12 Aug 2026 07:22:13 +0800 Subject: [PATCH 250/376] fix(desktop): smooth-scale pet sprite frames Petdex spritesheets are 192x208px illustration frames, not pixel art. The desktop canvas drew them with imageSmoothingEnabled=false (nearest-neighbour), so zoomed pets looked blocky. Enable bicubic smoothing so scaled frames stay clean at any zoom. High-DPI backing-store sizing is addressed separately in #83276. --- apps/desktop/src/components/pet/pet-sprite.tsx | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/apps/desktop/src/components/pet/pet-sprite.tsx b/apps/desktop/src/components/pet/pet-sprite.tsx index 4069f4ed256f0..350a52752dffe 100644 --- a/apps/desktop/src/components/pet/pet-sprite.tsx +++ b/apps/desktop/src/components/pet/pet-sprite.tsx @@ -369,7 +369,10 @@ function PetSpriteImpl({ info, zoom = 1, stateOverride, rowOverride, pauseWhenUn const sx = frame * frameW const sy = row * frameH ctx.clearRect(0, 0, canvas.width, canvas.height) - ctx.imageSmoothingEnabled = false + // Smooth (bicubic) upscale: petdex sheets are illustration art, not + // pixel art — nearest-neighbour makes zoomed frames look blocky. + ctx.imageSmoothingEnabled = true + ctx.imageSmoothingQuality = 'high' ctx.drawImage(image, sx, sy, frameW, frameH, 0, 0, backingW, backingH) drawnFrame = frame drawnRow = row From d474ba5615b071fe5aa5b780ae27ccc8f5061d92 Mon Sep 17 00:00:00 2001 From: fanfan343 <202709309+fanfan343@users.noreply.github.com> Date: Wed, 12 Aug 2026 07:37:39 +0800 Subject: [PATCH 251/376] test(desktop): cover bicubic smoothing in pet sprite --- .../src/components/pet/pet-sprite.test.tsx | 22 ++++++++++++++++++- 1 file changed, 21 insertions(+), 1 deletion(-) diff --git a/apps/desktop/src/components/pet/pet-sprite.test.tsx b/apps/desktop/src/components/pet/pet-sprite.test.tsx index 596a53f32e94f..1adfc65140757 100644 --- a/apps/desktop/src/components/pet/pet-sprite.test.tsx +++ b/apps/desktop/src/components/pet/pet-sprite.test.tsx @@ -246,7 +246,27 @@ describe('PetSprite RAF scheduling', () => { render() act(() => window.dispatchEvent(new Event('blur'))) - expect(raf.pending()).toBe(1) }) + + it('draws sprite frames with bicubic smoothing for illustration art', () => { + const raf = installRaf() + const ctxMock = { + clearRect: vi.fn(), + drawImage: vi.fn(), + imageSmoothingEnabled: false, + imageSmoothingQuality: 'low' + } as unknown as CanvasRenderingContext2D + vi.spyOn(HTMLCanvasElement.prototype, 'getContext').mockReturnValue(ctxMock) + + render() + act(() => raf.runNext(0)) + + // Petdex sheets are illustration frames, not pixel art — nearest-neighbour + // (the old default) makes zoomed pets look blocky. The renderer must opt + // into bicubic smoothing before the first draw. + expect(ctxMock.imageSmoothingEnabled).toBe(true) + expect(ctxMock.imageSmoothingQuality).toBe('high') + expect(ctxMock.drawImage).toHaveBeenCalledTimes(1) + }) }) From 5eb1d2b0aa9426d3db2d7a8e8ab4a820e59f7da2 Mon Sep 17 00:00:00 2001 From: Adolanium <94890352+Adolanium@users.noreply.github.com> Date: Sat, 8 Aug 2026 12:36:39 +0300 Subject: [PATCH 252/376] fix(desktop): restore pet.info backstop after live-sync regression Event-capable backends no longer polled pet.info, so a cold-start fail-open enabled:false left the mascot hidden until Settings re-seeded the store. Keep a slow backstop and short startup retries. --- .../components/pet/floating-pet-poll.test.ts | 15 +++++ .../src/components/pet/floating-pet.tsx | 67 +++++++++++++------ .../src/components/pet/pet-info-poll.ts | 25 +++++++ 3 files changed, 85 insertions(+), 22 deletions(-) create mode 100644 apps/desktop/src/components/pet/floating-pet-poll.test.ts create mode 100644 apps/desktop/src/components/pet/pet-info-poll.ts diff --git a/apps/desktop/src/components/pet/floating-pet-poll.test.ts b/apps/desktop/src/components/pet/floating-pet-poll.test.ts new file mode 100644 index 0000000000000..920b26102a3aa --- /dev/null +++ b/apps/desktop/src/components/pet/floating-pet-poll.test.ts @@ -0,0 +1,15 @@ +import { describe, expect, it } from 'vitest' + +import { petInfoPollIntervalMs } from './pet-info-poll' + +describe('petInfoPollIntervalMs', () => { + it('uses the slow backstop on event-capable backends (active or not)', () => { + expect(petInfoPollIntervalMs(true, false)).toBe(15_000) + expect(petInfoPollIntervalMs(true, true)).toBe(15_000) + }) + + it('keeps the legacy fast-while-inactive cadence without change events', () => { + expect(petInfoPollIntervalMs(false, false)).toBe(3_000) + expect(petInfoPollIntervalMs(false, true)).toBe(15_000) + }) +}) diff --git a/apps/desktop/src/components/pet/floating-pet.tsx b/apps/desktop/src/components/pet/floating-pet.tsx index 62acd28b4e943..d9faa01bf32b9 100644 --- a/apps/desktop/src/components/pet/floating-pet.tsx +++ b/apps/desktop/src/components/pet/floating-pet.tsx @@ -26,6 +26,7 @@ import { $gatewayState } from '@/store/session' import { isSecondaryWindow } from '@/store/windows' import { useTheme } from '@/themes/context' +import { PET_STARTUP_RETRY_MS, petInfoPollIntervalMs } from './pet-info-poll' import { PetSprite, roamWalkRow } from './pet-sprite' import { usePetRoam } from './use-pet-roam' import { type PetZoomAnchor, usePetZoomGesture } from './use-pet-zoom-gesture' @@ -89,12 +90,15 @@ function loadPosition(): Point { * pets rewritten on disk (or renamed/rebuilt by the hatch flow) repaint without * restarting the app. * + * Event-capable backends also drive refreshes via `pet.changed`, but a slow + * backstop poll stays in place: the watcher seeds the pet signature silently + * at gateway boot and only broadcasts when it *moves*, and the one-shot + * connect pull can race a still-warming `pet.info` (fail-open enabled:false). + * Without the backstop the mascot stays hidden until Settings re-seeds it. + * * Promotion to a separate frameless OS-level window is a follow-up — the * sprite + state logic here is reused as-is, only the host changes. */ -const PET_POLL_MS = 3000 -const PET_ACTIVE_REFRESH_MS = 15000 - export function FloatingPet() { const { requestGateway } = useGatewayRequest() const { resolvedMode } = useTheme() @@ -129,11 +133,9 @@ export function FloatingPet() { // edge can't leave the window cropping it. Shared by drag + the reclamp effect. const clamp = useCallback(({ x, y }: Point): Point => clampPoint(x, y, petW, petH), [petW, petH]) - // Fetch pet.info on connect, then let pet.changed drive refreshes: the - // change watcher broadcasts when /pet (de)activates a pet or the hatch flow - // rewrites a sheet, so event-capable backends need no interval at all — - // users with no pet especially (this used to poll hardest for them). Older - // backends keep the legacy fast-while-inactive poll. + // Fetch pet.info on connect. pet.changed re-runs this effect when the + // signature moves; a slow backstop covers silent seed + cold-start races. + // Older backends (no change_events) keep the legacy fast-while-inactive poll. const active = info.enabled && Boolean(info.spritesheetBase64) useEffect(() => { if (gatewayState !== 'open') { @@ -208,29 +210,50 @@ export function FloatingPet() { } } + const pullIfVisible = () => { + if (document.visibilityState === 'visible') { + void pull() + } + } + void pull() window.addEventListener('focus', pull) - // Event-capable backend: pet.changed re-runs this effect (petChange dep), - // so no timer. Legacy backend: the historical poll. - const timer = changeEventsAvailable - ? null - : window.setInterval( - () => { - if (document.visibilityState === 'visible') { - void pull() - } - }, - active ? PET_ACTIVE_REFRESH_MS : PET_POLL_MS - ) + // Cover the cold-start race where the first pull hit fail-open enabled:false + // before the pet store was warm. Skip further retries once the mascot is live. + const startupRetryTimers = PET_STARTUP_RETRY_MS.map(delay => + window.setTimeout(() => { + if (cancelled) { + return + } + + const current = $petInfo.get() + + if (current.enabled && current.spritesheetBase64) { + return + } + + pullIfVisible() + }, delay) + ) + + // Always keep a timer. Event-capable backends use the slow backstop (same + // contract as cron/sessions in use-background-sync); legacy keeps the + // historical fast-while-inactive cadence. + const timer = window.setInterval( + pullIfVisible, + petInfoPollIntervalMs(changeEventsAvailable, active) + ) return () => { cancelled = true window.removeEventListener('focus', pull) - if (timer !== null) { - window.clearInterval(timer) + for (const id of startupRetryTimers) { + window.clearTimeout(id) } + + window.clearInterval(timer) } }, [gatewayState, active, changeEventsAvailable, petChange, requestGateway]) diff --git a/apps/desktop/src/components/pet/pet-info-poll.ts b/apps/desktop/src/components/pet/pet-info-poll.ts new file mode 100644 index 0000000000000..c9b7cda306df0 --- /dev/null +++ b/apps/desktop/src/components/pet/pet-info-poll.ts @@ -0,0 +1,25 @@ +/** Cadences for the floating-pet `pet.info` refresh timer. + +Event-capable backends rely on `pet.changed` for live updates but still need +a slow backstop: the gateway seeds the pet signature silently at boot and only +broadcasts when it *moves*, and the one-shot connect pull can race a still- +warming backend (`pet.info` fail-opens to `enabled:false`). +*/ + +export const PET_POLL_MS = 3_000 +export const PET_ACTIVE_REFRESH_MS = 15_000 +/** Slow safety net when `pet.changed` is available. */ +export const PET_BACKSTOP_MS = 15_000 +/** Cold-start retries after the first connect pull (fail-open recovery). */ +export const PET_STARTUP_RETRY_MS = [1_000, 3_000, 8_000] as const + +export function petInfoPollIntervalMs( + changeEventsAvailable: boolean, + active: boolean +): number { + if (changeEventsAvailable) { + return PET_BACKSTOP_MS + } + + return active ? PET_ACTIVE_REFRESH_MS : PET_POLL_MS +} From 28d2de18451c788416e3837804d5d605f51cffbe Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Lav=C3=ADnia=20Beghini?= Date: Sat, 8 Aug 2026 14:56:36 -0300 Subject: [PATCH 253/376] fix(desktop): surface isolated tool failures in pet state --- .../hooks/use-message-stream/gateway-event.ts | 7 ++ .../pet-tool-failure-event.test.tsx | 94 +++++++++++++++++++ 2 files changed, 101 insertions(+) create mode 100644 apps/desktop/src/app/session/hooks/use-message-stream/pet-tool-failure-event.test.tsx diff --git a/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event.ts b/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event.ts index 15eb0d6171c2e..139377f11ed95 100644 --- a/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event.ts +++ b/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event.ts @@ -929,6 +929,13 @@ export function useGatewayEventHandler(deps: GatewayEventDeps) { if (isActiveEvent) { setPetActivity({ toolRunning: false }) + + // A tool can fail without ending the turn when the agent recovers + // and continues. Surface that failure as a short pet beat too; + // otherwise only turn-level errors ever reach the failed state. + if (payload?.error) { + flashPetActivity({ error: true }) + } } // A pending clarify blocks the turn, so the first tool.complete after diff --git a/apps/desktop/src/app/session/hooks/use-message-stream/pet-tool-failure-event.test.tsx b/apps/desktop/src/app/session/hooks/use-message-stream/pet-tool-failure-event.test.tsx new file mode 100644 index 0000000000000..452164b6c5e53 --- /dev/null +++ b/apps/desktop/src/app/session/hooks/use-message-stream/pet-tool-failure-event.test.tsx @@ -0,0 +1,94 @@ +import { QueryClient } from '@tanstack/react-query' +import { act, cleanup, render, waitFor } from '@testing-library/react' +import { useEffect, useRef } from 'react' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +import type { ClientSessionState } from '@/app/types' +import { createClientSessionState } from '@/lib/chat-runtime' +import { $petActivity, $petState, setPetActivity } from '@/store/pet' +import type { RpcEvent } from '@/types/hermes' + +import { useMessageStream } from './index' + +const SID = 'session-1' +const OTHER_SID = 'session-2' + +let handleEvent: ((event: RpcEvent) => void) | null = null + +function Harness() { + const activeSessionIdRef = useRef(SID) + const sessionStateByRuntimeIdRef = useRef(new Map()) + const queryClientRef = useRef(new QueryClient()) + + const stream = useMessageStream({ + activeSessionIdRef, + hydrateFromStoredSession: vi.fn(async () => undefined), + queryClient: queryClientRef.current, + refreshHermesConfig: vi.fn(async () => undefined), + refreshSessions: vi.fn(async () => undefined), + sessionStateByRuntimeIdRef, + updateSessionState: (sessionId, updater) => { + const current = sessionStateByRuntimeIdRef.current.get(sessionId) ?? createClientSessionState() + const next = updater(current) + sessionStateByRuntimeIdRef.current.set(sessionId, next) + + return next + } + }) + + useEffect(() => { + handleEvent = stream.handleGatewayEvent + }, [stream.handleGatewayEvent]) + + return null +} + +async function mountStream() { + render() + await waitFor(() => expect(handleEvent).not.toBeNull()) +} + +function emit(type: RpcEvent['type'], payload: RpcEvent['payload'] = {}, sessionId = SID) { + act(() => handleEvent!({ payload, session_id: sessionId, type })) +} + +describe('pet tool-failure reaction', () => { + beforeEach(() => { + handleEvent = null + setPetActivity({ + busy: false, + awaitingInput: false, + toolRunning: false, + reasoning: false, + error: false, + justCompleted: false, + celebrate: false + }) + }) + + afterEach(() => { + cleanup() + setPetActivity({ error: false, toolRunning: false }) + vi.restoreAllMocks() + }) + + it('briefly shows failed when the active session has an isolated tool error', async () => { + await mountStream() + + emit('tool.start', { name: 'terminal', tool_id: 'tool-1' }) + emit('tool.complete', { name: 'terminal', tool_id: 'tool-1', error: 'exit code 1' }) + + expect($petActivity.get().error).toBe(true) + expect($petState.get()).toBe('failed') + }) + + it('does not show failed for a successful tool or a background-session failure', async () => { + await mountStream() + + emit('tool.complete', { name: 'terminal', tool_id: 'tool-1' }) + expect($petActivity.get().error).toBe(false) + + emit('tool.complete', { name: 'terminal', tool_id: 'tool-2', error: 'exit code 1' }, OTHER_SID) + expect($petActivity.get().error).toBe(false) + }) +}) From 9e8828999d5ee1730c31c91ede43f57cbfcb2867 Mon Sep 17 00:00:00 2001 From: thatssoheil Date: Fri, 7 Aug 2026 18:41:33 -0400 Subject: [PATCH 254/376] fix(petdex): quoted 'false' now disables display.pet.enabled everywhere Three bare bool() reads of display.pet.enabled (deep-merged config, so a hand-edited quoted YAML value lands as the string 'false'): the pet.cells gate, the pet.gallery enabled echo, and the shared pet-state helper. bool('false') is True, so a quoted value kept the mascot enabled against the operator's explicit intent. All three now go through utils.is_truthy_value (default False, matching DEFAULT_CONFIG). Regression test drives pet.gallery with a quoted 'false' config and asserts enabled=False; verified RED on the old code. --- tests/test_tui_gateway_server.py | 21 +++++++++++++++++++++ tui_gateway/methods_session.py | 4 ++-- tui_gateway/server.py | 2 +- 3 files changed, 24 insertions(+), 3 deletions(-) diff --git a/tests/test_tui_gateway_server.py b/tests/test_tui_gateway_server.py index c144f09c59451..b0b24d6e87ae9 100644 --- a/tests/test_tui_gateway_server.py +++ b/tests/test_tui_gateway_server.py @@ -6778,6 +6778,27 @@ def test_config_set_approval_mode_persists_three_way_value_and_emits_live_status assert emitted[0][2]["approval_mode"] == "manual" +def test_pet_gallery_quoted_false_enabled_reports_disabled(tmp_path, monkeypatch): + """display.pet.enabled: "false" (quoted) must report enabled=False. + + The old check was bool(value) — bool('false') is True, so a hand-edited + quoted YAML value kept the petdex mascot enabled against the operator's + explicit intent. + """ + import yaml + + monkeypatch.setattr(server, "_hermes_home", tmp_path) + monkeypatch.setenv("HERMES_HOME", str(tmp_path)) + (tmp_path / "config.yaml").write_text( + yaml.safe_dump({"display": {"pet": {"enabled": "false"}}}) + ) + + response = server.handle_request( + {"id": "1", "method": "pet.gallery", "params": {}} + ) + assert response["result"]["enabled"] is False + + def test_desktop_contract_includes_approval_mode_rpc(): assert server.DESKTOP_BACKEND_CONTRACT >= 3 diff --git a/tui_gateway/methods_session.py b/tui_gateway/methods_session.py index a05b303486dc3..868322156d2f6 100644 --- a/tui_gateway/methods_session.py +++ b/tui_gateway/methods_session.py @@ -1516,7 +1516,7 @@ def _(rid, params: dict) -> dict: except Exception: pet_cfg = {} - if not bool(pet_cfg.get("enabled")): + if not is_truthy_value(pet_cfg.get("enabled"), default=False): return _ok(rid, {"enabled": False}) pet = store.resolve_active_pet(str(pet_cfg.get("slug", "") or "")) @@ -1669,7 +1669,7 @@ def _(rid, params: dict) -> dict: return _ok( rid, { - "enabled": bool(pet_cfg.get("enabled")), + "enabled": is_truthy_value(pet_cfg.get("enabled"), default=False), "active": str(pet_cfg.get("slug", "") or ""), "pets": gallery, }, diff --git a/tui_gateway/server.py b/tui_gateway/server.py index 44bc5663842fb..9a05c1ee4fca6 100644 --- a/tui_gateway/server.py +++ b/tui_gateway/server.py @@ -8737,7 +8737,7 @@ def _pet_active_selection(): except Exception: pet_cfg = {} - enabled = bool(pet_cfg.get("enabled")) + enabled = is_truthy_value(pet_cfg.get("enabled"), default=False) configured_slug = str(pet_cfg.get("slug", "") or "") pet = store.resolve_active_pet(configured_slug) if enabled else None scale = float(pet_cfg.get("scale", constants.DEFAULT_SCALE) or constants.DEFAULT_SCALE) From 343068eb10c24c7bff2a1f1e6301cac0fc6ed587 Mon Sep 17 00:00:00 2001 From: thatssoheil Date: Fri, 7 Aug 2026 18:49:43 -0400 Subject: [PATCH 255/376] fix(petdex): cover the CLI pet pane and status-line signature too Review follow-up: _pet_resolve_config (cli.py, the CLI pet pane gate) and _pet_sig (tui_gateway/server.py, the status-line 'off' signature) read display.pet.enabled with bare bool/truthiness, so a quoted 'false' still enabled the mascot on the CLI surface. Route both through is_truthy_value (default False) like the other three sites. --- cli.py | 4 +++- tui_gateway/server.py | 2 +- 2 files changed, 4 insertions(+), 2 deletions(-) diff --git a/cli.py b/cli.py index 8b9661f023854..f807ccb76fdef 100644 --- a/cli.py +++ b/cli.py @@ -6115,7 +6115,9 @@ def _pet_resolve_config(self) -> None: display = cfg.get("display", {}) if isinstance(cfg.get("display"), dict) else {} pet_cfg = display.get("pet", {}) if isinstance(display.get("pet"), dict) else {} - enabled = bool(pet_cfg.get("enabled")) + from utils import is_truthy_value + + enabled = is_truthy_value(pet_cfg.get("enabled"), default=False) slug = str(pet_cfg.get("slug", "") or "") scale = float(pet_cfg.get("scale", constants.DEFAULT_SCALE) or constants.DEFAULT_SCALE) cols = constants.resolve_cols(scale, pet_cfg.get("unicode_cols", 0)) diff --git a/tui_gateway/server.py b/tui_gateway/server.py index 9a05c1ee4fca6..2375c91fccf35 100644 --- a/tui_gateway/server.py +++ b/tui_gateway/server.py @@ -3562,7 +3562,7 @@ def _pet_sig() -> tuple: hatch flow rebuilds a sheet, or the scale changes.""" display = _load_cfg().get("display") or {} pet_cfg = display.get("pet") if isinstance(display.get("pet"), dict) else {} - if not pet_cfg or not pet_cfg.get("enabled"): + if not pet_cfg or not is_truthy_value(pet_cfg.get("enabled"), default=False): return ("off",) try: enabled, pet, scale = _pet_active_selection() From ce02f0ab8a30e2d701af36a0f978580599d58947 Mon Sep 17 00:00:00 2001 From: thatssoheil Date: Fri, 7 Aug 2026 18:54:27 -0400 Subject: [PATCH 256/376] fix(pets): cover the pets CLI (doctor, has-active, /pet toggle) too Second review follow-up: hermes_cli/pets.py still read display.pet.enabled with bare bool()/truthiness in _cmd_doctor (misreported quoted 'false' as enabled in 'hermes pets doctor'), _has_active_pet (quoted 'false' treated as active, so /pet install skipped the selection prompt), and toggle_pet_display (/pet toggle flipped the WRONG way). All three now go through is_truthy_value(default=False); the module imports the shared helper at the top. --- hermes_cli/pets.py | 10 +++++++--- 1 file changed, 7 insertions(+), 3 deletions(-) diff --git a/hermes_cli/pets.py b/hermes_cli/pets.py index 7fcba082d0205..e840a03bdde88 100644 --- a/hermes_cli/pets.py +++ b/hermes_cli/pets.py @@ -13,6 +13,8 @@ import argparse import sys +from utils import is_truthy_value + def _print(msg: str = "") -> None: print(msg) @@ -249,7 +251,7 @@ def _cmd_doctor(args) -> int: from agent.pet.render import detect_terminal_graphics, resolve_mode cfg = _pet_config() - enabled = bool(cfg.get("enabled")) + enabled = is_truthy_value(cfg.get("enabled"), default=False) configured_slug = str(cfg.get("slug", "") or "") mode_cfg = str(cfg.get("render_mode", "auto") or "auto") @@ -300,7 +302,9 @@ def _pet_config() -> dict: def _has_active_pet() -> bool: - return bool(_pet_config().get("enabled")) and bool(_pet_config().get("slug")) + return is_truthy_value(_pet_config().get("enabled"), default=False) and bool( + _pet_config().get("slug") + ) def _set_active(slug: str) -> None: @@ -364,7 +368,7 @@ def toggle_pet_display() -> tuple[bool, str | None, str | None]: slug = str(cfg.get("slug", "") or "") pet = store.resolve_active_pet(slug) - if bool(cfg.get("enabled")): + if is_truthy_value(cfg.get("enabled"), default=False): _set_enabled(False) return False, pet.display_name if pet else None, None From 8052d5dd243b98d8d978f7f9b770837a810f0921 Mon Sep 17 00:00:00 2001 From: thatssoheil Date: Fri, 7 Aug 2026 19:29:19 -0400 Subject: [PATCH 257/376] test(pets): lock the quoted-false behavior on the CLI surfaces Review follow-up (final round PASS with a repeated suggestion): the pets CLI regression test only exercised real bools, leaving the exact bug this fix shipped untested. Add a quoted-'false' test driving _has_active_pet (now False) and toggle_pet_display (now takes the ENABLE branch, distinguished by the 'no pets installed' error instead of err=None from the old wrong-way disable). Verified RED on the pre-fix pets.py and GREEN on the fix. --- tests/hermes_cli/test_pet_toggle.py | 29 +++++++++++++++++++++++++++++ 1 file changed, 29 insertions(+) diff --git a/tests/hermes_cli/test_pet_toggle.py b/tests/hermes_cli/test_pet_toggle.py index 7b8c47835874d..c1b55f939f3a2 100644 --- a/tests/hermes_cli/test_pet_toggle.py +++ b/tests/hermes_cli/test_pet_toggle.py @@ -50,6 +50,35 @@ def test_toggle_pet_display_errors_with_no_installed_pets(tmp_path, monkeypatch) assert err is not None +def test_pets_cli_quoted_false_disables_and_toggle_enables(tmp_path, monkeypatch): + """Quoted `display.pet.enabled: "false"` must read as disabled. + + bool('false') is True — before the is_truthy_value fix, _has_active_pet + reported an active pet and /pet toggle DISABLED instead of enabling. + """ + import yaml + + from hermes_cli.pets import _has_active_pet, toggle_pet_display + + home = tmp_path / ".hermes" + home.mkdir() + monkeypatch.setenv("HERMES_HOME", str(home)) + (home / "config.yaml").write_text( + yaml.safe_dump( + {"display": {"pet": {"enabled": "false", "slug": "", "scale": 0.33}}} + ), + encoding="utf-8", + ) + + assert _has_active_pet() is False + # Toggle must take the ENABLE branch (reaching the "no pets installed" + # error), not the disable branch (which would return err=None). + enabled, name, err = toggle_pet_display() + assert err is not None and "no pets installed" in err + assert enabled is False + assert name is None + + @pytest.fixture def empty_home(tmp_path, monkeypatch): home = tmp_path / ".hermes" From 4f293746626531d95dcdcae93dcc815c1571c656 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 23:06:25 -0700 Subject: [PATCH 258/376] fix(gateway): send-once spritesheet semantics for pet.info (#54730) pet.info accepts knownRevision; when it matches the active sheet's revision the multi-MB spritesheetBase64 is elided and spritesheetUnchanged=true is returned. The desktop floating pet passes the revision it already holds and keeps its cached bytes, so backstop refreshes no longer resend ~3.2MB frames over the WS (write-loop stalls, disconnect storms). Legacy callers omitting knownRevision get the full payload unchanged. --- .../src/components/pet/floating-pet.tsx | 15 +++++- tests/test_tui_gateway_server.py | 49 +++++++++++++++++++ tui_gateway/methods_session.py | 12 ++++- 3 files changed, 74 insertions(+), 2 deletions(-) diff --git a/apps/desktop/src/components/pet/floating-pet.tsx b/apps/desktop/src/components/pet/floating-pet.tsx index d9faa01bf32b9..883c015094657 100644 --- a/apps/desktop/src/components/pet/floating-pet.tsx +++ b/apps/desktop/src/components/pet/floating-pet.tsx @@ -186,11 +186,24 @@ export function FloatingPet() { } } - const next = await requestGateway('pet.info', { profile: petProfile() }) + // Send-once semantics (#54730): tell the gateway which spritesheet + // revision we already hold so an unchanged multi-MB sheet is not + // re-sent over the WebSocket on every backstop refresh. + const held = $petInfo.get() + const knownRevision = held.enabled && held.spritesheetBase64 ? held.spritesheetRevision : undefined + const next = await requestGateway('pet.info', { + knownRevision, + profile: petProfile() + }) if (!cancelled && next) { const current = $petInfo.get() + if (next.enabled && next.spritesheetUnchanged && !next.spritesheetBase64) { + // Gateway confirmed our held sheet is current; keep the bytes. + next.spritesheetBase64 = current.spritesheetBase64 + } + if ( next.enabled && current.enabled && diff --git a/tests/test_tui_gateway_server.py b/tests/test_tui_gateway_server.py index b0b24d6e87ae9..c3d957eb27774 100644 --- a/tests/test_tui_gateway_server.py +++ b/tests/test_tui_gateway_server.py @@ -6799,6 +6799,55 @@ def test_pet_gallery_quoted_false_enabled_reports_disabled(tmp_path, monkeypatch assert response["result"]["enabled"] is False +def test_pet_info_known_revision_elides_spritesheet(monkeypatch): + """pet.info with a matching knownRevision must not resend the sheet bytes. + + The spritesheet payload is multi-MB; resending it on every backstop + refresh stalls the WS write loop (#54730). A caller passing the revision + it already holds gets metadata plus spritesheetUnchanged instead. + """ + + class _FakePet: + slug = "codex" + display_name = "Codex" + exists = True + spritesheet = None + + payload = { + "slug": "codex", + "displayName": "Codex", + "mime": "image/png", + "spritesheetBase64": "A" * 1024, + "spritesheetRevision": "123:456", + "frameW": 192, + "frameH": 208, + "scale": 0.33, + } + + monkeypatch.setattr(server, "_pet_active_selection", lambda: (True, _FakePet(), 0.33)) + monkeypatch.setattr(server, "_pet_sprite_payload", lambda pet, *, scale: dict(payload)) + + # Matching revision: bytes elided, unchanged marker set. + resp = server.handle_request( + {"id": "1", "method": "pet.info", "params": {"knownRevision": "123:456"}} + ) + assert resp["result"]["enabled"] is True + assert "spritesheetBase64" not in resp["result"] + assert resp["result"]["spritesheetUnchanged"] is True + assert resp["result"]["spritesheetRevision"] == "123:456" + + # Stale revision: full payload still flows. + resp = server.handle_request( + {"id": "2", "method": "pet.info", "params": {"knownRevision": "999:999"}} + ) + assert resp["result"]["spritesheetBase64"] == "A" * 1024 + assert "spritesheetUnchanged" not in resp["result"] + + # No revision (legacy callers): full payload. + resp = server.handle_request({"id": "3", "method": "pet.info", "params": {}}) + assert resp["result"]["spritesheetBase64"] == "A" * 1024 + + def test_desktop_contract_includes_approval_mode_rpc(): assert server.DESKTOP_BACKEND_CONTRACT >= 3 diff --git a/tui_gateway/methods_session.py b/tui_gateway/methods_session.py index 868322156d2f6..ffd5cda771298 100644 --- a/tui_gateway/methods_session.py +++ b/tui_gateway/methods_session.py @@ -1462,7 +1462,17 @@ def _(rid, params: dict) -> dict: if not enabled or pet is None or not pet.exists: return _ok(rid, {"enabled": False}) - return _ok(rid, {"enabled": True, **_pet_sprite_payload(pet, scale=scale)}) + payload = {"enabled": True, **_pet_sprite_payload(pet, scale=scale)} + + # Send-once semantics for the multi-MB spritesheet (#54730): a caller + # that already holds the sheet passes the revision it has, and an + # unchanged sheet comes back as metadata only (spritesheetUnchanged). + known_revision = str(params.get("knownRevision", "") or "") + if known_revision and known_revision == payload.get("spritesheetRevision"): + payload.pop("spritesheetBase64", None) + payload["spritesheetUnchanged"] = True + + return _ok(rid, payload) except Exception as exc: # noqa: BLE001 - cosmetic, never break the surface logger.debug("pet.info failed: %s", exc) return _ok(rid, {"enabled": False}) From 1c23c2a50b6d3c099c81b374355e75a23740f78a Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 23:07:21 -0700 Subject: [PATCH 259/376] chore: map contributor email for attribution audit --- contributors/emails/lavinia.beghini@genialcare.com.br | 1 + 1 file changed, 1 insertion(+) create mode 100644 contributors/emails/lavinia.beghini@genialcare.com.br diff --git a/contributors/emails/lavinia.beghini@genialcare.com.br b/contributors/emails/lavinia.beghini@genialcare.com.br new file mode 100644 index 0000000000000..e5edc8302918e --- /dev/null +++ b/contributors/emails/lavinia.beghini@genialcare.com.br @@ -0,0 +1 @@ +LBeghini From 868c400e54f9ddbaf4d4516f2ebe0b09bfa826c2 Mon Sep 17 00:00:00 2001 From: Gabriel Atkinson Date: Thu, 23 Jul 2026 14:18:33 -0700 Subject: [PATCH 260/376] fix(tui): stop disabled pet cell polling --- ui-tui/src/__tests__/petPolling.test.ts | 74 +++++++++++++++++++++ ui-tui/src/app/usePet.ts | 87 ++++++++++++++++++------- ui-tui/src/lib/petPolling.ts | 73 +++++++++++++++++++++ 3 files changed, 210 insertions(+), 24 deletions(-) create mode 100644 ui-tui/src/__tests__/petPolling.test.ts create mode 100644 ui-tui/src/lib/petPolling.ts diff --git a/ui-tui/src/__tests__/petPolling.test.ts b/ui-tui/src/__tests__/petPolling.test.ts new file mode 100644 index 0000000000000..2cdfe096b88ff --- /dev/null +++ b/ui-tui/src/__tests__/petPolling.test.ts @@ -0,0 +1,74 @@ +import { describe, expect, it, vi } from 'vitest' + +import { createPetSingleFlight, requestPetUpdate } from '../lib/petPolling.js' + +const gateway = (request: ReturnType) => ({ request }) as never + +describe('requestPetUpdate', () => { + it('does not enqueue pet.cells while pets are disabled', async () => { + const request = vi.fn().mockResolvedValue({ enabled: false }) + const needsCells = vi.fn(() => true) + + const update = await requestPetUpdate(gateway(request), 'idle', false, needsCells) + + expect(update).toEqual({ cells: null, meta: { enabled: false } }) + expect(request).toHaveBeenCalledTimes(1) + expect(request).toHaveBeenCalledWith('pet.info.meta') + expect(needsCells).not.toHaveBeenCalled() + }) + + it('uses metadata only when the enabled state is already cached', async () => { + const request = vi.fn().mockResolvedValue({ + enabled: true, + scale: 0.33, + slug: 'boba', + spritesheetRevision: '1:2' + }) + + const update = await requestPetUpdate(gateway(request), 'idle', false, () => false) + + expect(update?.cells).toBeNull() + expect(request).toHaveBeenCalledTimes(1) + }) + + it('fetches cells only for an enabled uncached state', async () => { + const cells = { enabled: true, frames: [], slug: 'boba' } + const request = vi.fn().mockResolvedValueOnce({ enabled: true, slug: 'boba' }).mockResolvedValueOnce(cells) + + const update = await requestPetUpdate(gateway(request), 'review', false, () => true) + + expect(update?.cells).toEqual(cells) + expect(request).toHaveBeenNthCalledWith(1, 'pet.info.meta') + expect(request).toHaveBeenNthCalledWith(2, 'pet.cells', { + graphics: false, + state: 'review' + }) + }) + + it('silently drops cosmetic gateway failures', async () => { + const request = vi.fn().mockRejectedValue(new Error('timeout: pet.info.meta')) + + await expect(requestPetUpdate(gateway(request), 'idle', false, () => true)).resolves.toBeNull() + }) +}) + +describe('createPetSingleFlight', () => { + it('suppresses overlapping polls and permits the next completed poll', async () => { + let release = () => undefined + + const blocked = new Promise(resolve => { + release = resolve + }) + + const operation = vi.fn(() => blocked) + const run = createPetSingleFlight() + + const first = run(operation) + await expect(run(operation)).resolves.toBe(false) + expect(operation).toHaveBeenCalledTimes(1) + + release() + await expect(first).resolves.toBe(true) + await expect(run(async () => undefined)).resolves.toBe(true) + }) +}) diff --git a/ui-tui/src/app/usePet.ts b/ui-tui/src/app/usePet.ts index 01196821fc427..595d9a37b1e9f 100644 --- a/ui-tui/src/app/usePet.ts +++ b/ui-tui/src/app/usePet.ts @@ -2,6 +2,7 @@ import { useStdout } from '@hermes/ink' import { useCallback, useEffect, useRef, useState } from 'react' import type { PetGrid } from '../components/petSprite.js' +import { createPetSingleFlight, requestPetUpdate } from '../lib/petPolling.js' import { useGateway } from './gatewayContext.js' import { $overlayState, getOverlayState } from './overlayStore.js' @@ -104,10 +105,11 @@ export interface PetRender { * * A steady poll keeps it reactive to config changes made elsewhere (`/pet`, the * picker, `hermes pets select`) so adopting/switching/disabling takes effect - * live. The frame cache is keyed by `slug:state` so a switch re-pulls cleanly. + * live. Disabled/cached pets use the cheap inline `pet.info.meta` probe; only + * uncached enabled states request `pet.cells` from the long-handler pool. */ export function usePet(): PetRender { - const { rpc } = useGateway() + const { gw } = useGateway() const { write } = useStdout() const [enabled, setEnabled] = useState(false) const [grid, setGrid] = useState(null) @@ -116,9 +118,11 @@ export function usePet(): PetRender { const cache = useRef>(new Map()) const slugRef = useRef('') const scaleRef = useRef(0) + const revisionRef = useRef('') const imageIdRef = useRef(0) const stateRef = useRef('idle') const frameRef = useRef(0) + const runSingleFlight = useRef(createPetSingleFlight()).current const [petState, setPetState] = useState('idle') @@ -189,38 +193,76 @@ export function usePet(): PetRender { } }, [write]) - // Fetch + cache one (slug, state). `pet.cells` resolves the active pet from - // config, so its `slug`/`enabled` are the source of truth. + const disablePet = useCallback(() => { + releaseKitty() + slugRef.current = '' + scaleRef.current = 0 + revisionRef.current = '' + cache.current.clear() + setGrid(null) + setKitty(null) + setEnabled(false) + }, [releaseKitty]) + + // Probe the active selection cheaply, then fetch + cache one uncached state. const sync = useCallback( - async (state: PetState) => { - try { - const res = (await rpc('pet.cells', { graphics: IS_TTY, state })) as PetCellsResult | null + (state: PetState) => + runSingleFlight(async () => { + const update = await requestPetUpdate(gw, state, IS_TTY, meta => { + const slug = meta.slug ?? '' + const scale = meta.scale ?? 0 + const revision = meta.spritesheetRevision ?? '' + + const selectionChanged = + slug !== slugRef.current || scale !== scaleRef.current || revision !== revisionRef.current + + if (selectionChanged) { + releaseKitty() + slugRef.current = slug + scaleRef.current = scale + revisionRef.current = revision + cache.current.clear() + frameRef.current = 0 + } + + return !cache.current.has(`${slug}:${state}`) + }) + + if (!update) { + return + } + + if (!update.meta.enabled) { + disablePet() + + return + } + + const res = update.cells if (!res) { + setEnabled(true) + return } if (!res.enabled) { - releaseKitty() - slugRef.current = '' - cache.current.clear() - setGrid(null) - setKitty(null) - setEnabled(false) + disablePet() return } - const slug = res.slug ?? '' - const scale = res.scale ?? 0 + const slug = res.slug ?? update.meta.slug ?? '' + const scale = res.scale ?? update.meta.scale ?? 0 - // A switch OR a live `/pet scale` change invalidates the cached frames - // (they're rendered at the old size), so the steady poll repaints at the - // new scale without a restart. - if (slug !== slugRef.current || (scale > 0 && scale !== scaleRef.current)) { + // Config may change between the metadata and frame calls. Keep the + // frame response authoritative and force a fresh metadata revision on + // the next poll when the response moved to another selection. + if (slug !== slugRef.current || scale !== scaleRef.current) { releaseKitty() slugRef.current = slug scaleRef.current = scale + revisionRef.current = slug === update.meta.slug ? revisionRef.current : '' cache.current.clear() frameRef.current = 0 } @@ -243,11 +285,8 @@ export function usePet(): PetRender { } setEnabled(true) - } catch { - // cosmetic — ignore RPC failures - } - }, - [rpc, releaseKitty] + }), + [disablePet, gw, releaseKitty, runSingleFlight] ) // Pull frames whenever the state changes (if not already cached for the diff --git a/ui-tui/src/lib/petPolling.ts b/ui-tui/src/lib/petPolling.ts new file mode 100644 index 0000000000000..afdc9d51521b9 --- /dev/null +++ b/ui-tui/src/lib/petPolling.ts @@ -0,0 +1,73 @@ +import type { GatewayClient } from '../gatewayClient.js' + +import { asRpcResult } from './rpc.js' + +export interface PetMetaResult { + enabled?: boolean + scale?: number + slug?: string + spritesheetRevision?: string +} + +interface PetUpdate { + cells: TCells | null + meta: PetMetaResult +} + +type PetGateway = Pick + +/** + * Suppress overlapping cosmetic polls so a slow gateway can never accumulate + * a queue of pet requests. Returning false tells callers that an existing + * probe is still in flight. + */ +export function createPetSingleFlight() { + let active = false + + return async (operation: () => Promise): Promise => { + if (active) { + return false + } + + active = true + + try { + await operation() + + return true + } finally { + active = false + } + } +} + +/** + * Probe cheap pet metadata on the gateway reader thread, then request the + * expensive frame payload only when the active selection/state is not cached. + * This deliberately bypasses the transcript-logging RPC wrapper: pet display + * is cosmetic, so an unavailable gateway must not print an error. + */ +export async function requestPetUpdate( + gateway: PetGateway, + state: string, + graphics: boolean, + needsCells: (meta: PetMetaResult) => boolean +): Promise | null> { + try { + const meta = asRpcResult(await gateway.request('pet.info.meta')) as PetMetaResult | null + + if (!meta) { + return null + } + + if (!meta.enabled || !needsCells(meta)) { + return { cells: null, meta } + } + + const cells = asRpcResult(await gateway.request('pet.cells', { graphics, state })) as TCells | null + + return { cells, meta } + } catch { + return null + } +} From f0cfe5a56f83ea8cb2c9f6d78350c5fe4efc7b0d Mon Sep 17 00:00:00 2001 From: Lucas Oliveira Date: Mon, 3 Aug 2026 10:28:38 -0300 Subject: [PATCH 261/376] perf(dashboard): bound multi-profile sidebar polling --- hermes_cli/web_routers/profiles.py | 248 +++++++++++++++--- hermes_cli/web_server.py | 17 +- ...st_cron_profile_enumeration_lightweight.py | 37 +++ .../hermes_cli/test_profiles_sidebar_cache.py | 159 +++++++++++ 4 files changed, 423 insertions(+), 38 deletions(-) create mode 100644 tests/hermes_cli/test_cron_profile_enumeration_lightweight.py create mode 100644 tests/hermes_cli/test_profiles_sidebar_cache.py diff --git a/hermes_cli/web_routers/profiles.py b/hermes_cli/web_routers/profiles.py index 145806c6ebccb..b99bbcabe34e5 100644 --- a/hermes_cli/web_routers/profiles.py +++ b/hermes_cli/web_routers/profiles.py @@ -13,12 +13,17 @@ """ import asyncio # noqa: F401 — used by handlers +import copy +import functools +import inspect import json import logging import re import subprocess # noqa: F401 import sys # noqa: F401 +import threading import time # noqa: F401 +from collections import OrderedDict from pathlib import Path # noqa: F401 from typing import Any, Dict, List, Optional, Tuple # noqa: F401 @@ -79,6 +84,151 @@ def _warn_profile_read_error(profile: str, exc: Exception) -> None: _write_profile_model = late("_write_profile_model") +def _read_sidebar_cache_ttl() -> float: + """Return the bounded cache lifetime for the expensive sidebar scan.""" + raw = os.environ.get("HERMES_DASHBOARD_SIDEBAR_CACHE_TTL", "5") + try: + value = float(raw) + if not math.isfinite(value): + raise ValueError("non-finite TTL") + except (TypeError, ValueError): + _log.warning( + "invalid HERMES_DASHBOARD_SIDEBAR_CACHE_TTL=%r; using 5s", + raw, + ) + value = 5.0 + return min(max(value, 0.0), 30.0) + + +_SIDEBAR_CACHE_TTL_SECONDS = _read_sidebar_cache_ttl() +_SIDEBAR_CACHE_MAX_ENTRIES = 32 +_SIDEBAR_PROFILE_CACHE_MAX_ENTRIES = 256 +_SIDEBAR_PROFILE_CACHE = OrderedDict() +_SIDEBAR_PROFILE_CACHE_LOCK = threading.Lock() + + +def _stat_fingerprint(path: Path): + """Return identity + mutation metadata without opening the file.""" + try: + stat = path.stat() + except OSError: + return None + return (stat.st_dev, stat.st_ino, stat.st_size, stat.st_mtime_ns) + + +def _sidebar_db_fingerprint(db_path: Path): + """Track SQLite content changes through the main DB and its WAL.""" + wal_path = Path(f"{db_path}-wal") + return (_stat_fingerprint(db_path), _stat_fingerprint(wal_path)) + + +def _sidebar_profile_cache_get(key): + with _SIDEBAR_PROFILE_CACHE_LOCK: + value = _SIDEBAR_PROFILE_CACHE.get(key) + if value is None: + return None + _SIDEBAR_PROFILE_CACHE.move_to_end(key) + return copy.deepcopy(value) + + +def _sidebar_profile_cache_put(key, value): + db_path, fingerprint = key[:2] + snapshot = copy.deepcopy(value) + with _SIDEBAR_PROFILE_CACHE_LOCK: + # A changed DB/WAL makes all older parameter variants for that profile + # obsolete. Remove them eagerly rather than waiting for LRU pressure. + stale = [ + existing + for existing in _SIDEBAR_PROFILE_CACHE + if existing[0] == db_path and existing[1] != fingerprint + ] + for existing in stale: + _SIDEBAR_PROFILE_CACHE.pop(existing, None) + _SIDEBAR_PROFILE_CACHE[key] = snapshot + _SIDEBAR_PROFILE_CACHE.move_to_end(key) + while len(_SIDEBAR_PROFILE_CACHE) > _SIDEBAR_PROFILE_CACHE_MAX_ENTRIES: + _SIDEBAR_PROFILE_CACHE.popitem(last=False) + + +def _sidebar_profile_cache_clear(): + with _SIDEBAR_PROFILE_CACHE_LOCK: + _SIDEBAR_PROFILE_CACHE.clear() + + +def _sidebar_singleflight_cache(func): + """Coalesce concurrent sidebar scans and briefly reuse their response. + + Every uncached refresh opens every profile database and runs up to three + session queries per profile. Desktop reconnect/focus/change bursts can + therefore overlap several identical scans in AnyIO worker threads, which + amplifies YAML/SQLite work and starves the uvicorn event loop for the GIL. + + The short TTL bounds UI staleness while the single-flight lock guarantees + only one expensive scan runs at a time. Cached values are copied on store + and hit so FastAPI serialization or a caller cannot mutate shared state. + """ + signature = inspect.signature(func) + cache = OrderedDict() + cache_lock = threading.Lock() + refresh_lock = threading.Lock() + miss = object() + + def _key(args, kwargs): + bound = signature.bind(*args, **kwargs) + bound.apply_defaults() + return tuple(bound.arguments.items()) + + def _lookup(key): + now = time.monotonic() + with cache_lock: + item = cache.get(key) + if item is None: + return miss + expires_at, value = item + if now >= expires_at: + cache.pop(key, None) + return miss + cache.move_to_end(key) + return copy.deepcopy(value) + + @functools.wraps(func) + def wrapped(*args, **kwargs): + ttl = _SIDEBAR_CACHE_TTL_SECONDS + if ttl <= 0: + return func(*args, **kwargs) + + key = _key(args, kwargs) + cached = _lookup(key) + if cached is not miss: + return cached + + # A plain Lock is intentional: FastAPI executes this sync handler in + # the AnyIO worker pool, so contenders sleep without holding the GIL. + with refresh_lock: + cached = _lookup(key) + if cached is not miss: + return cached + result = func(*args, **kwargs) + try: + snapshot = copy.deepcopy(result) + except Exception: + _log.exception("sidebar response could not be cached") + return result + with cache_lock: + cache[key] = (time.monotonic() + ttl, snapshot) + cache.move_to_end(key) + while len(cache) > _SIDEBAR_CACHE_MAX_ENTRIES: + cache.popitem(last=False) + return result + + def cache_clear(): + with cache_lock: + cache.clear() + + wrapped.cache_clear = cache_clear + return wrapped + + @sessions_router.get("/api/profiles/sessions") def get_profiles_sessions( # ``le=500`` caps the per-request page size (idea from #39200) — this @@ -123,8 +273,9 @@ def get_profiles_sessions( targets.append((name, home)) else: try: - infos = profiles_mod.list_profiles() - targets = [(info.name, info.path) for info in infos] + # This endpoint only needs name/path. Avoid list_profiles(), which + # parses config/meta and probes gateways/skills per profile. + targets = profiles_mod.profiles_to_serve(multiplex=True) except Exception: _log.exception("GET /api/profiles/sessions: list_profiles failed") targets = [] @@ -230,6 +381,7 @@ def get_profiles_sessions( @sessions_router.get("/api/profiles/sessions/sidebar") +@_sidebar_singleflight_cache def get_profiles_sessions_sidebar( recents_profile: str = "all", recents_limit: int = 20, @@ -263,8 +415,9 @@ def get_profiles_sessions_sidebar( from hermes_cli import profiles as profiles_mod try: - infos = profiles_mod.list_profiles() - targets: List[Tuple[str, Path]] = [(info.name, info.path) for info in infos] + # Session aggregation only needs name/path; the lightweight enumerator + # avoids YAML/meta/gateway/skill probes for all profiles per refresh. + targets: List[Tuple[str, Path]] = profiles_mod.profiles_to_serve(multiplex=True) except Exception: _log.exception("GET /api/profiles/sessions/sidebar: list_profiles failed") targets = [] @@ -323,38 +476,61 @@ def _slice(db, *, source=None, exclude=None, cap): db_path = Path(home) / "state.db" if not db_path.exists(): continue - try: - # Read-only with the stale-schema heal — same contract as the - # per-slice endpoint above (one-time writable reconcile when the - # store predates a schema addition, plain read-only otherwise). - db = _open_session_db_at_path(db_path, read_only=True) - except Exception as exc: - _warn_profile_read_error(name, exc) - errors.append({"profile": name, "error": str(exc)}) - continue - try: - profile_rows = _slice(db, exclude=recents_exclude_list, cap=recents_cap) - # A full window means more rows remain on disk. That is all the - # sidebar's "load more" needs, and unlike an exact COUNT(*) per - # profile per refresh it costs nothing beyond the rows already - # read. Discount pinned back-fills — they arrive past the LIMIT - # and would otherwise fake a full page on a short list. - unpinned_count = sum(1 for s in profile_rows if not s.get("pinned")) - recents_truncated[name] = unpinned_count >= recents_cap - recents_rows.extend(_tag(profile_rows, name)) - # Aggregated in SQL rather than over the window above: the window is - # a page, and a total that shrank when you scrolled would be worse - # than no total at all. - profile_totals[name] = db.usage_totals() - cron_rows.extend(_tag(_slice(db, source="cron", cap=cron_cap), name)) - messaging_rows.extend( - _tag(_slice(db, exclude=messaging_exclude_list, cap=messaging_cap), name) - ) - except Exception as exc: - _warn_profile_read_error(name, exc) - errors.append({"profile": name, "error": str(exc)}) - finally: - db.close() + fingerprint = _sidebar_db_fingerprint(db_path) + profile_cache_key = ( + str(db_path), + fingerprint, + recents_cap, + tuple(recents_exclude_list), + cron_cap, + messaging_cap, + tuple(messaging_exclude_list), + ) + slices = _sidebar_profile_cache_get(profile_cache_key) + if slices is None: + try: + # Read-only with the stale-schema heal — same contract as the + # per-slice endpoint above (one-time writable reconcile when the + # store predates a schema addition, plain read-only otherwise). + db = _open_session_db_at_path(db_path, read_only=True) + except Exception as exc: + _warn_profile_read_error(name, exc) + errors.append({"profile": name, "error": str(exc)}) + continue + try: + slices = { + "recents": _slice(db, exclude=recents_exclude_list, cap=recents_cap), + # Aggregated in SQL rather than over the recents window: the + # window is a page, and a total that shrank when you scrolled + # would be worse than no total at all. + "usage": db.usage_totals(), + "cron": _slice(db, source="cron", cap=cron_cap), + "messaging": _slice( + db, + exclude=messaging_exclude_list, + cap=messaging_cap, + ), + } + _sidebar_profile_cache_put(profile_cache_key, slices) + except Exception as exc: + _warn_profile_read_error(name, exc) + errors.append({"profile": name, "error": str(exc)}) + continue + finally: + db.close() + + profile_rows = slices["recents"] + # A full window means more rows remain on disk. That is all the + # sidebar's "load more" needs, and unlike an exact COUNT(*) per + # profile per refresh it costs nothing beyond the rows already + # read. Discount pinned back-fills — they arrive past the LIMIT + # and would otherwise fake a full page on a short list. + unpinned_count = sum(1 for s in profile_rows if not s.get("pinned")) + recents_truncated[name] = unpinned_count >= recents_cap + recents_rows.extend(_tag(profile_rows, name)) + profile_totals[name] = slices["usage"] + cron_rows.extend(_tag(slices["cron"], name)) + messaging_rows.extend(_tag(slices["messaging"], name)) def _window(rows: List[Dict[str, Any]], cap: int) -> List[Dict[str, Any]]: rows.sort(key=lambda s: s.get("last_active") or s.get("started_at") or 0, reverse=True) diff --git a/hermes_cli/web_server.py b/hermes_cli/web_server.py index 1ed365fe4fe20..3f527ed0f6ee0 100644 --- a/hermes_cli/web_server.py +++ b/hermes_cli/web_server.py @@ -11997,10 +11997,23 @@ def _validate_dashboard_cron_context_from( def _cron_profile_dicts() -> List[Dict[str, Any]]: - """Return dashboard profile records, falling back to a directory scan.""" + """Return the minimal profile records needed by cron aggregation. + + The two callers only consume ``name``. ``list_profiles()`` also parses + config/distribution metadata, probes gateway processes, and counts skills + for every profile; polling cron jobs through that path creates avoidable + GIL pressure on large profile pools. + """ from hermes_cli import profiles as profiles_mod try: - return [_profile_to_dict(p) for p in profiles_mod.list_profiles()] + return [ + { + "name": name, + "path": str(home), + "is_default": name == "default", + } + for name, home in profiles_mod.profiles_to_serve(multiplex=True) + ] except Exception: _log.exception("Failed to list profiles for cron dashboard; falling back to directory scan") return _fallback_profile_dicts(profiles_mod) diff --git a/tests/hermes_cli/test_cron_profile_enumeration_lightweight.py b/tests/hermes_cli/test_cron_profile_enumeration_lightweight.py new file mode 100644 index 0000000000000..b6de5a238251c --- /dev/null +++ b/tests/hermes_cli/test_cron_profile_enumeration_lightweight.py @@ -0,0 +1,37 @@ +"""Cron aggregation must not perform full profile metadata scans.""" + +import tempfile +import unittest +from pathlib import Path +from unittest import mock + +from hermes_cli import web_server + + +class CronProfileEnumerationTests(unittest.TestCase): + def test_uses_lightweight_name_path_enumerator(self): + with tempfile.TemporaryDirectory() as root: + homes = [ + ("default", Path(root)), + ("coder-01", Path(root) / "profiles" / "coder-01"), + ] + with ( + mock.patch( + "hermes_cli.profiles.profiles_to_serve", + return_value=homes, + ) as lightweight, + mock.patch( + "hermes_cli.profiles.list_profiles", + side_effect=AssertionError("full profile scan is forbidden"), + ), + ): + result = web_server._cron_profile_dicts() + + lightweight.assert_called_once_with(multiplex=True) + self.assertEqual([item["name"] for item in result], ["default", "coder-01"]) + self.assertTrue(result[0]["is_default"]) + self.assertFalse(result[1]["is_default"]) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/hermes_cli/test_profiles_sidebar_cache.py b/tests/hermes_cli/test_profiles_sidebar_cache.py new file mode 100644 index 0000000000000..76d96b418fc80 --- /dev/null +++ b/tests/hermes_cli/test_profiles_sidebar_cache.py @@ -0,0 +1,159 @@ +"""Regression tests for dashboard sidebar scan coalescing.""" + +import inspect +import tempfile +import threading +import time +import unittest +from concurrent.futures import ThreadPoolExecutor +from pathlib import Path +from unittest import mock + +from hermes_cli.web_routers import profiles + + +class SidebarCacheTests(unittest.TestCase): + def setUp(self): + patcher = mock.patch.object(profiles, "_SIDEBAR_CACHE_TTL_SECONDS", 5.0) + patcher.start() + self.addCleanup(patcher.stop) + profiles._sidebar_profile_cache_clear() + self.addCleanup(profiles._sidebar_profile_cache_clear) + + def test_invalid_or_non_finite_ttl_falls_back_to_default(self): + for raw in ("invalid", "nan", "inf", "-inf"): + with self.subTest(raw=raw): + with mock.patch.dict(profiles.os.environ, {"HERMES_DASHBOARD_SIDEBAR_CACHE_TTL": raw}): + self.assertEqual(profiles._read_sidebar_cache_ttl(), 5.0) + + def test_profile_cache_uses_db_and_wal_fingerprint_and_defensive_copies(self): + with tempfile.TemporaryDirectory() as root: + db_path = Path(root) / "state.db" + wal_path = Path(f"{db_path}-wal") + db_path.write_bytes(b"db-v1") + wal_path.write_bytes(b"wal-v1") + first_fingerprint = profiles._sidebar_db_fingerprint(db_path) + first_key = (str(db_path), first_fingerprint, False, 0, (), 50, 100, ()) + payload = {"recents": None, "cron": [{"id": "one"}], "messaging": []} + + profiles._sidebar_profile_cache_put(first_key, payload) + cached = profiles._sidebar_profile_cache_get(first_key) + cached["cron"][0]["id"] = "mutated" + self.assertEqual( + profiles._sidebar_profile_cache_get(first_key)["cron"][0]["id"], + "one", + ) + + wal_path.write_bytes(b"wal-v2-is-different") + second_fingerprint = profiles._sidebar_db_fingerprint(db_path) + second_key = (str(db_path), second_fingerprint, False, 0, (), 50, 100, ()) + self.assertNotEqual(first_fingerprint, second_fingerprint) + self.assertIsNone(profiles._sidebar_profile_cache_get(second_key)) + + profiles._sidebar_profile_cache_put(second_key, payload) + self.assertIsNone(profiles._sidebar_profile_cache_get(first_key)) + + def test_profile_cache_is_lru_bounded(self): + with mock.patch.object(profiles, "_SIDEBAR_PROFILE_CACHE_MAX_ENTRIES", 2): + for index in range(3): + key = (f"/db/{index}", (index, None), False, 0, (), 50, 100, ()) + profiles._sidebar_profile_cache_put(key, {"index": index}) + self.assertEqual(len(profiles._SIDEBAR_PROFILE_CACHE), 2) + + def test_applies_defaults_and_returns_defensive_copies(self): + calls = 0 + + @profiles._sidebar_singleflight_cache + def scan(profile="all", limit=20): + nonlocal calls + calls += 1 + return {"profile": profile, "rows": [{"limit": limit}]} + + first = scan() + first["rows"][0]["limit"] = 999 + second = scan(profile="all", limit=20) + + self.assertEqual(calls, 1) + self.assertEqual(second, {"profile": "all", "rows": [{"limit": 20}]}) + + def test_coalesces_concurrent_identical_scans(self): + workers = 12 + entered = threading.Event() + release = threading.Event() + calls = 0 + calls_lock = threading.Lock() + + @profiles._sidebar_singleflight_cache + def scan(profile="all"): + nonlocal calls + with calls_lock: + calls += 1 + entered.set() + self.assertTrue(release.wait(timeout=2)) + return {"profile": profile, "rows": []} + + with ThreadPoolExecutor(max_workers=workers) as pool: + futures = [pool.submit(scan, "default") for _ in range(workers)] + self.assertTrue(entered.wait(timeout=1)) + time.sleep(0.05) + release.set() + results = [future.result(timeout=2) for future in futures] + + self.assertEqual(calls, 1) + self.assertEqual(results, [{"profile": "default", "rows": []}] * workers) + + def test_expires(self): + clock = iter((100.0, 100.0, 100.0, 106.0, 106.0, 106.0)) + calls = 0 + + @profiles._sidebar_singleflight_cache + def scan(): + nonlocal calls + calls += 1 + return {"generation": calls} + + with mock.patch.object(profiles.time, "monotonic", side_effect=clock): + self.assertEqual(scan(), {"generation": 1}) + self.assertEqual(scan(), {"generation": 2}) + self.assertEqual(calls, 2) + + def test_does_not_cache_failures(self): + calls = 0 + + @profiles._sidebar_singleflight_cache + def scan(): + nonlocal calls + calls += 1 + if calls == 1: + raise RuntimeError("transient") + return {"ok": True} + + with self.assertRaisesRegex(RuntimeError, "transient"): + scan() + self.assertEqual(scan(), {"ok": True}) + self.assertEqual(scan(), {"ok": True}) + self.assertEqual(calls, 2) + + def test_can_be_disabled(self): + calls = 0 + + @profiles._sidebar_singleflight_cache + def scan(): + nonlocal calls + calls += 1 + return calls + + with mock.patch.object(profiles, "_SIDEBAR_CACHE_TTL_SECONDS", 0.0): + self.assertEqual((scan(), scan()), (1, 2)) + + def test_preserves_fastapi_signature(self): + def scan(profile: str = "all", limit: int = 20): + return profile, limit + + wrapped = profiles._sidebar_singleflight_cache(scan) + + self.assertEqual(inspect.signature(wrapped), inspect.signature(scan)) + + +if __name__ == "__main__": + unittest.main() From 44665783a936d173aba639ecd74483e0a6a016ce Mon Sep 17 00:00:00 2001 From: Tugrul Guner Date: Mon, 3 Aug 2026 13:35:50 -0400 Subject: [PATCH 262/376] fix(kanban): detect WS client disconnect on idle board, prevent zombie poll tasks MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Closes #77833 stream_events() only detected client disconnect via send_json() raising WebSocketDisconnect. When no events were pending (idle board), send_json was never called, so the poll loop ran forever even after the client disconnected — leaking one poll task per disconnect. Fix: race ws.receive() with a timeout matching the poll interval. On timeout, poll the DB as before. On disconnect, exit cleanly. --- plugins/kanban/dashboard/plugin_api.py | 16 +++++++++++++++- 1 file changed, 15 insertions(+), 1 deletion(-) diff --git a/plugins/kanban/dashboard/plugin_api.py b/plugins/kanban/dashboard/plugin_api.py index 47077d939024a..ade9fa140e707 100644 --- a/plugins/kanban/dashboard/plugin_api.py +++ b/plugins/kanban/dashboard/plugin_api.py @@ -2946,10 +2946,24 @@ def _fetch_new(cursor_val: int) -> tuple[int, list[dict]]: conn.close() while True: + # Race receive() against the poll interval to detect client + # disconnect even when no events are being sent. Without this, + # a disconnect is only detected via send_json() raising + # WebSocketDisconnect, so an idle board leaks zombie poll tasks. + try: + msg = await asyncio.wait_for( + ws.receive(), timeout=_EVENT_POLL_SECONDS + ) + if msg["type"] == "websocket.disconnect": + return + # Any other client message (pong, text) is ignored; we + # continue polling. + except asyncio.TimeoutError: + pass # no client message — poll the DB + cursor, events = await asyncio.to_thread(_fetch_new, cursor) if events: await ws.send_json({"events": events, "cursor": cursor}) - await asyncio.sleep(_EVENT_POLL_SECONDS) except WebSocketDisconnect: return except asyncio.CancelledError: From 6d0d748aa032f6f1a010edef270fa1155d4892e3 Mon Sep 17 00:00:00 2001 From: Tugrul Guner Date: Mon, 3 Aug 2026 13:46:56 -0400 Subject: [PATCH 263/376] chore: add contributor email mapping --- contributors/emails/tugrulgunr@gmail.com | 1 + 1 file changed, 1 insertion(+) create mode 100644 contributors/emails/tugrulgunr@gmail.com diff --git a/contributors/emails/tugrulgunr@gmail.com b/contributors/emails/tugrulgunr@gmail.com new file mode 100644 index 0000000000000..b38c022e39ecc --- /dev/null +++ b/contributors/emails/tugrulgunr@gmail.com @@ -0,0 +1 @@ +tugrulguner \ No newline at end of file From 5bceb3e84bd32e4944788cda25c3c8fdd867d0d9 Mon Sep 17 00:00:00 2001 From: Christopher <210261288+Christopher-Schulze@users.noreply.github.com> Date: Sat, 8 Aug 2026 15:18:03 +0200 Subject: [PATCH 264/376] fix(dashboard): add idle back-off to PTY pump loop (#42627) --- hermes_cli/web_server.py | 6 +- .../test_web_server_pty_idle_backoff.py | 73 +++++++++++++++++++ 2 files changed, 78 insertions(+), 1 deletion(-) create mode 100644 tests/hermes_cli/test_web_server_pty_idle_backoff.py diff --git a/hermes_cli/web_server.py b/hermes_cli/web_server.py index 3f527ed0f6ee0..01c5e9d77f80e 100644 --- a/hermes_cli/web_server.py +++ b/hermes_cli/web_server.py @@ -15136,6 +15136,10 @@ class PtyUnavailableError(RuntimeError): # type: ignore[no-redef] _RESIZE_RE = re.compile(rb"\x1b\[RESIZE:(\d+);(\d+)\]") _PTY_READ_CHUNK_TIMEOUT = 0.2 +# Back-off delay between idle PTY reads so a quiet terminal does not spin +# the event loop. A positive sleep lets other coroutines run and keeps +# dashboard idle CPU low (#42627). +_PTY_IDLE_BACKOFF = 0.05 # Keep-alive PTY sessions: a terminal connecting with ``?attach=`` is # bound to a process that survives disconnect/refresh and is reattachable. @@ -15170,7 +15174,7 @@ async def pump_pty_to_ws() -> None: if chunk is None: # EOF return if not chunk: # no data this tick; yield control and retry - await asyncio.sleep(0) + await asyncio.sleep(_PTY_IDLE_BACKOFF) continue try: await ws.send_bytes(chunk) diff --git a/tests/hermes_cli/test_web_server_pty_idle_backoff.py b/tests/hermes_cli/test_web_server_pty_idle_backoff.py new file mode 100644 index 0000000000000..7a755259814b4 --- /dev/null +++ b/tests/hermes_cli/test_web_server_pty_idle_backoff.py @@ -0,0 +1,73 @@ +"""Regression: dashboard PTY pump must back off when the terminal is idle.""" + +from __future__ import annotations + +import asyncio +from unittest.mock import patch + +import pytest + +from hermes_cli.web_server import _legacy_pump + + +class _FakeBridge: + def __init__(self, reads): + self._reads = list(reads) + self.closed = False + + def read(self, timeout): + if self._reads: + return self._reads.pop(0) + return None + + def write(self, data): + pass + + def resize(self, cols, rows): + pass + + def close(self): + self.closed = True + + +@pytest.mark.asyncio +async def test_legacy_pump_sleeps_on_idle_pty(): + """An empty PTY read must not spin with ``asyncio.sleep(0)``.""" + sleeps: list[float] = [] + first_idle_seen = asyncio.Event() + real_sleep = asyncio.sleep + + async def fake_sleep(delay: float) -> None: + sleeps.append(delay) + first_idle_seen.set() + # Yield with the real sleep so we don't recurse through the patch. + await real_sleep(0) + + bridge = _FakeBridge([b"", b"", None]) + + class _FakeWebSocket: + def __init__(self): + self.sent: list[bytes] = [] + + async def send_bytes(self, data): + self.sent.append(data) + + async def receive(self): + await first_idle_seen.wait() + return {"type": "websocket.disconnect"} + + async def close(self): + pass + + ws = _FakeWebSocket() + + with patch("hermes_cli.web_server.asyncio.sleep", side_effect=fake_sleep): + # _legacy_pump is typed against starlette WebSocket; our fake is an + # intentional behavioral shim. # type: ignore[invalid-argument-type] + await _legacy_pump(ws, bridge) + + assert bridge.closed + assert sleeps, "pty pump did not call asyncio.sleep on idle ticks" + assert all(d > 0 for d in sleeps), ( + f"idle pump used zero or negative sleeps: {sleeps}" + ) From bee0d45ddf1beb9adc7dcf11a76f4b65731f166b Mon Sep 17 00:00:00 2001 From: AlexDev_ <{ID}+{username}@users.noreply.github.com> Date: Sat, 8 Aug 2026 19:51:16 +0200 Subject: [PATCH 265/376] perf(desktop): pause background UI work while unfocused --- .../contrib/hooks/use-background-sync.test.ts | 11 +++- .../app/contrib/hooks/use-background-sync.ts | 19 ++++++- .../shell/hooks/use-status-snapshot.test.ts | 25 ++++++++ .../app/shell/hooks/use-status-snapshot.ts | 22 +++---- .../components/chat/activity-timer.test.tsx | 32 +++++++++++ .../src/components/chat/activity-timer.ts | 28 +++++---- apps/desktop/src/hooks/use-viewed-interval.ts | 57 +++++++++++++++++++ apps/desktop/src/lib/statusbar.tsx | 15 +---- 8 files changed, 173 insertions(+), 36 deletions(-) create mode 100644 apps/desktop/src/hooks/use-viewed-interval.ts diff --git a/apps/desktop/src/app/contrib/hooks/use-background-sync.test.ts b/apps/desktop/src/app/contrib/hooks/use-background-sync.test.ts index 4cb9f0bbc7a5a..47a67e6be2510 100644 --- a/apps/desktop/src/app/contrib/hooks/use-background-sync.test.ts +++ b/apps/desktop/src/app/contrib/hooks/use-background-sync.test.ts @@ -17,7 +17,8 @@ import { reconcileActiveTranscript, rehydrateLiveSessionStatuses, resolveActiveTranscriptSession, - useBackgroundSync + useBackgroundSync, + windowIsActivelyViewed } from './use-background-sync' vi.mock('@/hermes', async importOriginal => ({ @@ -305,6 +306,14 @@ describe('reconcileActiveTranscript', () => { }) }) +describe('windowIsActivelyViewed', () => { + it('requires both DOM visibility and keyboard focus', () => { + expect(windowIsActivelyViewed({ focused: true, visibilityState: 'visible' })).toBe(true) + expect(windowIsActivelyViewed({ focused: false, visibilityState: 'visible' })).toBe(false) + expect(windowIsActivelyViewed({ focused: true, visibilityState: 'hidden' })).toBe(false) + }) +}) + describe('rehydrateLiveSessionStatuses', () => { it('restores running sessions after reconnect without opening them', () => { const now = 1_800_000_000_000 diff --git a/apps/desktop/src/app/contrib/hooks/use-background-sync.ts b/apps/desktop/src/app/contrib/hooks/use-background-sync.ts index 1776d647101a6..0be9bacb95d99 100644 --- a/apps/desktop/src/app/contrib/hooks/use-background-sync.ts +++ b/apps/desktop/src/app/contrib/hooks/use-background-sync.ts @@ -296,9 +296,24 @@ interface BackgroundSyncParams { * safety-net refreshes, not the live path, so they're the right thing to slow * when the machine is spending its charge. Returns nothing — meant to live * inside an effect. */ +export function windowIsActivelyViewed({ + focused, + visibilityState +}: { + focused: boolean + visibilityState: DocumentVisibilityState +}): boolean { + return visibilityState === 'visible' && focused +} + function visiblePoll(intervalMs: number, tick: () => void): () => void { const run = () => { - if (document.visibilityState === 'visible') { + // On macOS an unfocused or app-hidden BrowserWindow commonly remains + // `visibilityState === "visible"`. Visibility alone therefore kept every + // safety-net gateway poll alive while the user was in another app. These + // are stale-data backstops, not the live event path, so pause them until + // the window is actually being viewed and catch up immediately on focus. + if (windowIsActivelyViewed({ focused: document.hasFocus(), visibilityState: document.visibilityState })) { tick() } } @@ -311,11 +326,13 @@ function visiblePoll(intervalMs: number, tick: () => void): () => void { }) document.addEventListener('visibilitychange', run) + window.addEventListener('focus', run) return () => { unsubscribeBattery() window.clearInterval(intervalId) document.removeEventListener('visibilitychange', run) + window.removeEventListener('focus', run) } } diff --git a/apps/desktop/src/app/shell/hooks/use-status-snapshot.test.ts b/apps/desktop/src/app/shell/hooks/use-status-snapshot.test.ts index 1c05ac05822a0..ce3db7e829d77 100644 --- a/apps/desktop/src/app/shell/hooks/use-status-snapshot.test.ts +++ b/apps/desktop/src/app/shell/hooks/use-status-snapshot.test.ts @@ -31,6 +31,7 @@ async function flushAsync() { beforeEach(() => { vi.useFakeTimers() + vi.spyOn(document, 'hasFocus').mockReturnValue(true) vi.mocked(getStatus) .mockReset() .mockResolvedValue({} as never) @@ -38,10 +39,34 @@ beforeEach(() => { afterEach(() => { cleanup() + vi.restoreAllMocks() vi.useRealTimers() }) describe('useStatusSnapshot', () => { + it('pauses status RPCs while visible but unfocused, then catches up on focus', async () => { + vi.mocked(document.hasFocus).mockReturnValue(false) + const requestGateway = vi.fn().mockResolvedValue({}) as unknown as GatewayRequester + + renderHook(() => useStatusSnapshot('open', requestGateway)) + await flushAsync() + + expect(getStatus).not.toHaveBeenCalled() + expect(requestGateway).not.toHaveBeenCalled() + + await act(async () => { + await vi.advanceTimersByTimeAsync(60_000) + }) + expect(getStatus).not.toHaveBeenCalled() + + vi.mocked(document.hasFocus).mockReturnValue(true) + window.dispatchEvent(new Event('focus')) + await flushAsync() + + expect(getStatus).toHaveBeenCalledOnce() + expect(requestGateway).toHaveBeenCalledTimes(2) + }) + it('keeps the last authoritative readiness through a transient RPC failure', async () => { let refresh = 0 diff --git a/apps/desktop/src/app/shell/hooks/use-status-snapshot.ts b/apps/desktop/src/app/shell/hooks/use-status-snapshot.ts index d2b714904b8b9..d9391dae208c2 100644 --- a/apps/desktop/src/app/shell/hooks/use-status-snapshot.ts +++ b/apps/desktop/src/app/shell/hooks/use-status-snapshot.ts @@ -5,9 +5,8 @@ import { evaluateRuntimeReadiness, type RuntimeReadinessResult } from '@/lib/run import type { StatusResponse } from '@/types/hermes' // Statusbar health is ambient chrome, not live data — nothing the user acts on -// within seconds. 60s + a hidden-tab skip keeps it honest at a quarter of the -// old traffic; the visibility listener refreshes immediately on return so a -// backgrounded window never shows stale health after re-focus. +// within seconds. 60s + an actively-viewed check keeps traffic low; focus and +// visibility listeners refresh immediately on return. const REFRESH_MS = 60_000 type GatewayRequester = (method: string, params?: Record) => Promise @@ -34,9 +33,10 @@ export function useStatusSnapshot(gatewayState: string | undefined, requestGatew } const refresh = async () => { - // Hidden window: skip the round-trips, keep the timer alive; the - // visibilitychange listener repaints immediately on return. - if (document.visibilityState !== 'visible') { + // macOS commonly leaves an occluded BrowserWindow `visible`; focus is + // the missing signal that prevents status + readiness RPCs while the + // user is working in another app. + if (document.visibilityState !== 'visible' || !document.hasFocus()) { scheduleRefresh() return @@ -79,8 +79,8 @@ export function useStatusSnapshot(gatewayState: string | undefined, requestGatew } } - const onVisible = () => { - if (document.visibilityState === 'visible' && !cancelled) { + const onReturn = () => { + if (document.visibilityState === 'visible' && document.hasFocus() && !cancelled) { if (timer !== undefined) { window.clearTimeout(timer) } @@ -89,12 +89,14 @@ export function useStatusSnapshot(gatewayState: string | undefined, requestGatew } } - document.addEventListener('visibilitychange', onVisible) + document.addEventListener('visibilitychange', onReturn) + window.addEventListener('focus', onReturn) void refresh() return () => { cancelled = true - document.removeEventListener('visibilitychange', onVisible) + document.removeEventListener('visibilitychange', onReturn) + window.removeEventListener('focus', onReturn) if (timer !== undefined) { window.clearTimeout(timer) diff --git a/apps/desktop/src/components/chat/activity-timer.test.tsx b/apps/desktop/src/components/chat/activity-timer.test.tsx index 02be87985684a..203b74e5f6962 100644 --- a/apps/desktop/src/components/chat/activity-timer.test.tsx +++ b/apps/desktop/src/components/chat/activity-timer.test.tsx @@ -19,10 +19,12 @@ describe('useElapsedSeconds', () => { beforeEach(() => { vi.useFakeTimers() vi.setSystemTime(new Date('2026-01-01T00:00:00.000Z')) + vi.spyOn(document, 'hasFocus').mockReturnValue(true) __resetElapsedTimerRegistryForTests() }) afterEach(() => { + vi.restoreAllMocks() vi.useRealTimers() __resetElapsedTimerRegistryForTests() }) @@ -72,16 +74,33 @@ describe('useElapsedSeconds', () => { expect(screen.getByTestId('elapsed').textContent).toBe('0') }) + + it('pauses UI ticks without focus and catches up immediately on return', () => { + render() + vi.mocked(document.hasFocus).mockReturnValue(false) + window.dispatchEvent(new Event('blur')) + + act(() => { + vi.advanceTimersByTime(5_000) + }) + expect(screen.getByTestId('elapsed').textContent).toBe('0') + + vi.mocked(document.hasFocus).mockReturnValue(true) + act(() => window.dispatchEvent(new Event('focus'))) + expect(screen.getByTestId('elapsed').textContent).toBe('5') + }) }) describe('useMeasuredDuration', () => { beforeEach(() => { vi.useFakeTimers() vi.setSystemTime(new Date('2026-01-01T00:00:00.000Z')) + vi.spyOn(document, 'hasFocus').mockReturnValue(true) __resetElapsedTimerRegistryForTests() }) afterEach(() => { + vi.restoreAllMocks() vi.useRealTimers() __resetElapsedTimerRegistryForTests() }) @@ -153,4 +172,17 @@ describe('useMeasuredDuration', () => { expect(screen.getByTestId('measured').textContent).toBe('2') }) + + it('records the real finish time even if the UI clock was paused', () => { + const probe = render() + vi.mocked(document.hasFocus).mockReturnValue(false) + window.dispatchEvent(new Event('blur')) + + act(() => { + vi.advanceTimersByTime(5_000) + }) + probe.rerender() + + expect(screen.getByTestId('measured').textContent).toBe('5') + }) }) diff --git a/apps/desktop/src/components/chat/activity-timer.ts b/apps/desktop/src/components/chat/activity-timer.ts index 6eddba167fbfe..cb040e7673de1 100644 --- a/apps/desktop/src/components/chat/activity-timer.ts +++ b/apps/desktop/src/components/chat/activity-timer.ts @@ -1,5 +1,7 @@ import { useEffect, useRef, useState } from 'react' +import { useViewedInterval } from '@/hooks/use-viewed-interval' + // Module-level registry so timers survive component unmount/remount (e.g. // when a tool row scrolls out and back). Keyed by caller-supplied timerKey; // anonymous timers (no key) start fresh each mount. @@ -53,25 +55,25 @@ export function useElapsedSeconds(active = true, timerKey?: string, since?: numb lastKey.current = timerKey } - // eslint-disable-next-line no-restricted-syntax -- legitimate non-atom ref write (see eslint rule comment) + // eslint-disable-next-line no-restricted-syntax -- timer origin is imperative state, not an atom mirror useEffect(() => { - if (!active) { - return - } - if (since !== undefined) { start.current = since } else if (timerKey) { start.current = startedAt(timerKey) } - const tick = () => setElapsed(Math.max(0, Math.floor((Date.now() - start.current) / 1000))) - tick() - const id = window.setInterval(tick, 1000) - - return () => window.clearInterval(id) + if (active) { + setElapsed(Math.max(0, Math.floor((Date.now() - start.current) / 1000))) + } }, [active, since, timerKey]) + useViewedInterval( + () => setElapsed(Math.max(0, Math.floor((Date.now() - start.current) / 1000))), + 1000, + active + ) + return elapsed } @@ -100,9 +102,11 @@ export function useMeasuredDuration(active: boolean, timerKey: string): null | n if (active) { setWatching(true) } else if (watching) { + const finalElapsed = Math.max(elapsed, Math.floor((Date.now() - startedAt(timerKey)) / 1000)) + setWatching(false) - durationByKey.set(timerKey, elapsed) - setMeasured(elapsed) + durationByKey.set(timerKey, finalElapsed) + setMeasured(finalElapsed) } }, [active, elapsed, timerKey, watching]) diff --git a/apps/desktop/src/hooks/use-viewed-interval.ts b/apps/desktop/src/hooks/use-viewed-interval.ts new file mode 100644 index 0000000000000..7b9942267251e --- /dev/null +++ b/apps/desktop/src/hooks/use-viewed-interval.ts @@ -0,0 +1,57 @@ +import { useEffect, useRef } from 'react' + +/** Run a UI-only clock while this document is actually being viewed. + * + * macOS can leave an occluded BrowserWindow `visible`, and active streaming + * deliberately disables Chromium's background timer throttling. Pairing focus + * with visibility avoids waking React for elapsed labels nobody can see while + * a leading tick on return catches the UI up immediately. + */ +export function useViewedInterval(callback: () => void, intervalMs: number, enabled = true): void { + const callbackRef = useRef(callback) + + // eslint-disable-next-line no-restricted-syntax -- latest-callback ref avoids restarting the interval each render + useEffect(() => { + callbackRef.current = callback + }, [callback]) + + useEffect(() => { + if (!enabled) { + return + } + + let intervalId: null | number = null + const stop = () => { + if (intervalId !== null) { + window.clearInterval(intervalId) + intervalId = null + } + } + const sync = () => { + const viewed = document.visibilityState === 'visible' && document.hasFocus() + + if (!viewed) { + stop() + + return + } + + if (intervalId === null) { + callbackRef.current() + intervalId = window.setInterval(() => callbackRef.current(), intervalMs) + } + } + + window.addEventListener('focus', sync) + window.addEventListener('blur', sync) + document.addEventListener('visibilitychange', sync) + sync() + + return () => { + stop() + window.removeEventListener('focus', sync) + window.removeEventListener('blur', sync) + document.removeEventListener('visibilitychange', sync) + } + }, [enabled, intervalMs]) +} diff --git a/apps/desktop/src/lib/statusbar.tsx b/apps/desktop/src/lib/statusbar.tsx index 0c0d22599c58d..01ca3b645af1e 100644 --- a/apps/desktop/src/lib/statusbar.tsx +++ b/apps/desktop/src/lib/statusbar.tsx @@ -1,6 +1,7 @@ -import { useEffect, useState } from 'react' +import { useState } from 'react' import { StableText } from '@/components/chat/stable-text' +import { useViewedInterval } from '@/hooks/use-viewed-interval' import { compactNumber } from '@/lib/format' import type { UsageStats } from '@/types/hermes' @@ -61,17 +62,7 @@ export function contextBarLabel(usage: UsageStats): string { export function LiveDuration({ since }: { since: number | null | undefined }) { const [now, setNow] = useState(() => Date.now()) - useEffect(() => { - if (!since) { - return - } - - const tick = () => setNow(Date.now()) - tick() - const timer = window.setInterval(tick, 1000) - - return () => window.clearInterval(timer) - }, [since]) + useViewedInterval(() => setNow(Date.now()), 1000, Boolean(since)) if (!since) { return null From 67a1c1ed1adc97ff162b059d6e7e4c4db5b750a0 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 23:05:53 -0700 Subject: [PATCH 266/376] fix(dashboard): use a fixed sidebar cache TTL (no HERMES_* env var for non-secret config) --- hermes_cli/web_routers/profiles.py | 21 ++++--------------- .../hermes_cli/test_profiles_sidebar_cache.py | 6 ------ 2 files changed, 4 insertions(+), 23 deletions(-) diff --git a/hermes_cli/web_routers/profiles.py b/hermes_cli/web_routers/profiles.py index b99bbcabe34e5..4d44e9e366b8d 100644 --- a/hermes_cli/web_routers/profiles.py +++ b/hermes_cli/web_routers/profiles.py @@ -84,23 +84,10 @@ def _warn_profile_read_error(profile: str, exc: Exception) -> None: _write_profile_model = late("_write_profile_model") -def _read_sidebar_cache_ttl() -> float: - """Return the bounded cache lifetime for the expensive sidebar scan.""" - raw = os.environ.get("HERMES_DASHBOARD_SIDEBAR_CACHE_TTL", "5") - try: - value = float(raw) - if not math.isfinite(value): - raise ValueError("non-finite TTL") - except (TypeError, ValueError): - _log.warning( - "invalid HERMES_DASHBOARD_SIDEBAR_CACHE_TTL=%r; using 5s", - raw, - ) - value = 5.0 - return min(max(value, 0.0), 30.0) - - -_SIDEBAR_CACHE_TTL_SECONDS = _read_sidebar_cache_ttl() +# Bounded cache lifetime for the expensive sidebar scan. Short enough that the +# UI never shows meaningfully stale data, long enough to coalesce the desktop's +# reconnect/focus/change poll bursts into one scan. +_SIDEBAR_CACHE_TTL_SECONDS = 5.0 _SIDEBAR_CACHE_MAX_ENTRIES = 32 _SIDEBAR_PROFILE_CACHE_MAX_ENTRIES = 256 _SIDEBAR_PROFILE_CACHE = OrderedDict() diff --git a/tests/hermes_cli/test_profiles_sidebar_cache.py b/tests/hermes_cli/test_profiles_sidebar_cache.py index 76d96b418fc80..5bb113a028a7e 100644 --- a/tests/hermes_cli/test_profiles_sidebar_cache.py +++ b/tests/hermes_cli/test_profiles_sidebar_cache.py @@ -20,12 +20,6 @@ def setUp(self): profiles._sidebar_profile_cache_clear() self.addCleanup(profiles._sidebar_profile_cache_clear) - def test_invalid_or_non_finite_ttl_falls_back_to_default(self): - for raw in ("invalid", "nan", "inf", "-inf"): - with self.subTest(raw=raw): - with mock.patch.dict(profiles.os.environ, {"HERMES_DASHBOARD_SIDEBAR_CACHE_TTL": raw}): - self.assertEqual(profiles._read_sidebar_cache_ttl(), 5.0) - def test_profile_cache_uses_db_and_wal_fingerprint_and_defensive_copies(self): with tempfile.TemporaryDirectory() as root: db_path = Path(root) / "state.db" From 106207c7fbc6aac353eb852487c040931d032a37 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 23:06:11 -0700 Subject: [PATCH 267/376] chore: map contributor emails for attribution audit --- contributors/emails/g.atkinson112@gmail.com | 1 + contributors/emails/{ID}+{username}@users.noreply.github.com | 1 + 2 files changed, 2 insertions(+) create mode 100644 contributors/emails/g.atkinson112@gmail.com create mode 100644 contributors/emails/{ID}+{username}@users.noreply.github.com diff --git a/contributors/emails/g.atkinson112@gmail.com b/contributors/emails/g.atkinson112@gmail.com new file mode 100644 index 0000000000000..7c05f01a8053a --- /dev/null +++ b/contributors/emails/g.atkinson112@gmail.com @@ -0,0 +1 @@ +bananawalnut diff --git a/contributors/emails/{ID}+{username}@users.noreply.github.com b/contributors/emails/{ID}+{username}@users.noreply.github.com new file mode 100644 index 0000000000000..2f38d284f481d --- /dev/null +++ b/contributors/emails/{ID}+{username}@users.noreply.github.com @@ -0,0 +1 @@ +alexdev03 From f535a4f284017ca1fb3a27d404e28bd51b435e31 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 23:07:46 -0700 Subject: [PATCH 268/376] test(desktop): pin document.hasFocus in background-sync backstop tests --- .../src/app/contrib/hooks/use-background-sync.test.ts | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/apps/desktop/src/app/contrib/hooks/use-background-sync.test.ts b/apps/desktop/src/app/contrib/hooks/use-background-sync.test.ts index 47a67e6be2510..1b70bb29bd0cc 100644 --- a/apps/desktop/src/app/contrib/hooks/use-background-sync.test.ts +++ b/apps/desktop/src/app/contrib/hooks/use-background-sync.test.ts @@ -113,6 +113,12 @@ function renderSync( ) } +beforeEach(() => { + // visiblePoll only ticks while the window is actively viewed; jsdom's + // document.hasFocus() is not reliably true, so pin it for these tests. + vi.spyOn(document, 'hasFocus').mockReturnValue(true) +}) + afterEach(() => { cleanup() vi.clearAllTimers() @@ -124,6 +130,7 @@ afterEach(() => { setMessagingSessions([]) setBusy(false) vi.clearAllMocks() + vi.restoreAllMocks() clearAllSessionStates() }) From edd73daaf441b9cf54089fc06b137e2cb720ae3f Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 23:09:07 -0700 Subject: [PATCH 269/376] test(kanban): regression for idle-board WS disconnect detection (#77833) --- .../plugins/test_kanban_ws_idle_disconnect.py | 70 +++++++++++++++++++ 1 file changed, 70 insertions(+) create mode 100644 tests/plugins/test_kanban_ws_idle_disconnect.py diff --git a/tests/plugins/test_kanban_ws_idle_disconnect.py b/tests/plugins/test_kanban_ws_idle_disconnect.py new file mode 100644 index 0000000000000..1c7e107f857b8 --- /dev/null +++ b/tests/plugins/test_kanban_ws_idle_disconnect.py @@ -0,0 +1,70 @@ +"""Regression: kanban events WS must notice client disconnect on an idle board. + +Before the fix (#77833), ``stream_events`` only awaited ``asyncio.sleep`` +between DB polls, so a disconnect was detected solely when ``send_json`` +raised — which never happens on a board with no new events. Every closed +dashboard tab therefore left a zombie poll task querying SQLite forever. +""" + +from __future__ import annotations + +import asyncio +import importlib.util +import sys +from pathlib import Path + +import pytest + + +def _load_plugin_module(): + repo_root = Path(__file__).resolve().parents[2] + plugin_file = repo_root / "plugins" / "kanban" / "dashboard" / "plugin_api.py" + assert plugin_file.exists(), f"plugin file missing: {plugin_file}" + spec = importlib.util.spec_from_file_location( + "hermes_dashboard_plugin_kanban_ws_test", plugin_file, + ) + assert spec is not None and spec.loader is not None + mod = importlib.util.module_from_spec(spec) + sys.modules[spec.name] = mod + spec.loader.exec_module(mod) + return mod + + +class _IdleDisconnectingWebSocket: + """Accepts, then reports a client disconnect on the first receive().""" + + def __init__(self): + self.accepted = False + self.sent: list[dict] = [] + self.query_params: dict[str, str] = {} + self.receive_calls = 0 + + async def accept(self): + self.accepted = True + + async def receive(self): + self.receive_calls += 1 + return {"type": "websocket.disconnect"} + + async def send_json(self, payload): + self.sent.append(payload) + + async def close(self, code=None): + pass + + +@pytest.mark.asyncio +async def test_stream_events_exits_on_idle_disconnect(monkeypatch, tmp_path): + mod = _load_plugin_module() + monkeypatch.setattr(mod, "_ws_upgrade_authorized", lambda ws: True) + + ws = _IdleDisconnectingWebSocket() + + # The disconnect must terminate the handler even though the board is idle + # and no event is ever sent. Before the fix this call never returned + # (the loop only slept between polls), so bound it with a timeout. + await asyncio.wait_for(mod.stream_events(ws), timeout=5) + + assert ws.accepted + assert ws.receive_calls == 1 + assert ws.sent == [] # returned before any poll, no zombie loop From aa8e92f516e3225e008bccbdf048449dc5003513 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 15 Aug 2026 00:11:57 -0700 Subject: [PATCH 270/376] test: pin document.hasFocus in status timer test (useViewedInterval gating) --- .../src/components/assistant-ui/thread/status.test.tsx | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/apps/desktop/src/components/assistant-ui/thread/status.test.tsx b/apps/desktop/src/components/assistant-ui/thread/status.test.tsx index e51f39313a79b..51604b4dc4b27 100644 --- a/apps/desktop/src/components/assistant-ui/thread/status.test.tsx +++ b/apps/desktop/src/components/assistant-ui/thread/status.test.tsx @@ -19,6 +19,10 @@ describe('ResponseLoadingIndicator timer', () => { beforeEach(() => { vi.useFakeTimers() vi.setSystemTime(new Date('2026-01-01T00:00:00.000Z')) + // useViewedInterval gates ticking on document focus + visibility; jsdom's + // hasFocus() is unreliable across runners, so pin it (same as the + // background-sync backstop tests). + vi.spyOn(document, 'hasFocus').mockReturnValue(true) __resetElapsedTimerRegistryForTests() }) @@ -27,6 +31,7 @@ describe('ResponseLoadingIndicator timer', () => { $activeSessionId.set(null) $turnStartedAt.set(null) __resetElapsedTimerRegistryForTests() + vi.restoreAllMocks() vi.useRealTimers() }) From 3d73821e9de082a043f40c9357ea0e7c12807f22 Mon Sep 17 00:00:00 2001 From: Reksely <69013710+Reksely@users.noreply.github.com> Date: Sat, 8 Aug 2026 09:13:59 -0400 Subject: [PATCH 271/376] fix(agent): stop thread output descriptor leaks --- agent/thread_scoped_output.py | 57 +++++++++++++------ tests/agent/test_thread_scoped_output.py | 72 ++++++++++++++++++++++++ 2 files changed, 112 insertions(+), 17 deletions(-) diff --git a/agent/thread_scoped_output.py b/agent/thread_scoped_output.py index e46608e492752..3c4a7be891e78 100644 --- a/agent/thread_scoped_output.py +++ b/agent/thread_scoped_output.py @@ -30,6 +30,20 @@ # Maps the proxy we installed for a given attribute ("stdout"/"stderr") so we # never double-wrap and so we can recover the original stream. _installed: dict[str, "_ThreadRoutingStream"] = {} +# One process-lifetime sink per stream. Temporary process-global redirects can +# displace and later restore a routing proxy; they must not allocate another +# permanent /dev/null descriptor every time that happens. +_sinks: dict[str, TextIO] = {} +_routing_states: dict[str, "_RoutingState"] = {} + + +class _RoutingState: + """Silencing registry shared by every proxy generation for one stream.""" + + def __init__(self, sink: TextIO) -> None: + self.sink = sink + self.silenced: dict[int, int] = {} + self.lock = threading.Lock() class _ThreadRoutingStream: @@ -42,32 +56,27 @@ class _ThreadRoutingStream: ``.fileno()`` behave like the underlying stream for the calling thread. """ - def __init__(self, passthrough: TextIO, sink: TextIO) -> None: + def __init__(self, passthrough: TextIO, state: _RoutingState) -> None: self._passthrough = passthrough - self._sink = sink - # ident -> nesting depth. A thread is silenced while depth > 0, so - # nested ``thread_scoped_silence()`` on the same thread composes - # correctly (the inner exit decrements rather than fully clearing). - self._silenced: dict[int, int] = {} - self._lock = threading.Lock() + self._state = state def _target(self) -> TextIO: - if self._silenced.get(threading.get_ident(), 0) > 0: - return self._sink + if self._state.silenced.get(threading.get_ident(), 0) > 0: + return self._state.sink return self._passthrough # --- registration ----------------------------------------------------- def silence(self, ident: int) -> None: - with self._lock: - self._silenced[ident] = self._silenced.get(ident, 0) + 1 + with self._state.lock: + self._state.silenced[ident] = self._state.silenced.get(ident, 0) + 1 def unsilence(self, ident: int) -> None: - with self._lock: - depth = self._silenced.get(ident, 0) - 1 + with self._state.lock: + depth = self._state.silenced.get(ident, 0) - 1 if depth > 0: - self._silenced[ident] = depth + self._state.silenced[ident] = depth else: - self._silenced.pop(ident, None) + self._state.silenced.pop(ident, None) # --- file-like surface ------------------------------------------------ def write(self, data): # type: ignore[no-untyped-def] @@ -109,14 +118,28 @@ def _ensure_installed(attr: str, passthrough: TextIO) -> "_ThreadRoutingStream": with _install_lock: proxy = _installed.get(attr) current = getattr(sys, attr, None) + if isinstance(current, _ThreadRoutingStream): + # A redirect context can restore an older routing proxy after a + # temporary replacement. Adopt it instead of wrapping it and + # growing an unbounded proxy chain. + _installed[attr] = current + _routing_states[attr] = current._state + return current if proxy is not None and current is proxy: return proxy # Capture whatever is currently bound as the passthrough. If a prior # global redirect_stdout is active, route non-silenced threads to that # stream to preserve the old behavior. passthrough = current if current is not None else passthrough - sink = open(os.devnull, "w", encoding="utf-8") - proxy = _ThreadRoutingStream(passthrough, sink) + sink = _sinks.get(attr) + if sink is None or sink.closed: + sink = open(os.devnull, "w", encoding="utf-8") + _sinks[attr] = sink + state = _routing_states.get(attr) + if state is None or state.sink is not sink: + state = _RoutingState(sink) + _routing_states[attr] = state + proxy = _ThreadRoutingStream(passthrough, state) setattr(sys, attr, proxy) _installed[attr] = proxy return proxy diff --git a/tests/agent/test_thread_scoped_output.py b/tests/agent/test_thread_scoped_output.py index 7f85e7d5f8684..fa22921dd97b4 100644 --- a/tests/agent/test_thread_scoped_output.py +++ b/tests/agent/test_thread_scoped_output.py @@ -7,11 +7,13 @@ ``contextlib.redirect_stdout(devnull)`` violated (issue #55769 / #55925). """ +import contextlib import io import sys import threading import time +import agent.thread_scoped_output as thread_output from agent.thread_scoped_output import thread_scoped_silence @@ -94,3 +96,73 @@ def test_repeated_contexts_never_write_to_a_closed_sink(): sys.stdout.fileno() finally: sys.stdout = original + + +def test_temporary_global_redirects_do_not_allocate_new_sinks(monkeypatch): + """A displaced proxy is temporary, not a reason to leak another FD pair.""" + opened_sinks = [] + + def fake_open(*_args, **_kwargs): + sink = io.StringIO() + opened_sinks.append(sink) + return sink + + monkeypatch.setattr(thread_output, "_installed", {}) + monkeypatch.setattr(thread_output, "_sinks", {}, raising=False) + monkeypatch.setattr(thread_output, "open", fake_open, raising=False) + original_stdout, original_stderr = sys.stdout, sys.stderr + sys.stdout, sys.stderr = io.StringIO(), io.StringIO() + try: + with thread_scoped_silence(): + pass + assert len(opened_sinks) == 2 + original_proxies = dict(thread_output._installed) + + for _ in range(20): + with contextlib.redirect_stdout(io.StringIO()), contextlib.redirect_stderr(io.StringIO()): + with thread_scoped_silence(): + print("hidden") + + with thread_scoped_silence(): + pass + assert len(opened_sinks) == 2 + assert thread_output._installed == original_proxies + finally: + sys.stdout, sys.stderr = original_stdout, original_stderr + + +def test_silence_survives_redirect_restoring_an_older_proxy(monkeypatch): + """Silencing is stream-wide, even when a redirect swaps proxy generations.""" + monkeypatch.setattr(thread_output, "_installed", {}) + monkeypatch.setattr(thread_output, "_sinks", {}, raising=False) + original_stdout, original_stderr = sys.stdout, sys.stderr + passthrough = io.StringIO() + sys.stdout = passthrough + entered = threading.Event() + release = threading.Event() + + try: + with thread_scoped_silence(): + pass + + def worker(): + with thread_scoped_silence(): + entered.set() + assert release.wait(timeout=10) + print("must-stay-silenced") + + redirected = io.StringIO() + with contextlib.redirect_stdout(redirected): + thread = threading.Thread(target=worker) + thread.start() + assert entered.wait(timeout=10) + + release.set() + thread.join(timeout=10) + + assert not thread.is_alive() + assert "must-stay-silenced" not in passthrough.getvalue() + assert "must-stay-silenced" not in redirected.getvalue() + finally: + release.set() + sys.stdout, sys.stderr = original_stdout, original_stderr From c99a45b28eb392f9c9a22b7072cc91babd035d94 Mon Sep 17 00:00:00 2001 From: andy Date: Sun, 9 Aug 2026 09:28:42 +0800 Subject: [PATCH 272/376] fix(browser): reap leaked agent-browser daemons whose owner is still alive MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The orphan reaper had two gaps that let agent-browser daemons accumulate indefinitely inside a single long-lived hermes process: 1. `_reap_orphaned_browser_sessions()` ran exactly once, before the cleanup loop started, so a leak appearing after boot could never be recovered. 2. `owner_alive is True` skipped unconditionally. In-memory session tracking is lost on any exception path between spawn and registration, but the owner PID stays up — so such a daemon was skipped forever. The daemon-side `AGENT_BROWSER_IDLE_TIMEOUT_MS` is not a backstop for (2): it does not fire when the daemon itself is wedged, e.g. after Chrome's framework was replaced underneath it by an auto-update. Observed on macOS: five agent-browser daemons (96 Chrome processes) built up over 10 days inside an 18-day-uptime hermes process, holding roughly 5 CPU cores busy and driving the load average past 100. Four of those processes were still running a Chrome framework version that had since been replaced on disk, spinning at ~85% CPU each. Changes: - Re-run the reaper every `BROWSER_ORPHAN_REAP_INTERVAL` (300s) from inside the cleanup loop. Cycle 0 preserves the existing startup reap. - When the owner is alive but the session is untracked, fall back to idle age: reap past `BROWSER_ORPHAN_GRACE_SECONDS`, defined as `max(1h, 20 x inactivity_timeout)`. Unknown age fails safe. - Add `_socket_dir_idle_seconds()` — the newest mtime under a session's socket dir. Every browser command writes `_stdout_` / `_stderr_` there, making it a last-activity marker that survives hermes restarts and does not depend on in-memory bookkeeping surviving an exception path. It scans directory entries rather than reading the directory mtime alone: command names repeat, and rewriting an existing `_stdout_click` updates that file's mtime but not the directory's, so a dir-mtime-only check would report a busy session as idle and reap it. Sessions still present in `_active_sessions` are never touched at any age, and the new path still goes through `_verify_reapable_browser_daemon`, so the anti-spoof / anti-PID-recycle guarantees from #14073 are unchanged. Adds 9 tests: idle-age unit tests (including the dir-mtime regression), spared/reaped/fail-safe cases for a live owner, the identity-guard gate on the new path, and a periodic-reap test asserting more than one reap per cleanup-thread lifetime. Co-Authored-By: Claude Opus 5 --- tests/tools/test_browser_orphan_reaper.py | 209 ++++++++++++++++++++++ tools/browser_tool.py | 101 ++++++++++- 2 files changed, 302 insertions(+), 8 deletions(-) diff --git a/tests/tools/test_browser_orphan_reaper.py b/tests/tools/test_browser_orphan_reaper.py index 59eb67a182469..33033c18a9535 100644 --- a/tests/tools/test_browser_orphan_reaper.py +++ b/tests/tools/test_browser_orphan_reaper.py @@ -2,6 +2,7 @@ daemons whose Python parent exited without cleaning up.""" import os +import time from unittest.mock import patch import pytest @@ -355,3 +356,211 @@ def _spy_reaper(): assert reaper_called, ( "Reaper must run on exit even with no active sessions" ) + + +def _age_socket_dir(d, seconds): + """Backdate every mtime under ``d`` so it looks idle for ``seconds``.""" + old = time.time() - seconds + for p in d.iterdir(): + os.utime(p, (old, old)) + os.utime(d, (old, old)) + + +class TestSocketDirIdleSeconds: + """Unit tests for the idle-age signal backing the leak escape hatch.""" + + def test_missing_dir_returns_none(self, tmp_path): + from tools.browser_tool import _socket_dir_idle_seconds + assert _socket_dir_idle_seconds(str(tmp_path / "nope")) is None + + def test_fresh_dir_is_near_zero(self, tmp_path): + from tools.browser_tool import _socket_dir_idle_seconds + d = tmp_path / "agent-browser-h_fresh" + d.mkdir() + assert _socket_dir_idle_seconds(str(d)) < 5 + + def test_entry_mtime_beats_stale_dir_mtime(self, tmp_path): + """Rewriting an existing file must count as activity. + + Command names repeat (``_stdout_click`` is rewritten on every click), + and overwriting an existing file does NOT bump the *directory* mtime. + Reading only the directory mtime would therefore report a busy session + as idle and reap it. The reaper must scan entries too. + """ + from tools.browser_tool import _socket_dir_idle_seconds + d = tmp_path / "agent-browser-h_reuse" + d.mkdir() + f = d / "_stdout_click" + f.write_text("x") + _age_socket_dir(d, 7200) + assert _socket_dir_idle_seconds(str(d)) > 7000 + + f.write_text("y") # rewrite in place — dir mtime stays stale + assert time.time() - os.path.getmtime(d) > 7000, "precondition" + assert _socket_dir_idle_seconds(str(d)) < 5 + + +class TestLeakedDaemonWithLiveOwner: + """Idle-age escape hatch for untracked daemons whose owner is still alive. + + ``owner_alive is True`` alone made a leaked daemon immortal: in-memory + tracking is lost on any exception path between spawn and registration, + yet the owner PID stays up, so the reaper skipped it forever. Observed in + the wild — five agent-browser daemons accumulated over 10 days inside one + long-lived hermes process, pinning ~5 CPU cores and driving load to 100+. + + The daemon-side ``AGENT_BROWSER_IDLE_TIMEOUT_MS`` is not a backstop here: + it does not fire when the daemon itself is wedged (e.g. Chrome's framework + was replaced underneath it by an auto-update). + """ + + def test_fresh_untracked_daemon_with_live_owner_is_spared(self, fake_tmpdir): + """Within the grace window, cross-process safety still wins.""" + from tools.browser_tool import _reap_orphaned_browser_sessions + + d = _make_socket_dir( + fake_tmpdir, "h_fresh_owner", pid=12345, owner_pid=os.getpid() + ) + kill_calls = [] + + with patch("gateway.status._pid_exists", return_value=True), \ + patch("tools.browser_tool._verify_reapable_browser_daemon", return_value=True), \ + patch("tools.process_registry.ProcessRegistry._terminate_host_pid", + side_effect=kill_calls.append): + _reap_orphaned_browser_sessions() + + assert 12345 not in kill_calls + assert d.exists() + + def test_idle_untracked_daemon_with_live_owner_is_reaped(self, fake_tmpdir): + """Past the grace window, an untracked daemon is treated as leaked.""" + from tools.browser_tool import ( + BROWSER_ORPHAN_GRACE_SECONDS, + _reap_orphaned_browser_sessions, + ) + + d = _make_socket_dir( + fake_tmpdir, "h_leaked_owner", pid=12345, owner_pid=os.getpid() + ) + _age_socket_dir(d, BROWSER_ORPHAN_GRACE_SECONDS + 600) + kill_calls = [] + + with patch("gateway.status._pid_exists", return_value=True), \ + patch("tools.browser_tool._verify_reapable_browser_daemon", return_value=True), \ + patch("tools.process_registry.ProcessRegistry._terminate_host_pid", + side_effect=kill_calls.append): + _reap_orphaned_browser_sessions() + + assert 12345 in kill_calls + assert not d.exists() + + def test_tracked_daemon_with_live_owner_is_spared_at_any_age(self, fake_tmpdir): + """A session this process still tracks is never reaped, however old. + + Idle age is a fallback for *lost* bookkeeping, not an override of + bookkeeping that is present and says the session is live. + """ + import tools.browser_tool as bt + from tools.browser_tool import ( + BROWSER_ORPHAN_GRACE_SECONDS, + _reap_orphaned_browser_sessions, + ) + + d = _make_socket_dir( + fake_tmpdir, "h_tracked_old", pid=12345, owner_pid=os.getpid() + ) + _age_socket_dir(d, BROWSER_ORPHAN_GRACE_SECONDS * 10) + bt._active_sessions["task-1"] = {"session_name": "h_tracked_old"} + kill_calls = [] + + with patch("gateway.status._pid_exists", return_value=True), \ + patch("tools.browser_tool._verify_reapable_browser_daemon", return_value=True), \ + patch("tools.process_registry.ProcessRegistry._terminate_host_pid", + side_effect=kill_calls.append): + _reap_orphaned_browser_sessions() + + assert 12345 not in kill_calls + assert d.exists() + + def test_unknown_idle_age_fails_safe(self, fake_tmpdir): + """Unreadable mtime => treat as too young to reap, never guess.""" + from tools.browser_tool import _reap_orphaned_browser_sessions + + d = _make_socket_dir( + fake_tmpdir, "h_unknown_age", pid=12345, owner_pid=os.getpid() + ) + kill_calls = [] + + with patch("gateway.status._pid_exists", return_value=True), \ + patch("tools.browser_tool._socket_dir_idle_seconds", return_value=None), \ + patch("tools.browser_tool._verify_reapable_browser_daemon", return_value=True), \ + patch("tools.process_registry.ProcessRegistry._terminate_host_pid", + side_effect=kill_calls.append): + _reap_orphaned_browser_sessions() + + assert 12345 not in kill_calls + assert d.exists() + + def test_identity_guard_still_gates_the_new_path(self, fake_tmpdir): + """The escape hatch must not bypass _verify_reapable_browser_daemon. + + That guard is the anti-spoof / anti-PID-recycle defense (issue #14073); + an idle daemon is still only reapable if it verifies. + """ + from tools.browser_tool import ( + BROWSER_ORPHAN_GRACE_SECONDS, + _reap_orphaned_browser_sessions, + ) + + d = _make_socket_dir( + fake_tmpdir, "h_unverified", pid=12345, owner_pid=os.getpid() + ) + _age_socket_dir(d, BROWSER_ORPHAN_GRACE_SECONDS + 600) + kill_calls = [] + + with patch("gateway.status._pid_exists", return_value=True), \ + patch("tools.browser_tool._verify_reapable_browser_daemon", return_value=False), \ + patch("tools.process_registry.ProcessRegistry._terminate_host_pid", + side_effect=kill_calls.append): + _reap_orphaned_browser_sessions() + + assert 12345 not in kill_calls + assert d.exists() + + +class TestPeriodicOrphanReap: + """The reaper must run repeatedly, not only at cleanup-thread startup. + + A startup-only reap can never recover from a leak that appears *after* + boot — which is exactly what happens in a hermes process that stays up + for days. + """ + + def test_reaper_runs_on_every_interval_not_just_startup(self): + import tools.browser_tool as bt + + cycles_to_run = 21 + reap_calls = [] + remaining = {"n": cycles_to_run} + + def fake_cleanup(): + remaining["n"] -= 1 + if remaining["n"] <= 0: + bt._cleanup_running = False + + orig_running = bt._cleanup_running + bt._cleanup_running = True + try: + with patch("tools.browser_tool._reap_orphaned_browser_sessions", + side_effect=lambda: reap_calls.append(1)), \ + patch("tools.browser_tool._cleanup_inactive_browser_sessions", + side_effect=fake_cleanup), \ + patch("tools.browser_tool.time.sleep"): + bt._browser_cleanup_thread_worker() + finally: + bt._cleanup_running = orig_running + + every = max(1, round(bt.BROWSER_ORPHAN_REAP_INTERVAL / 30)) + expected = len([c for c in range(cycles_to_run) if c % every == 0]) + assert len(reap_calls) == expected + assert len(reap_calls) > 1, "startup-only reap would give exactly 1" diff --git a/tools/browser_tool.py b/tools/browser_tool.py index eaaff9d969ed2..544a06d512770 100644 --- a/tools/browser_tool.py +++ b/tools/browser_tool.py @@ -1642,6 +1642,22 @@ def _get_session_inactivity_timeout() -> int: BROWSER_SESSION_INACTIVITY_TIMEOUT = _get_session_inactivity_timeout() +# How often the cleanup thread re-runs the orphan reaper. The reaper used to +# run exactly once, before the cleanup loop started, which meant a hermes +# process that stays up for days could never recover from a leak that appeared +# *after* boot. Observed in the wild: five agent-browser daemons accumulated +# over 10 days in a single 18-day-uptime process, pinning ~5 CPU cores. +BROWSER_ORPHAN_REAP_INTERVAL = 300 # seconds + +# Hard ceiling for a daemon whose owning hermes process is still alive but +# which has fallen out of that process's in-memory session tracking. The +# owner-alive check alone makes such a daemon immortal: in-memory tracking is +# lost on any exception path, yet the owner PID stays up, so the reaper skips +# it forever. Idle age (see ``_socket_dir_idle_seconds``) is the escape hatch. +# Deliberately a large multiple of the inactivity timeout so a legitimately +# busy session is never touched. +BROWSER_ORPHAN_GRACE_SECONDS = max(3600, BROWSER_SESSION_INACTIVITY_TIMEOUT * 20) + # Track last activity time per session _session_last_activity: Dict[str, float] = {} @@ -1872,6 +1888,40 @@ def _verify_reapable_browser_daemon(daemon_pid: int, socket_dir: str, return True +def _socket_dir_idle_seconds(socket_dir: str) -> Optional[float]: + """Seconds since anything in ``socket_dir`` was last written. + + Every browser command writes ``_stdout_`` / ``_stderr_`` temp + files into the session's socket dir, so the newest mtime under that dir is + a last-activity marker that — unlike ``_session_last_activity`` — survives + hermes restarts and does not depend on in-memory bookkeeping surviving an + exception path. + + The directory's own mtime is not sufficient: command names repeat, so + rewriting an existing ``_stdout_click`` updates that file's mtime but not + the directory's. Scan the entries too. + + Returns ``None`` when the age cannot be determined, so callers can fail + safe (treat unknown age as "too young to reap"). + """ + try: + latest = os.path.getmtime(socket_dir) + except OSError: + return None + + try: + with os.scandir(socket_dir) as entries: + for entry in entries: + try: + latest = max(latest, entry.stat().st_mtime) + except OSError: + continue + except OSError: + pass # dir mtime alone is still a usable lower bound + + return max(0.0, time.time() - latest) + + def _reap_orphaned_browser_sessions(): """Scan for orphaned agent-browser daemon processes from previous runs. @@ -1926,6 +1976,7 @@ def _reap_orphaned_browser_sessions(): # Ownership check: prefer owner_pid file (cross-process safe). owner_pid_file = os.path.join(socket_dir, f"{session_name}.owner_pid") + owner_pid: Optional[int] = None owner_alive: Optional[bool] = None # None = owner_pid missing/unreadable if os.path.isfile(owner_pid_file): try: @@ -1935,11 +1986,34 @@ def _reap_orphaned_browser_sessions(): from gateway.status import _pid_exists owner_alive = _pid_exists(owner_pid) except (ValueError, OSError): + owner_pid = None owner_alive = None # corrupt file — fall through if owner_alive is True: - # Owner is alive — this session belongs to a live hermes process. - continue + # Owner is alive. Normally that means the session belongs to a + # live hermes process and must not be touched — but "owner alive" + # alone made leaked daemons immortal: if the owner lost its + # in-memory tracking (any exception path between spawn and + # registration), nothing would ever reap the daemon, and the + # daemon-side AGENT_BROWSER_IDLE_TIMEOUT_MS does not fire when the + # daemon itself is wedged (e.g. Chrome's framework was swapped out + # from under it by an auto-update). + # + # So: still trust live tracking, but fall back to idle age. + if session_name in tracked_names: + continue + + idle_s = _socket_dir_idle_seconds(socket_dir) + if idle_s is None or idle_s < BROWSER_ORPHAN_GRACE_SECONDS: + # Unknown age, or still within the grace window — fail safe. + continue + + logger.warning( + "Browser session %s has a live owner (PID %s) but is untracked " + "and idle for %ds (grace %ds) — treating as leaked and reaping", + session_name, owner_pid, int(idle_s), + BROWSER_ORPHAN_GRACE_SECONDS) + # fall through to the reap path below if owner_alive is None: # No owner_pid file (legacy daemon). Fall back to in-process @@ -2003,15 +2077,26 @@ def _browser_cleanup_thread_worker(): Runs every 30 seconds and checks for sessions that haven't been used within the BROWSER_SESSION_INACTIVITY_TIMEOUT period. - On first run, also reaps orphaned sessions from previous process lifetimes. + + Also reaps orphaned daemons — on startup (sessions left by previous + process lifetimes) *and* every BROWSER_ORPHAN_REAP_INTERVAL seconds + thereafter. The periodic pass matters because a leak is not only a + across-restart phenomenon: a daemon can fall out of in-memory tracking + at any point in a long-lived process, and a startup-only reap can never + recover from that. """ - # One-time orphan reap on startup - try: - _reap_orphaned_browser_sessions() - except Exception as e: - logger.warning("Orphan reap error: %s", e) + reap_every_cycles = max(1, round(BROWSER_ORPHAN_REAP_INTERVAL / 30)) + cycle = 0 while _cleanup_running: + # cycle 0 is the startup reap; then every reap_every_cycles. + if cycle % reap_every_cycles == 0: + try: + _reap_orphaned_browser_sessions() + except Exception as e: + logger.warning("Orphan reap error: %s", e) + cycle += 1 + try: _cleanup_inactive_browser_sessions() except Exception as e: From 7de5a6590626ffa1b42d15bbc5cddc5a6c0c63e4 Mon Sep 17 00:00:00 2001 From: Adolanium <94890352+Adolanium@users.noreply.github.com> Date: Thu, 25 Jun 2026 20:31:05 +0300 Subject: [PATCH 273/376] fix(pets): delete row strips after extracting their frames Each hatch generates one row strip per state into cache/images and extracts the animation frames from it, but never removes the strip. The only cleanup for that directory runs in the gateway housekeeping loop, which a CLI, desktop, or cron hatch never starts, so the strips accumulate for good. Drop the strip after every attempt once its frames are decoded into memory, including failed or retried attempts, so a hatch no longer grows the image cache without bound. --- agent/pet/generate/orchestrate.py | 13 +++++ tests/agent/test_pet_generate.py | 91 +++++++++++++++++++++++++++++++ 2 files changed, 104 insertions(+) diff --git a/agent/pet/generate/orchestrate.py b/agent/pet/generate/orchestrate.py index 54a1adf5b078c..13f002ea8e511 100644 --- a/agent/pet/generate/orchestrate.py +++ b/agent/pet/generate/orchestrate.py @@ -246,6 +246,7 @@ def _gen_row(spec: tuple[str, int, int]) -> tuple[str, list | None]: if cancelled(): return state, None strict = attempt < _ROW_GEN_ATTEMPTS - 1 + strips: list[Path] = [] try: strips = imagegen.generate( prompts.build_row_prompt(state, count, label, style=style), @@ -274,6 +275,18 @@ def _gen_row(spec: tuple[str, int, int]) -> tuple[str, list | None]: "pet hatch %r: row %r attempt %d/%d failed: %s", slug, state, attempt + 1, _ROW_GEN_ATTEMPTS, exc, ) + finally: + # The strip is an intermediate. extract_strip_frames has already + # decoded its frames into memory, so drop the row image after + # every attempt (success or failure). Nothing prunes + # cache/images outside the gateway housekeeping loop, so a CLI + # or desktop hatch would otherwise leave each strip behind for + # good and grow the cache without bound. + for strip in strips: + try: + Path(strip).unlink(missing_ok=True) + except OSError: + pass logger.warning( "pet hatch %r: row %r gave up after %.1fs: %s", slug, state, time.monotonic() - t0, last_exc, diff --git a/tests/agent/test_pet_generate.py b/tests/agent/test_pet_generate.py index 2715dce46bc8f..6d78c209b98b0 100644 --- a/tests/agent/test_pet_generate.py +++ b/tests/agent/test_pet_generate.py @@ -8,6 +8,7 @@ from __future__ import annotations import os +from pathlib import Path import pytest @@ -267,6 +268,96 @@ def fake_generate(prompt, *, n=1, reference_images=None, provider=None, prefix=" assert store.load_pet("mocky").exists +def test_hatch_pet_removes_row_strips_after_extraction(monkeypatch, tmp_path): + """Row strips are intermediates. Once their frames are decoded, the strip + files are removed so the image cache does not grow on every hatch (nothing + prunes cache/images outside the gateway housekeeping loop).""" + from agent.pet.generate import atlas as atlas_mod + from agent.pet.generate import imagegen, orchestrate + + base = tmp_path / "base.png" + _strip(1).save(base) + + produced: list = [] + + def fake_generate(prompt, *, n=1, reference_images=None, provider=None, prefix="pet", aspect_ratio="square"): + state = prefix.replace("pet_row_", "") + count = atlas_mod.FRAME_COUNTS.get(state, 6) + p = tmp_path / f"{prefix}.png" + _strip(count).save(p) + produced.append(p) + return [p] + + monkeypatch.setattr(imagegen, "resolve_provider", lambda **_: object()) + monkeypatch.setattr(imagegen, "generate", fake_generate) + + orchestrate.hatch_pet(base_image=base, slug="cleanup", concept="a fox") + + assert produced, "expected row strips to be generated" + leftover = [p for p in produced if p.exists()] + assert leftover == [], f"row strips left in cache: {leftover}" + + +def test_hatch_pet_removes_row_strips_after_failed_attempt(monkeypatch, tmp_path): + from agent.pet.generate import atlas as atlas_mod + from agent.pet.generate import imagegen, orchestrate + + base = tmp_path / "base.png" + _strip(1).save(base) + + attempts: dict[str, int] = {} + + def fake_generate(prompt, *, n=1, reference_images=None, provider=None, prefix="pet", aspect_ratio="square"): + attempts[prefix] = attempts.get(prefix, 0) + 1 + state = prefix.replace("pet_row_", "") + count = atlas_mod.FRAME_COUNTS.get(state, 6) + path = tmp_path / f"{prefix}_{attempts[prefix]}.png" + _strip(count).save(path) + return [path] + + extract_strip_frames = atlas_mod.extract_strip_frames + failed_once = False + + def flaky_extract(strip, count, *args, **kwargs): + nonlocal failed_once + if Path(strip).name.startswith("pet_row_idle_") and not failed_once: + failed_once = True + raise ValueError("retry idle row") + return extract_strip_frames(strip, count, *args, **kwargs) + + monkeypatch.setattr(imagegen, "resolve_provider", lambda **_: object()) + monkeypatch.setattr(imagegen, "generate", fake_generate) + monkeypatch.setattr(atlas_mod, "extract_strip_frames", flaky_extract) + + orchestrate.hatch_pet(base_image=base, slug="retry-cleanup", concept="a fox") + + assert failed_once + assert attempts["pet_row_idle"] == 2 + assert not list(tmp_path.glob("pet_row_*")) + + +def test_hatch_pet_idle_fallback_when_row_fails(monkeypatch, tmp_path): + from agent.pet.generate import atlas as atlas_mod + from agent.pet.generate import imagegen, orchestrate + from agent.pet.generate.imagegen import GenerationError + + base = tmp_path / "base.png" + _strip(1).save(base) + + def fake_generate(prompt, *, n=1, reference_images=None, provider=None, prefix="pet", aspect_ratio="square"): + if prefix == "pet_row_idle": + raise GenerationError("boom") + state = prefix.replace("pet_row_", "") + count = atlas_mod.FRAME_COUNTS.get(state, 6) + p = tmp_path / f"{prefix}.png" + _strip(count).save(p) + return [p] + + monkeypatch.setattr(imagegen, "resolve_provider", lambda **_: object()) + monkeypatch.setattr(imagegen, "generate", fake_generate) + + result = orchestrate.hatch_pet(base_image=base, slug="fallbacky", concept="a fox") + assert "idle" in result.states # filled by the base-image fallback From b9672ea24e9b984b29d46848d23a946f4dd73e04 Mon Sep 17 00:00:00 2001 From: Adolanium <94890352+Adolanium@users.noreply.github.com> Date: Thu, 25 Jun 2026 20:33:27 +0300 Subject: [PATCH 274/376] fix(pets): remove the non-PNG base draft after hardening generate_base_drafts hardens every base draft to a transparent PNG with _harden_transparency. When the provider returned a non-PNG file (webp, jpg, or gif), the hardened PNG is saved under a new path and the original draft is left in cache/images. Nothing prunes that directory outside the gateway housekeeping loop, so a CLI or desktop draft round leaks one original per non-PNG draft. Remove the original after a successful hardening when the output path differs from the input. PNG inputs (including mixed-case suffixes like .PNG) are hardened in place so a case-insensitive filesystem cannot treat with_suffix(".png") as a different file and unlink the output. --- agent/pet/generate/orchestrate.py | 18 +++++++++++- tests/agent/test_pet_generate.py | 49 +++++++++++++++++++++++++++++++ 2 files changed, 66 insertions(+), 1 deletion(-) diff --git a/agent/pet/generate/orchestrate.py b/agent/pet/generate/orchestrate.py index 13f002ea8e511..1b267a4de2100 100644 --- a/agent/pet/generate/orchestrate.py +++ b/agent/pet/generate/orchestrate.py @@ -71,8 +71,24 @@ def _harden_transparency(path: Path) -> Path: # Zero the RGB of any leftover semi-transparent edge pixels so a keyed # draft has no colored halo when composited on the dark UI. keyed = atlas._clear_transparent_rgb(keyed) - out = path.with_suffix(".png") + # PNG inputs are hardened in place, including mixed-case suffixes like + # .PNG. with_suffix(".png") would name a different Path string that still + # resolves to the same file on case-insensitive filesystems (macOS APFS, + # Windows), and unlinking path after save would delete the hardened output. + if path.suffix.lower() == ".png": + out = path + else: + out = path.with_suffix(".png") keyed.save(out, format="PNG") + if out != path: + # The hardened PNG stands in for the draft. When the provider handed + # back a non-PNG file (webp, jpg, gif), out is a different path, so + # remove the original instead of leaving it behind in cache/images + # (nothing prunes that directory outside the gateway loop). + try: + path.unlink(missing_ok=True) + except OSError: + pass return out except Exception as exc: # noqa: BLE001 - cosmetic; fall back to the raw image logger.debug("base draft transparency hardening failed for %s: %s", path, exc) diff --git a/tests/agent/test_pet_generate.py b/tests/agent/test_pet_generate.py index 6d78c209b98b0..84d303281c0b4 100644 --- a/tests/agent/test_pet_generate.py +++ b/tests/agent/test_pet_generate.py @@ -231,6 +231,55 @@ def fake_generate(prompt, *, n=1, reference_images=None, provider=None, prefix=" assert rgba.getpixel((rgba.width // 2, rgba.height // 2))[3] > 0 +def test_harden_transparency_removes_non_png_original(tmp_path): + """A non-PNG base draft is replaced by a hardened PNG, and the original + draft file is not left behind in the image cache.""" + from agent.pet.generate import orchestrate + + src = tmp_path / "pet_base_sample.webp" + _strip(1).save(src, format="WEBP") + + out = orchestrate._harden_transparency(src) + + assert out.suffix == ".png" + assert out.exists() + assert not src.exists() + + +def test_harden_transparency_keeps_png_input_in_place(tmp_path): + """A PNG base draft is hardened in place, so the returned path is the input + path and there is no separate original to remove.""" + from agent.pet.generate import orchestrate + + src = tmp_path / "pet_base_sample.png" + _strip(1).save(src, format="PNG") + + out = orchestrate._harden_transparency(src) + + assert out == src + assert out.exists() + + +def test_harden_transparency_keeps_mixed_case_png_in_place(tmp_path): + """A PNG path with a mixed-case suffix is hardened in place. + + path.with_suffix('.png') yields a different Path string than 'pet.PNG', but + on case-insensitive filesystems (macOS APFS, Windows) both resolve to the + same file. Unlinking the input after save would delete the hardened output. + """ + from agent.pet.generate import orchestrate + + src = tmp_path / "pet_base_sample.PNG" + _strip(1).save(src, format="PNG") + + out = orchestrate._harden_transparency(src) + + assert out == src + assert out.suffix == ".PNG" + assert out.exists() + assert src.exists() + + def test_hatch_pet_end_to_end(monkeypatch, tmp_path): from agent.pet import store from agent.pet.generate import atlas as atlas_mod From aba493427430d41c73dbd46045a5a9dd4e8c92b3 Mon Sep 17 00:00:00 2001 From: rainbowgits <164521089+rainbowgits@users.noreply.github.com> Date: Wed, 5 Aug 2026 05:45:54 +0300 Subject: [PATCH 275/376] fix(agent): omit unsupported metadata on Relay scope.pop Older nemo-relay bindings reject metadata= on scope.pop, which aborted turn finalization and left scopes open. Filter kwargs to what the live binding accepts so close paths can complete. --- agent/relay_llm.py | 3 +- agent/relay_runtime.py | 43 ++++++- .../observability/relay_shared_metrics.py | 3 +- tests/agent/test_relay_scope_pop_metadata.py | 121 ++++++++++++++++++ 4 files changed, 165 insertions(+), 5 deletions(-) create mode 100644 tests/agent/test_relay_scope_pop_metadata.py diff --git a/agent/relay_llm.py b/agent/relay_llm.py index 58a5c1bd5e7d4..7758a2bdcfa71 100644 --- a/agent/relay_llm.py +++ b/agent/relay_llm.py @@ -897,7 +897,8 @@ def _complete_logical( output["response_model"] = response_model_name lease.host.run_in_session( lease.session, - lease.host.relay.scope.pop, + relay_runtime.pop_relay_scope, + lease.host.relay, handle, output=output, metadata={ diff --git a/agent/relay_runtime.py b/agent/relay_runtime.py index f4f15b552dfea..003213d87d259 100644 --- a/agent/relay_runtime.py +++ b/agent/relay_runtime.py @@ -94,6 +94,40 @@ def _target() -> None: return result[0] if result else None +def pop_relay_scope( + relay: Any, + handle: Any, + *, + output: Any = None, + metadata: Any = None, + timestamp: Any = None, +) -> Any: + """Pop a Relay scope without passing kwargs the binding rejects. + + NeMo Relay ``scope.pop`` gained ``metadata`` in 0.4+. Older wheels (e.g. + 0.3.x) raise ``TypeError: pop() got an unexpected keyword argument + 'metadata'`` when Hermes finalization forwards runtime metadata. Filter to + parameters the live binding accepts so turn/session close can complete. + """ + pop = relay.scope.pop + kwargs: dict[str, Any] = {} + if output is not None: + kwargs["output"] = output + if metadata is not None: + kwargs["metadata"] = metadata + if timestamp is not None: + kwargs["timestamp"] = timestamp + try: + params = inspect.signature(pop).parameters + except (TypeError, ValueError): + params = {} + if params and not any( + param.kind == inspect.Parameter.VAR_KEYWORD for param in params.values() + ): + kwargs = {key: value for key, value in kwargs.items() if key in params} + return pop(handle, **kwargs) + + @dataclass class RelaySession: """One isolated Relay scope stack owned by a Hermes session.""" @@ -609,7 +643,8 @@ def same_handle(a: Any, b: Any) -> bool: return a_uuid is not None and a_uuid == b_uuid try: - self.relay.scope.pop( + pop_relay_scope( + self.relay, handle, output=close_output, metadata=metadata, @@ -630,7 +665,8 @@ def same_handle(a: Any, b: Any) -> bool: ): break try: - self.relay.scope.pop( + pop_relay_scope( + self.relay, top, output={ "outcome": "cancelled", @@ -654,7 +690,8 @@ def same_handle(a: Any, b: Any) -> bool: handle, ) try: - self.relay.scope.pop( + pop_relay_scope( + self.relay, handle, output=close_output, metadata=metadata, diff --git a/hermes_cli/observability/relay_shared_metrics.py b/hermes_cli/observability/relay_shared_metrics.py index 756c6caa88362..2ab88f51c3ece 100644 --- a/hermes_cli/observability/relay_shared_metrics.py +++ b/hermes_cli/observability/relay_shared_metrics.py @@ -1026,7 +1026,8 @@ def _finish_task( try: self._run_in_task( task, - self.relay.scope.pop, + relay_runtime.pop_relay_scope, + self.relay, task.handle, output=fields, metadata=self._event_metadata(), diff --git a/tests/agent/test_relay_scope_pop_metadata.py b/tests/agent/test_relay_scope_pop_metadata.py new file mode 100644 index 0000000000000..b307d7eee7c28 --- /dev/null +++ b/tests/agent/test_relay_scope_pop_metadata.py @@ -0,0 +1,121 @@ +"""Regression for #78993: scope.pop metadata kwarg on older NeMo Relay.""" + +from __future__ import annotations + +import inspect +import logging +import tempfile +from types import SimpleNamespace + +import pytest + +from agent import relay_runtime + + +def test_pop_relay_scope_omits_unsupported_metadata_kwarg(): + calls: list[tuple[object, dict]] = [] + + def pop_without_metadata(handle, *, output=None): + calls.append((handle, {"output": output})) + + relay = SimpleNamespace(scope=SimpleNamespace(pop=pop_without_metadata)) + handle = ("scope", "hermes.turn", 1) + + relay_runtime.pop_relay_scope( + relay, + handle, + output={"outcome": "success"}, + metadata={"hermes.relay.schema_version": "hermes.relay.runtime.v1"}, + ) + + assert calls == [(handle, {"output": {"outcome": "success"}})] + + +def test_pop_relay_scope_forwards_metadata_when_supported(): + calls: list[tuple[object, dict]] = [] + + def pop_with_metadata(handle, *, output=None, metadata=None, timestamp=None): + calls.append( + ( + handle, + { + "output": output, + "metadata": metadata, + "timestamp": timestamp, + }, + ) + ) + + relay = SimpleNamespace(scope=SimpleNamespace(pop=pop_with_metadata)) + handle = ("scope", "hermes.turn", 2) + metadata = {"hermes.relay.runtime_instance": "abc"} + + relay_runtime.pop_relay_scope( + relay, + handle, + output={"outcome": "error"}, + metadata=metadata, + ) + + assert calls == [ + ( + handle, + { + "output": {"outcome": "error"}, + "metadata": metadata, + "timestamp": None, + }, + ) + ] + + +def test_end_turn_finalization_survives_pop_without_metadata(monkeypatch, caplog): + """Mirror #78993: nemo-relay 0.3.x rejects metadata= on scope.pop.""" + pytest.importorskip("nemo_relay") + + monkeypatch.setenv("HERMES_HOME", tempfile.mkdtemp()) + relay_runtime._reset_for_tests() + lease = relay_runtime.SESSION_COORDINATOR.acquire_conversation( + profile_key=relay_runtime.current_profile_key(), + session_id="session-78993", + platform="cli", + ) + turn = relay_runtime.SESSION_COORDINATOR.begin_turn( + lease, + turn_id="turn-1", + task_id="task-1", + ) + lease.host.retain_managed_execution("test.relay_scope_pop") + + original_pop = lease.host.relay.scope.pop + assert "metadata" in inspect.signature(original_pop).parameters + + def pop_without_metadata(handle, *, output=None, timestamp=None): + return original_pop(handle, output=output, timestamp=timestamp) + + monkeypatch.setattr(lease.host.relay.scope, "pop", pop_without_metadata) + + logical = lease.host.run_in_session( + lease.session, + lease.host.relay.scope.push, + "logical-llm", + lease.host.relay.ScopeType.Custom, + handle=turn.handle, + input={}, + metadata={"hermes.test": True}, + ) + turn.logical_llm_calls["api-1"] = logical + + with caplog.at_level(logging.WARNING, logger="agent.relay_runtime"): + relay_runtime.SESSION_COORDINATOR.end_turn(turn, outcome="success") + + joined = "\n".join(record.getMessage() for record in caplog.records) + assert "unexpected keyword argument 'metadata'" not in joined + assert "turn finalization failed" not in joined + assert "logical LLM finalization failed" not in joined + assert turn.logical_llm_calls == {} + assert turn.closed is True + + lease.host.release_managed_execution("test.relay_scope_pop") + relay_runtime.SESSION_COORDINATOR.release_conversation(lease) + relay_runtime._reset_for_tests() From 94ce8396e84ccdaa96d2a85d1fc2ced4bc2e0c61 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 23:08:09 -0700 Subject: [PATCH 276/376] fix(sessions): release active-session leases against their acquisition registry MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A gateway active-session lease is acquired against the root HERMES_HOME, but release_active_session()/transfer_active_session() re-resolved the registry path from the *current* HERMES_HOME. Under native multiplex a routed turn runs agent cleanup inside _profile_runtime_scope, so the release looked under the named profile while the root entry stayed alive — after max_concurrent_sessions routed turns every new session was rejected with 'Hermes is at the active session limit' (#85431). Pin state/lock paths on the lease at acquisition time and prefer them on release and transfer. Fixes #85431. --- hermes_cli/active_sessions.py | 22 +++++-- tests/hermes_cli/test_active_sessions.py | 73 ++++++++++++++++++++++++ 2 files changed, 91 insertions(+), 4 deletions(-) diff --git a/hermes_cli/active_sessions.py b/hermes_cli/active_sessions.py index a572c74093294..13aa1e41b32f8 100644 --- a/hermes_cli/active_sessions.py +++ b/hermes_cli/active_sessions.py @@ -261,6 +261,14 @@ class ActiveSessionLease: surface: str enabled: bool = True released: bool = False + # Registry paths pinned at acquisition time. A lease acquired under the + # root ``HERMES_HOME`` must release against the same registry even when + # ``release()`` runs inside a profile home override (native multiplex + # routes turns under ``_profile_runtime_scope``), otherwise the root + # entry survives until process exit and the session cap fills with + # phantom leases (#85431). + state_path: Optional[Path] = None + lock_path: Optional[Path] = None def release(self) -> None: if self.released or not self.enabled: @@ -331,13 +339,18 @@ def try_acquire_active_session( lease_id=lease_id, session_id=str(session_id), surface=str(surface), + state_path=state_path, + lock_path=_lock_path(), ), None def release_active_session(lease: ActiveSessionLease) -> None: - state_path = _state_path() + # Prefer the registry the lease was acquired against: the caller may be + # running under a profile HERMES_HOME override (#85431). + state_path = lease.state_path or _state_path() + lock_path = lease.lock_path or _lock_path() try: - with _FileLock(_lock_path()): + with _FileLock(lock_path): entries = _prune_dead(_read_entries(state_path)) kept = [ entry @@ -366,8 +379,9 @@ def transfer_active_session( lease.session_id = new_session_id return True - state_path = _state_path() - with _FileLock(_lock_path()): + state_path = lease.state_path or _state_path() + lock_path = lease.lock_path or _lock_path() + with _FileLock(lock_path): entries = _prune_dead(_read_entries(state_path)) updated = False for entry in entries: diff --git a/tests/hermes_cli/test_active_sessions.py b/tests/hermes_cli/test_active_sessions.py index 2d4dd949eae39..dcbc36af55927 100644 --- a/tests/hermes_cli/test_active_sessions.py +++ b/tests/hermes_cli/test_active_sessions.py @@ -167,3 +167,76 @@ def test_release_orphaned_leases_reclaims_only_unowned_own_pid_entries(tmp_path, for entry in active_sessions.active_session_registry_snapshot() ) == ["kept", "other"] assert orphan is not None + + +def test_release_under_profile_home_override_targets_acquisition_registry( + tmp_path, monkeypatch +): + """Regression for #85431: a lease acquired against the root HERMES_HOME + must release from the root registry even when ``release()`` runs inside a + profile home override (native multiplex runs agent cleanup under + ``_profile_runtime_scope``). Before the fix the root entry survived and + the session cap filled with phantom leases.""" + from hermes_constants import ( + reset_hermes_home_override, + set_hermes_home_override, + ) + + root = tmp_path / "hermes" + profile = root / "profiles" / "worker" + profile.mkdir(parents=True) + monkeypatch.setenv("HERMES_HOME", str(root)) + + lease, error = active_sessions.try_acquire_active_session( + session_id="agent:worker:telegram:dm:synthetic", + surface="gateway:telegram", + config={"max_concurrent_sessions": 2}, + ) + assert lease is not None and error is None + root_registry = root / "runtime" / "active_sessions.json" + assert root_registry.exists() + + token = set_hermes_home_override(str(profile)) + try: + lease.release() + finally: + reset_hermes_home_override(token) + + assert lease.released is True + remaining = active_sessions._read_entries(root_registry) + assert remaining == [] + # No phantom registry created under the profile home. + assert not (profile / "runtime" / "active_sessions.json").exists() + + +def test_transfer_under_profile_home_override_targets_acquisition_registry( + tmp_path, monkeypatch +): + """Sibling site of #85431: transfer must also update the registry the + lease was acquired against, not one resolved from the current override.""" + from hermes_constants import ( + reset_hermes_home_override, + set_hermes_home_override, + ) + + root = tmp_path / "hermes" + profile = root / "profiles" / "worker" + profile.mkdir(parents=True) + monkeypatch.setenv("HERMES_HOME", str(root)) + + lease, error = active_sessions.try_acquire_active_session( + session_id="before", + surface="gateway:telegram", + config={"max_concurrent_sessions": 2}, + ) + assert lease is not None and error is None + + token = set_hermes_home_override(str(profile)) + try: + assert active_sessions.transfer_active_session(lease, session_id="after") + finally: + reset_hermes_home_override(token) + + root_registry = root / "runtime" / "active_sessions.json" + entries = active_sessions._read_entries(root_registry) + assert [entry["session_id"] for entry in entries] == ["after"] From 2fecf392a28d4d91d77cd34739072d70b74a823b Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 23:08:50 -0700 Subject: [PATCH 277/376] chore: map contributor email for Hangzian --- contributors/emails/andy@andydeMac-mini-2.local | 1 + 1 file changed, 1 insertion(+) create mode 100644 contributors/emails/andy@andydeMac-mini-2.local diff --git a/contributors/emails/andy@andydeMac-mini-2.local b/contributors/emails/andy@andydeMac-mini-2.local new file mode 100644 index 0000000000000..2abfa5e2f3e71 --- /dev/null +++ b/contributors/emails/andy@andydeMac-mini-2.local @@ -0,0 +1 @@ +Hangzian From 0d07fe63f9c2d4b1728601deb9627fffd4f21ceb Mon Sep 17 00:00:00 2001 From: Tranquil-Flow <66773372+Tranquil-Flow@users.noreply.github.com> Date: Sat, 27 Jun 2026 02:10:18 +0200 Subject: [PATCH 278/376] fix(projects): use _branch_lane_id for non-git folders to prevent duplicate lanes (#53329) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit _place_by_heuristic used the raw path as the lane key for non-git project folders, while the desktop overlay independently computed ::branch::main for the same session (since git_branch was null). The ID mismatch caused duplicate lanes — one from the backend with the folder name, one from the overlay labeled 'main'. Use _branch_lane_id(path, DEFAULT_BRANCH_LABEL) so the backend's lane key matches the overlay's expected ::branch::main scheme, eliminating the duplicate lane. --- tests/tui_gateway/test_project_tree.py | 32 ++++++++++++++++++++++++++ tui_gateway/project_tree.py | 2 +- 2 files changed, 33 insertions(+), 1 deletion(-) diff --git a/tests/tui_gateway/test_project_tree.py b/tests/tui_gateway/test_project_tree.py index 0faeac7f1e6e9..69fc2b0176d52 100644 --- a/tests/tui_gateway/test_project_tree.py +++ b/tests/tui_gateway/test_project_tree.py @@ -561,3 +561,35 @@ def test_colliding_repo_basenames_disambiguate_labels(): labels = sorted(p["label"] for p in tree["projects"]) assert labels == ["x/proj", "y/proj"] + + +def test_non_git_folder_uses_branch_lane_id(): + """#53329: _place_by_heuristic must use _branch_lane_id for non-git folders. + + Before the fix, non-git folders got a lane key equal to the raw path, + while the desktop overlay expected ::branch::main. This caused duplicate + lanes (one from backend, one from overlay). + """ + result = pt._place_by_heuristic("/home/user/my-project") + assert result is not None + assert result["lane_key"] == pt._branch_lane_id( + "/home/user/my-project", pt.DEFAULT_BRANCH_LABEL + ), ( + f"Expected lane_key to use _branch_lane_id scheme but got " + f"{result['lane_key']!r}" + ) + # The label should still be the folder basename + assert result["lane_label"] == "my-project" + # Must be marked as main lane + assert result["is_main"] is True + + +def test_non_git_folder_lane_matches_overlay_scheme(): + """#53329: verify the lane key format matches what the overlay expects.""" + result = pt._place_by_heuristic("/data/work/folder-x") + assert result is not None + # Overlay expects: ::branch::main + expected = "/data/work/folder-x::branch::main" + assert result["lane_key"] == expected, ( + f"Expected lane_key={expected!r} but got {result['lane_key']!r}" + ) diff --git a/tui_gateway/project_tree.py b/tui_gateway/project_tree.py index cd5c966a5578f..5a92b38c97df6 100644 --- a/tui_gateway/project_tree.py +++ b/tui_gateway/project_tree.py @@ -221,7 +221,7 @@ def _place_by_heuristic(path: str) -> Optional[dict]: repo_path = _with_base_name(path, m.group(1)) return _placement(repo_path, path, m.group(2), path, False, False) - return _placement(path, path, base, path, True, False) + return _placement(path, _branch_lane_id(path, DEFAULT_BRANCH_LABEL), base, path, True, False) def _place(cwd: str, branch: str, resolve: Optional[Resolve], persisted_root: str) -> Optional[dict]: From f7de2ca4169fa48f024c5f4ddee7edb5fe44d0ee Mon Sep 17 00:00:00 2001 From: briandevans <252620095+briandevans@users.noreply.github.com> Date: Sun, 12 Jul 2026 20:40:50 -0700 Subject: [PATCH 279/376] fix(desktop): order project-tree lanes by recency in the overview, not alphabetically MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit _build_repos emptied each lane's sessions array for the overview (hydrate=False) payload BEFORE _sort_lanes ran. _lane_sort_key derives a lane's activity from max(_session_time(s) for s in group["sessions"]), so with the rows already gone every non-trunk lane scored activity=0.0 and the sort key (is_trunk, is_kanban, -activity, label) collapsed to alphabetical-by-label. The documented intent — branches and linked worktrees sort by most-recent activity, then label — was silently defeated on the projects.tree RPC that feeds the desktop sidebar overview, while the drill-in path (hydrate=True) kept the rows and sorted correctly. Any repo with two or more non-trunk lanes showed a different order in the overview than when opened. Move the session-clearing to after _sort_lanes/_disambiguate_labels so the sort reads real recency. Lane counts are still captured before clearing, so sessionCount and the slim overview payload are unchanged — only the order is fixed, and the overview now matches the drill-in. --- tests/tui_gateway/test_project_tree.py | 36 ++++++++++++++++++++++++++ tui_gateway/project_tree.py | 11 ++++++-- 2 files changed, 45 insertions(+), 2 deletions(-) diff --git a/tests/tui_gateway/test_project_tree.py b/tests/tui_gateway/test_project_tree.py index 69fc2b0176d52..067316740b0e3 100644 --- a/tests/tui_gateway/test_project_tree.py +++ b/tests/tui_gateway/test_project_tree.py @@ -125,6 +125,42 @@ def test_linked_worktrees_fold_under_their_common_repo_root(): assert linked["path"] == "/elsewhere/wt" +def test_overview_orders_lanes_by_recency_not_alphabetically(): + # Two linked-worktree lanes under one common repo root whose ALPHABETICAL + # order (wt-aaa, wt-zzz) is the OPPOSITE of their activity order (wt-zzz is + # the more recently active). The overview (hydrate=False) empties lane + # session arrays for payload slimness — but the lane sort must still run on + # real recency, matching the drill-in (hydrate=True) order, not collapse to + # alphabetical because the rows were dropped before sorting. + resolve = _resolver( + { + "/repo": ("/repo", "/repo"), + "/wt-aaa": ("/repo", "/wt-aaa"), + "/wt-zzz": ("/repo", "/wt-zzz"), + } + ) + sessions = [ + _session("/repo", branch="main", last_active=5000), + _session("/wt-aaa", last_active=1000), # alphabetically first, older + _session("/wt-zzz", last_active=9000), # alphabetically last, newer + ] + + def _non_trunk_labels(hydrate): + tree = pt.build_tree([], sessions, [], resolve, hydrate=hydrate) + project = tree["projects"][0] + return [ + g["label"] + for repo in project["repos"] + for g in repo["groups"] + if not g["isMain"] + ] + + # Overview path: recency order (newer first), NOT alphabetical. + assert _non_trunk_labels(hydrate=False) == ["wt-zzz", "wt-aaa"] + # Drill-in path already sorts by recency — the two paths must agree. + assert _non_trunk_labels(hydrate=True) == ["wt-zzz", "wt-aaa"] + + def test_kanban_task_worktrees_collapse_into_one_bucket(): resolve = _resolver( { diff --git a/tui_gateway/project_tree.py b/tui_gateway/project_tree.py index 5a92b38c97df6..4f2d1eaaae107 100644 --- a/tui_gateway/project_tree.py +++ b/tui_gateway/project_tree.py @@ -376,8 +376,6 @@ def _build_repos(sessions: list[dict], resolve: Optional[Resolve], hydrate: bool group = entry["group"] group["sessions"].sort(key=_session_time, reverse=True) count = len(group["sessions"]) - if not hydrate: - group["sessions"] = [] repo_identity = _path_key(entry["repo_key"]) repo = repos.get(repo_identity) @@ -397,6 +395,15 @@ def _build_repos(sessions: list[dict], resolve: Optional[Resolve], hydrate: bool for repo in repo_list: repo["groups"] = _sort_lanes(repo["groups"]) _disambiguate_labels(repo["groups"]) + # Drop per-lane session rows only AFTER sorting: _lane_sort_key ranks + # non-trunk lanes by most-recent activity, which it derives from the + # session rows. Clearing them earlier makes every lane look inactive on + # the overview (hydrate=False) path and collapses the sort to + # alphabetical. Counts were already captured above, so the payload stays + # slim without losing the recency order. + if not hydrate: + for group in repo["groups"]: + group["sessions"] = [] _disambiguate_labels(repo_list) return repo_list From 0fd223d1bd4e9322c11ad57c6bbe4dccd92fbf5a Mon Sep 17 00:00:00 2001 From: rsk-731 Date: Mon, 20 Jul 2026 11:21:14 +0000 Subject: [PATCH 280/376] fix(desktop): stop dual main lanes for non-git project workspaces Live overlay always placed sessions under `::branch::main`, while the backend non-git heuristic keys the lane by folder path (label = basename). Overlay missed that lane by id/label and forked a phantom `main` group with the same sessions. Match existing path-keyed isMain lanes before creating a branch-style main lane. Covers the codex-research-guardian-style drill-in duplicate. --- .../sidebar/projects/workspace-groups.test.ts | 71 +++++++++++++++++++ .../chat/sidebar/projects/workspace-groups.ts | 15 ++++ 2 files changed, 86 insertions(+) diff --git a/apps/desktop/src/app/chat/sidebar/projects/workspace-groups.test.ts b/apps/desktop/src/app/chat/sidebar/projects/workspace-groups.test.ts index b2a893af583e2..6548d4a3b6e73 100644 --- a/apps/desktop/src/app/chat/sidebar/projects/workspace-groups.test.ts +++ b/apps/desktop/src/app/chat/sidebar/projects/workspace-groups.test.ts @@ -697,6 +697,77 @@ describe('overlayLiveLanes', () => { expect(overlaid.repos[0].groups.flatMap(g => g.sessions.map(s => s.id))).toEqual(['dup']) }) + it('does not fork a phantom main lane for a non-git backend workspace lane', () => { + // Backend non-git heuristic (`project_tree._place_by_heuristic`): lane id = + // folder path, label = basename, isMain=true. Live overlay used to always + // place under `::branch::main` / label "main", miss that lane by id+label, + // and CREATE a second main lane with the same sessions — dual lanes in the + // project drill-in (e.g. main + codex-research-guardian). + const root = '/home/hermes/hermes-workspace/codex-research-guardian' + const a = makeSession(root, { id: 's1' }) // empty git_branch / git_repo_root + const b = makeSession(root, { id: 's2' }) + + const project = projectNode({ + id: root, + isAuto: true, + path: root, + repos: [ + { + id: root, + label: 'codex-research-guardian', + path: root, + sessionCount: 2, + groups: [ + lane({ + id: root, + label: 'codex-research-guardian', + isMain: true, + path: root, + sessions: [a, b] + }) + ] + } + ] + }) + + const overlaid = overlayLiveLanes(project, [a, b]) + const groups = overlaid.repos[0].groups + + expect(groups).toHaveLength(1) + expect(groups[0].id).toBe(root) + expect(groups[0].label).toBe('codex-research-guardian') + expect(groups[0].sessions.map(s => s.id).sort()).toEqual(['s1', 's2']) + expect(groups.some(g => g.label === 'main' || g.id.endsWith('::branch::main'))).toBe(false) + }) + + it('joins a fresh live session into an existing non-git workspace lane (no branch id)', () => { + const root = '/work/notes' + const existing = makeSession(root, { id: 'old' }) + + const project = projectNode({ + id: root, + isAuto: true, + path: root, + repos: [ + { + id: root, + label: 'notes', + path: root, + sessionCount: 1, + groups: [lane({ id: root, label: 'notes', isMain: true, path: root, sessions: [existing] })] + } + ] + }) + + const fresh = makeSession(root, { id: 'fresh' }) + const overlaid = overlayLiveLanes(project, [existing, fresh]) + const groups = overlaid.repos[0].groups + + expect(groups).toHaveLength(1) + expect(groups[0].id).toBe(root) + expect(groups[0].sessions.map(s => s.id).sort()).toEqual(['fresh', 'old']) + }) + it('adds a new session to an existing worktree lane keyed by a divergent id (matches by path)', () => { // Backend keyed the worktree lane off a branch-style id (no live git probe), // but the lane PATH is the worktree dir. A new session under that worktree diff --git a/apps/desktop/src/app/chat/sidebar/projects/workspace-groups.ts b/apps/desktop/src/app/chat/sidebar/projects/workspace-groups.ts index 353e9d4ad7000..756600d495cee 100644 --- a/apps/desktop/src/app/chat/sidebar/projects/workspace-groups.ts +++ b/apps/desktop/src/app/chat/sidebar/projects/workspace-groups.ts @@ -563,6 +563,21 @@ export function overlayRepoLanes( (placed.isMain ? lanes.find(g => g.isMain && g.label.toLowerCase() === placed.label.toLowerCase()) : undefined) ?? + // Non-git backend heuristic (`project_tree._place_by_heuristic`): one + // isMain lane keyed by the folder path itself (id === path, label = + // basename) — not `::branch::`. Live placement always emits + // `::branch::main` / label "main", so id+label miss and used to FORK a + // phantom second main lane with the same sessions. Prefer the existing + // path-keyed main lane when present. + (placed.isMain && placedKey + ? lanes.find( + g => + g.isMain && + pathKey(g.path) === placedKey && + !g.id.includes('::branch::') && + !g.id.includes('::kanban') + ) + : undefined) ?? (!placed.isMain && placedKey ? lanes.find(g => pathKey(g.path) === placedKey) : undefined) if (!lane) { From 3defb25a428d2563d2a67b2b65d8ed9c9cd1ce71 Mon Sep 17 00:00:00 2001 From: liuhao1024 Date: Tue, 30 Jun 2026 23:40:42 +0800 Subject: [PATCH 281/376] fix(desktop): evict stale lane entry when overlay moves session to worktree When a session's cwd moves from the main checkout to a newly created worktree, overlayRepoLanes places it into the matching worktree lane but never removed the stale entry from the main lane. The session appeared under both groups until the user left and re-entered the project view. Add a cross-lane eviction loop that removes the session from all other lanes before inserting it into the target lane. --- .../sidebar/projects/workspace-groups.test.ts | 33 +++++++++++++++++++ .../chat/sidebar/projects/workspace-groups.ts | 16 +++++++++ 2 files changed, 49 insertions(+) diff --git a/apps/desktop/src/app/chat/sidebar/projects/workspace-groups.test.ts b/apps/desktop/src/app/chat/sidebar/projects/workspace-groups.test.ts index 6548d4a3b6e73..931a4dad5c66d 100644 --- a/apps/desktop/src/app/chat/sidebar/projects/workspace-groups.test.ts +++ b/apps/desktop/src/app/chat/sidebar/projects/workspace-groups.test.ts @@ -901,6 +901,39 @@ describe('overlayLiveLanes', () => { expect(overlayLiveLanes(home, [makeSession('/www/app', { id: 'fresh' })])).toBe(home) }) + + it('evicts a session from the main lane when the live overlay places it into a worktree lane', () => { + // Session was in main when the backend tree was captured, but the live + // $sessions cache now has it under a worktree cwd. The overlay must place + // it ONLY in the worktree lane — not both. + const session = makeSession('/www/app/.worktrees/feature', { id: 'moved', git_branch: 'feature' }) + + const project = projectNode({ + id: '/www/app', + repos: [ + { + id: '/www/app', + label: 'app', + path: '/www/app', + sessionCount: 1, + groups: [ + lane({ id: '/www/app::branch::main', label: 'main', isMain: true, path: '/www/app', sessions: [session] }), + lane({ id: '/www/app/.worktrees/feature', label: 'feature', path: '/www/app/.worktrees/feature', sessions: [] }) + ] + } + ] + }) + + const overlaid = overlayLiveLanes(project, [session]) + const mainLane = overlaid.repos[0].groups.find(g => g.isMain) + const featureLane = overlaid.repos[0].groups.find(g => g.path === '/www/app/.worktrees/feature') + + // Session must NOT appear in the main lane + expect(mainLane?.sessions ?? []).toHaveLength(0) + // Session must appear only in the worktree lane + expect(featureLane?.sessions.map(s => s.id)).toEqual(['moved']) + expect(overlaid.sessionCount).toBe(1) + }) }) describe('overlayLivePreviews', () => { diff --git a/apps/desktop/src/app/chat/sidebar/projects/workspace-groups.ts b/apps/desktop/src/app/chat/sidebar/projects/workspace-groups.ts index 756600d495cee..52370653baab3 100644 --- a/apps/desktop/src/app/chat/sidebar/projects/workspace-groups.ts +++ b/apps/desktop/src/app/chat/sidebar/projects/workspace-groups.ts @@ -586,6 +586,22 @@ export function overlayRepoLanes( } } + // Evict the session from any OTHER lane the backend snapshot may have + // placed it in (e.g. a turn that moved the session's cwd from main to a + // new worktree — the overlay places it into the worktree lane, but without + // this eviction the stale main-lane entry persists and the session appears + // under both groups until the next backend tree refresh). + for (const g of lanes) { + if (g !== lane) { + const idx = g.sessions.findIndex(s => s.id === session.id) + + if (idx >= 0) { + g.sessions = [...g.sessions.slice(0, idx), ...g.sessions.slice(idx + 1)] + changed = true + } + } + } + lane.sessions = upsertSession(lane.sessions, session) changed = true } From d40dda7db6c5a2e244719b989024d71e27306e46 Mon Sep 17 00:00:00 2001 From: Mustafa Date: Sun, 12 Jul 2026 11:51:09 +0100 Subject: [PATCH 282/376] fix(desktop): preserve lane session recency order --- .../sidebar/projects/workspace-groups.test.ts | 44 +++++++++++++++++++ .../chat/sidebar/projects/workspace-groups.ts | 2 +- 2 files changed, 45 insertions(+), 1 deletion(-) diff --git a/apps/desktop/src/app/chat/sidebar/projects/workspace-groups.test.ts b/apps/desktop/src/app/chat/sidebar/projects/workspace-groups.test.ts index 931a4dad5c66d..ccea21f80a8a2 100644 --- a/apps/desktop/src/app/chat/sidebar/projects/workspace-groups.test.ts +++ b/apps/desktop/src/app/chat/sidebar/projects/workspace-groups.test.ts @@ -768,6 +768,50 @@ describe('overlayLiveLanes', () => { expect(groups[0].sessions.map(s => s.id).sort()).toEqual(['fresh', 'old']) }) + it('preserves backend recency order when live sessions overlay a lane', () => { + const recentlyActive = makeSession('/www/app', { + id: 'recently-active', + git_branch: 'main', + started_at: 1, + last_active: 3 + }) + + const newlyCreated = makeSession('/www/app', { + id: 'newly-created', + git_branch: 'main', + started_at: 2, + last_active: 2 + }) + + const project = projectNode({ + id: '/www/app', + repos: [ + { + id: '/www/app', + label: 'app', + path: '/www/app', + sessionCount: 2, + groups: [ + lane({ + id: '/www/app::branch::main', + label: 'main', + isMain: true, + path: '/www/app', + sessions: [recentlyActive, newlyCreated] + }) + ] + } + ] + }) + + const overlaid = overlayLiveLanes(project, [recentlyActive, newlyCreated]) + + expect(overlaid.repos[0].groups[0].sessions.map(session => session.id)).toEqual([ + 'recently-active', + 'newly-created' + ]) + }) + it('adds a new session to an existing worktree lane keyed by a divergent id (matches by path)', () => { // Backend keyed the worktree lane off a branch-style id (no live git probe), // but the lane PATH is the worktree dir. A new session under that worktree diff --git a/apps/desktop/src/app/chat/sidebar/projects/workspace-groups.ts b/apps/desktop/src/app/chat/sidebar/projects/workspace-groups.ts index 52370653baab3..605ce38eef7bd 100644 --- a/apps/desktop/src/app/chat/sidebar/projects/workspace-groups.ts +++ b/apps/desktop/src/app/chat/sidebar/projects/workspace-groups.ts @@ -449,7 +449,7 @@ export function sessionProjectColor(session: SessionInfo, projects: ProjectInfo[ } const upsertSession = (rows: SessionInfo[], session: SessionInfo): SessionInfo[] => - [session, ...rows.filter(row => row.id !== session.id)].sort((a, b) => b.started_at - a.started_at) + [session, ...rows.filter(row => row.id !== session.id)].sort((a, b) => sessionRecency(b) - sessionRecency(a)) /** * The lane a live session belongs to WITHIN a known repo root, by path — the From db1d8983e50003899ea64253ed6985068cb2f6d6 Mon Sep 17 00:00:00 2001 From: Inspired-atm <1265291278@qq.com> Date: Mon, 10 Aug 2026 16:09:27 +0800 Subject: [PATCH 283/376] fix(desktop): no-op branch switch on non-repo project lanes Non-repo explicit projects (plain folders) get a main-checkout lane whose label is the folder basename, not a branch. Clicking "+" (new session) on such a lane calls switchBranchInRepo -> switchBranch, which sanitizes the basename to "" and throws "Branch name is required.", aborting the session creation. Short-circuit switchBranch for roots that are not git work trees so callers proceed with a plain session. Fixes #83028 --- .../desktop/electron/git-worktree-ops.test.ts | 35 +++++++++++++++++++ apps/desktop/electron/git-worktree-ops.ts | 19 ++++++++++ 2 files changed, 54 insertions(+) diff --git a/apps/desktop/electron/git-worktree-ops.test.ts b/apps/desktop/electron/git-worktree-ops.test.ts index 43af70e83dc3d..d3cf6a19e9a0a 100644 --- a/apps/desktop/electron/git-worktree-ops.test.ts +++ b/apps/desktop/electron/git-worktree-ops.test.ts @@ -435,3 +435,38 @@ test('addWorktree: a remote default branch gets its own worktree, not a home swi fs.rmSync(cloneDir, { recursive: true, force: true }) } }) + +test('switchBranch: non-repo dir short-circuits instead of throwing', async () => { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'hermes-sw-')) + + try { + // A plain folder pinned as a project (no .git): its lane label is the + // folder basename, not a branch — switching must no-op, not error, so + // callers like "+" new session can proceed with a plain session. + const result = await switchBranch(dir, '国创大赛', 'git') + + assert.deepEqual(result, { branch: null }) + } finally { + fs.rmSync(dir, { recursive: true, force: true }) + } +}) + +test('switchBranch: repo dir still validates the branch name and switches', async () => { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'hermes-sw-')) + + try { + execFileSync('git', ['init', '-b', 'main'], { cwd: dir }) + execFileSync('git', ['config', 'user.email', 't@example.com'], { cwd: dir }) + execFileSync('git', ['config', 'user.name', 'test'], { cwd: dir }) + execFileSync('git', ['commit', '--allow-empty', '-m', 'root'], { cwd: dir }) + + // Existing behaviour preserved: an illegal branch name still errors. + await assert.rejects(() => switchBranch(dir, '///', 'git'), /Branch name is required/) + + // And switching to a real branch still works. + const result = await switchBranch(dir, 'main', 'git') + assert.deepEqual(result, { branch: 'main' }) + } finally { + fs.rmSync(dir, { recursive: true, force: true }) + } +}) diff --git a/apps/desktop/electron/git-worktree-ops.ts b/apps/desktop/electron/git-worktree-ops.ts index 2ff43862c32f7..384e8c3fbe393 100644 --- a/apps/desktop/electron/git-worktree-ops.ts +++ b/apps/desktop/electron/git-worktree-ops.ts @@ -437,6 +437,25 @@ async function listBranches(repoPath, gitBin) { async function switchBranch(repoPath, branch, gitBin) { const resolved = resolveRequestedPathForIpc(repoPath, { purpose: 'Branch switch' }) + + // Sidebar lanes exist for plain folders too (non-repo explicit projects), + // and their lane label is the folder basename — not a branch. `git switch` + // there is meaningless, and sanitizing that label would throw a misleading + // "Branch name is required." — so short-circuit for non-repo roots and let + // callers (e.g. "+" new session on the project lane) proceed with a plain + // session instead of aborting. + let inside = 'false' + + try { + inside = (await runGit(gitBin, ['rev-parse', '--is-inside-work-tree'], resolved)).trim() + } catch { + // Not a git repo (or git unavailable): fall through to the short-circuit. + } + + if (inside !== 'true') { + return { branch: null } + } + const target = sanitizeBranch(branch) if (!target) { From e89532d97eb2c969a97766af0342e17f04631fbe Mon Sep 17 00:00:00 2001 From: embwl0x Date: Sun, 2 Aug 2026 03:51:34 -0600 Subject: [PATCH 284/376] fix(desktop): order async session git metadata --- hermes_state.py | 103 ++++++- hermes_state_common.py | 3 +- .../test_session_git_metadata_generation.py | 257 ++++++++++++++++++ .../test_session_git_metadata_generation.py | 72 +++++ tui_gateway/server.py | 94 ++++--- 5 files changed, 476 insertions(+), 53 deletions(-) create mode 100644 tests/state/test_session_git_metadata_generation.py create mode 100644 tests/tui_gateway/test_session_git_metadata_generation.py diff --git a/hermes_state.py b/hermes_state.py index 3a4459f275f3e..325bd485c0ea7 100644 --- a/hermes_state.py +++ b/hermes_state.py @@ -5318,11 +5318,11 @@ def update_session_cwd( self, session_id: str, cwd: str, - git_branch: str = None, - git_repo_root: str = None, + git_branch: Optional[str] = None, + git_repo_root: Optional[str] = None, replace_git_meta: bool = False, - ) -> None: - """Persist the session working directory when a frontend knows it. + ) -> Optional[int]: + """Persist the authoritative cwd and claim a Git metadata generation. ``git_branch`` records the git branch checked out in ``cwd`` at the time the session started/resumed. The sidebar groups main-checkout sessions @@ -5340,27 +5340,100 @@ def update_session_cwd( MOVE (re-homing a session into another project) must overwrite the old repo identity even when the new cwd resolves to none — keeping the stale root would leave the session grouped under the project it just left. + + Every call increments ``git_metadata_generation`` in the same write + transaction. Async Git probes must publish through + :meth:`publish_session_git_metadata` with the returned generation, so + an older worker cannot overwrite a newer cwd claim even after an + A -> B -> A transition or from another process sharing this database. + Metadata from a different cwd is cleared atomically with the move. """ if not session_id or not cwd: - return + return None + + branch = (git_branch or "").strip() + repo_root = (git_repo_root or "").strip() + + def _do(conn): + current = conn.execute( + "SELECT cwd FROM sessions WHERE id = ?", (session_id,) + ).fetchone() + if current is None: + return None + + current_cwd = current["cwd"] if isinstance(current, sqlite3.Row) else current[0] + sets = [ + "cwd = ?", + "git_metadata_generation = COALESCE(git_metadata_generation, 0) + 1", + ] + params: List[Any] = [cwd] + if current_cwd != cwd or replace_git_meta: + sets.extend(("git_branch = ?", "git_repo_root = ?")) + params.extend((branch or None, repo_root or None)) + elif branch: + sets.append("git_branch = ?") + params.append(branch) + if repo_root and current_cwd == cwd and not replace_git_meta: + sets.append("git_repo_root = ?") + params.append(repo_root) + params.append(session_id) + conn.execute( + f"UPDATE sessions SET {', '.join(sets)} WHERE id = ?", params + ) + row = conn.execute( + "SELECT git_metadata_generation FROM sessions WHERE id = ?", + (session_id,), + ).fetchone() + if row is None: + return None + value = row["git_metadata_generation"] if isinstance(row, sqlite3.Row) else row[0] + return int(value) + + return self._execute_write(_do) + + def publish_session_git_metadata( + self, + session_id: str, + cwd: str, + generation: int, + git_branch: Optional[str] = None, + git_repo_root: Optional[str] = None, + ) -> bool: + """Publish async Git enrichment only while its cwd claim is current.""" + if ( + not session_id + or not cwd + or isinstance(generation, bool) + or not isinstance(generation, int) + or generation < 1 + ): + return False branch = (git_branch or "").strip() repo_root = (git_repo_root or "").strip() + if not branch and not repo_root: + return False - sets = ["cwd = ?"] - params: List[Any] = [cwd] - if branch or replace_git_meta: + sets: List[str] = [] + params: List[Any] = [] + if branch: sets.append("git_branch = ?") - params.append(branch or None) - if repo_root or replace_git_meta: + params.append(branch) + if repo_root: sets.append("git_repo_root = ?") - params.append(repo_root or None) - params.append(session_id) + params.append(repo_root) + params.extend((session_id, cwd, generation)) def _do(conn): - conn.execute(f"UPDATE sessions SET {', '.join(sets)} WHERE id = ?", params) + cursor = conn.execute( + f"UPDATE sessions SET {', '.join(sets)} " + "WHERE id = ? AND cwd = ? " + "AND git_metadata_generation = ?", + params, + ) + return cursor.rowcount == 1 - self._execute_write(_do) + return bool(self._execute_write(_do)) def backfill_repo_roots(self, cwd_to_root: Dict[str, str]) -> None: """Persist resolved git repo roots for cwds that don't have one yet. @@ -7872,7 +7945,7 @@ def get_compression_tip(self, session_id: str) -> Optional[str]: # declarative reconciliation are included automatically instead of # silently dropping out of list rows. _SESSION_COMPACT_EXCLUDED = frozenset( - {"system_prompt", "system_prompt_hash"} + {"system_prompt", "system_prompt_hash", "git_metadata_generation"} ) _session_compact_cols_sql: Optional[str] = None diff --git a/hermes_state_common.py b/hermes_state_common.py index 5a9e4893e6c75..7186c8205db1e 100644 --- a/hermes_state_common.py +++ b/hermes_state_common.py @@ -216,7 +216,7 @@ def _sql_session_last_active_by_id(session_id_expr: str) -> str: ) -SCHEMA_VERSION = 25 +SCHEMA_VERSION = 26 # FTS storage-layout version, tracked INDEPENDENTLY of SCHEMA_VERSION in the @@ -285,6 +285,7 @@ def _sql_session_last_active_by_id(session_id_expr: str) -> str: cwd TEXT, git_branch TEXT, git_repo_root TEXT, + git_metadata_generation INTEGER NOT NULL DEFAULT 0, billing_provider TEXT, billing_base_url TEXT, billing_mode TEXT, diff --git a/tests/state/test_session_git_metadata_generation.py b/tests/state/test_session_git_metadata_generation.py new file mode 100644 index 0000000000000..2e1cb309889c4 --- /dev/null +++ b/tests/state/test_session_git_metadata_generation.py @@ -0,0 +1,257 @@ +"""Cross-process ordering for asynchronous session Git metadata probes.""" + +from __future__ import annotations + +import sqlite3 +import threading + +from hermes_state import SCHEMA_VERSION, SessionDB + + +def _open_pair(tmp_path): + path = tmp_path / "state.db" + first = SessionDB(db_path=path) + second = SessionDB(db_path=path) + first.create_session("session", "desktop", cwd="/repo/A") + return first, second + + +def _require_generation(value: int | None) -> int: + assert isinstance(value, int) and not isinstance(value, bool) + return value + + +def test_delayed_probe_cannot_overwrite_newer_a_b_a_claim(tmp_path): + first, second = _open_pair(tmp_path) + release_old = threading.Event() + old_finished = threading.Event() + old_result = [] + try: + old_generation = _require_generation( + first.update_session_cwd("session", "/repo/A") + ) + + def publish_old_probe(): + assert release_old.wait(5) + old_result.append( + first.publish_session_git_metadata( + "session", + "/repo/A", + old_generation, + "stale-branch", + "/repo/stale-root", + ) + ) + old_finished.set() + + worker = threading.Thread(target=publish_old_probe) + worker.start() + + second.update_session_cwd("session", "/repo/B") + new_generation = _require_generation( + second.update_session_cwd("session", "/repo/A") + ) + assert new_generation > old_generation + assert second.publish_session_git_metadata( + "session", + "/repo/A", + new_generation, + "new-branch", + "/repo/new-root", + ) + + release_old.set() + assert old_finished.wait(5) + worker.join(timeout=5) + assert not worker.is_alive() + assert old_result == [False] + + row = second.get_session("session") + assert row is not None + assert row["cwd"] == "/repo/A" + assert row["git_branch"] == "new-branch" + assert row["git_repo_root"] == "/repo/new-root" + finally: + release_old.set() + first.close() + second.close() + + +def test_repeated_same_cwd_claim_invalidates_older_probe(tmp_path): + first, second = _open_pair(tmp_path) + try: + old_generation = _require_generation( + first.update_session_cwd("session", "/repo/A") + ) + new_generation = _require_generation( + second.update_session_cwd("session", "/repo/A") + ) + + assert new_generation > old_generation + assert second.publish_session_git_metadata( + "session", "/repo/A", new_generation, "new", "/repo/A" + ) + assert not first.publish_session_git_metadata( + "session", "/repo/A", old_generation, "old", "/repo/old" + ) + row = second.get_session("session") + assert row is not None + assert row["git_branch"] == "new" + finally: + first.close() + second.close() + + +def test_cwd_move_clears_metadata_in_same_claim(tmp_path): + db = SessionDB(db_path=tmp_path / "state.db") + try: + db.create_session("session", "desktop", cwd="/repo/A") + generation = _require_generation( + db.update_session_cwd("session", "/repo/A") + ) + assert db.publish_session_git_metadata( + "session", "/repo/A", generation, "main", "/repo/A" + ) + + moved_generation = _require_generation( + db.update_session_cwd("session", "/repo/B") + ) + row = db.get_session("session") + assert row is not None + assert moved_generation > generation + assert row["cwd"] == "/repo/B" + assert row["git_branch"] is None + assert row["git_repo_root"] is None + finally: + db.close() + + +def test_explicit_move_replaces_metadata_and_claims_generation(tmp_path): + db = SessionDB(db_path=tmp_path / "state.db") + try: + db.create_session("session", "desktop", cwd="/repo/A") + initial_generation = _require_generation( + db.update_session_cwd( + "session", + "/repo/A", + git_branch="main", + git_repo_root="/repo/A", + ) + ) + + moved_generation = _require_generation( + db.update_session_cwd( + "session", + "/outside-git", + replace_git_meta=True, + ) + ) + + row = db.get_session("session") + assert row is not None + assert moved_generation > initial_generation + assert row["cwd"] == "/outside-git" + assert row["git_branch"] is None + assert row["git_repo_root"] is None + finally: + db.close() + + +def test_failed_new_probe_still_invalidates_older_worker(tmp_path): + first, second = _open_pair(tmp_path) + try: + baseline = _require_generation( + first.update_session_cwd("session", "/repo/A") + ) + assert first.publish_session_git_metadata( + "session", "/repo/A", baseline, "baseline", "/repo/A" + ) + old_generation = _require_generation( + first.update_session_cwd("session", "/repo/A") + ) + second.update_session_cwd("session", "/repo/A") + + assert not first.publish_session_git_metadata( + "session", "/repo/A", old_generation, "stale", "/repo/stale" + ) + row = second.get_session("session") + assert row is not None + assert row["git_branch"] == "baseline" + assert row["git_repo_root"] == "/repo/A" + finally: + first.close() + second.close() + + +def test_generation_authority_is_scoped_to_each_profile_database(tmp_path): + first = SessionDB(db_path=tmp_path / "profile-a.db") + second = SessionDB(db_path=tmp_path / "profile-b.db") + try: + first.create_session("same-id", "desktop", cwd="/a") + second.create_session("same-id", "desktop", cwd="/b") + first_generation = _require_generation( + first.update_session_cwd("same-id", "/a") + ) + second_generation = _require_generation( + second.update_session_cwd("same-id", "/b") + ) + + assert first.publish_session_git_metadata( + "same-id", "/a", first_generation, "a", "/a" + ) + assert second.publish_session_git_metadata( + "same-id", "/b", second_generation, "b", "/b" + ) + first_row = first.get_session("same-id") + second_row = second.get_session("same-id") + assert first_row is not None + assert second_row is not None + assert first_row["git_branch"] == "a" + assert second_row["git_branch"] == "b" + finally: + first.close() + second.close() + + +def test_legacy_sessions_table_reconciles_generation_column(tmp_path): + path = tmp_path / "state.db" + SessionDB(db_path=path).close() + conn = sqlite3.connect(path) + try: + conn.execute("ALTER TABLE sessions DROP COLUMN git_metadata_generation") + conn.execute("UPDATE schema_version SET version = 25") + conn.commit() + finally: + conn.close() + + reopened = SessionDB(db_path=path) + try: + verify = sqlite3.connect(path) + try: + columns = { + row[1] + for row in verify.execute("PRAGMA table_info('sessions')") + } + finally: + verify.close() + assert "git_metadata_generation" in columns + assert reopened._conn.execute( + "SELECT version FROM schema_version" + ).fetchone()[0] == SCHEMA_VERSION == 26 + reopened.create_session("session", "desktop", cwd="/repo") + assert reopened.update_session_cwd("session", "/repo") == 1 + finally: + reopened.close() + + +def test_compact_session_rows_do_not_expose_internal_generation(tmp_path): + db = SessionDB(db_path=tmp_path / "state.db") + try: + db.create_session("session", "desktop", cwd="/repo") + db.update_session_cwd("session", "/repo") + + rows = db.list_sessions_rich(compact_rows=True) + assert len(rows) == 1 + assert "git_metadata_generation" not in rows[0] + finally: + db.close() diff --git a/tests/tui_gateway/test_session_git_metadata_generation.py b/tests/tui_gateway/test_session_git_metadata_generation.py new file mode 100644 index 0000000000000..89dca8120130f --- /dev/null +++ b/tests/tui_gateway/test_session_git_metadata_generation.py @@ -0,0 +1,72 @@ +"""Gateway wiring for generation-scoped Git metadata publication.""" + +from __future__ import annotations + +import tui_gateway.server as server + + +class _ImmediateThread: + def __init__(self, *, target, **_kwargs): + self._target = target + + def start(self): + self._target() + + +def test_cwd_claim_precedes_probe_and_generation_reaches_publish(monkeypatch): + events = [] + + class DB: + def update_session_cwd(self, session_id, cwd): + events.append(("claim", session_id, cwd)) + return 17 + + def publish_session_git_metadata( + self, session_id, cwd, generation, branch, root + ): + events.append( + ("publish", session_id, cwd, generation, branch, root) + ) + return True + + monkeypatch.setattr(server, "_get_db", lambda: DB()) + monkeypatch.setattr(server.threading, "Thread", _ImmediateThread) + monkeypatch.setattr( + server, + "_git_branch_for_cwd", + lambda cwd: events.append(("probe", cwd)) or "feature", + ) + monkeypatch.setattr(server, "_git_common_repo_root_for_cwd", lambda _cwd: "/repo") + + generation = server._persist_session_cwd_and_schedule_git_meta( + {"session_key": "session"}, "/repo/worktree" + ) + + assert generation == 17 + assert events == [ + ("claim", "session", "/repo/worktree"), + ("probe", "/repo/worktree"), + ( + "publish", + "session", + "/repo/worktree", + 17, + "feature", + "/repo", + ), + ] + + +def test_missing_db_claim_never_starts_git_probe(monkeypatch): + probed = [] + monkeypatch.setattr(server, "_get_db", lambda: None) + monkeypatch.setattr( + server, "_git_branch_for_cwd", lambda cwd: probed.append(cwd) + ) + + generation = server._persist_session_cwd_and_schedule_git_meta( + {"session_key": "session"}, "/repo" + ) + + assert generation is None + assert probed == [] diff --git a/tui_gateway/server.py b/tui_gateway/server.py index 2375c91fccf35..9638e8879492b 100644 --- a/tui_gateway/server.py +++ b/tui_gateway/server.py @@ -2723,13 +2723,7 @@ def _display_session_cwd(session: dict | None) -> str: healed = _heal_dead_cwd(cwd) if healed and healed != cwd and session is not None: session["cwd"] = healed - try: - with _session_db(session) as db: - if db is not None: - db.update_session_cwd(session.get("session_key", ""), healed) - except Exception: - logger.debug("failed to persist healed session cwd", exc_info=True) - _persist_session_git_meta(session, healed) + _persist_session_cwd_and_schedule_git_meta(session, healed) return healed @@ -2812,14 +2806,7 @@ def _reconcile_session_cwd_from_terminal(session: dict | None) -> bool: session["cwd_from_settle"] = True _register_session_cwd(session) - with _session_db(session) as db: - if db is not None: - try: - db.update_session_cwd(session.get("session_key", ""), resolved) - except Exception: - logger.debug("failed to persist settled session cwd", exc_info=True) - - _persist_session_git_meta(session, resolved) + _persist_session_cwd_and_schedule_git_meta(session, resolved) return True @@ -3092,7 +3079,7 @@ def _session_db(session: dict): db.close() -def _persist_session_git_meta(session: dict, cwd: str) -> None: +def _persist_session_git_meta(session: dict, cwd: str, generation: int) -> None: """Resolve + persist a session's git branch / repo root WITHOUT blocking. Branch and root come from ``git`` subprocess probes; running them inline on @@ -3106,7 +3093,13 @@ def _persist_session_git_meta(session: dict, cwd: str) -> None: probe never delays gateway shutdown. """ session_key = session.get("session_key", "") - if not session_key or not cwd: + if ( + not session_key + or not cwd + or isinstance(generation, bool) + or not isinstance(generation, int) + or generation < 1 + ): return # Snapshot the routing fields now; the live session dict may be gone by the # time the thread runs. `_session_db` reopens the profile-correct db inside. @@ -3120,13 +3113,52 @@ def _run() -> None: return with _session_db(db_session) as db: if db is not None: - db.update_session_cwd(session_key, cwd, branch, root) + db.publish_session_git_metadata( + session_key, + cwd, + generation, + branch, + root, + ) except Exception: logger.debug("failed to persist session git metadata", exc_info=True) threading.Thread(target=_run, name="git-meta", daemon=True).start() +def _persist_session_cwd_and_schedule_git_meta( + session: dict, + cwd: str, + *, + db=None, +) -> int | None: + """Claim a DB-backed probe generation, then start Git enrichment.""" + try: + if db is not None: + generation = db.update_session_cwd( + session.get("session_key", ""), cwd + ) + else: + with _session_db(session) as owner_db: + if owner_db is None: + return None + generation = owner_db.update_session_cwd( + session.get("session_key", ""), cwd + ) + except Exception: + logger.debug("failed to persist session cwd", exc_info=True) + return None + + if ( + isinstance(generation, bool) + or not isinstance(generation, int) + or generation < 1 + ): + return None + _persist_session_git_meta(session, cwd, generation) + return generation + + def _set_session_cwd(session: dict, cwd: str) -> str: from hermes_constants import translate_cwd_for_wsl_backend @@ -3142,14 +3174,9 @@ def _set_session_cwd(session: dict, cwd: str) -> str: # the terminal wandering must not move the workspace again. session["cwd_from_settle"] = False _register_session_cwd(session) - with _session_db(session) as db: - if db is not None: - try: - db.update_session_cwd(session.get("session_key", ""), resolved) - except Exception: - logger.debug("failed to persist session cwd", exc_info=True) - # Branch/repo-root probes are git subprocesses — capture them off the hot path. - _persist_session_git_meta(session, resolved) + # The synchronous DB write claims ordering authority; Git subprocesses stay + # off the hot path and may publish only for that exact generation. + _persist_session_cwd_and_schedule_git_meta(session, resolved) try: from tools.terminal_tool import cleanup_vm @@ -6137,14 +6164,7 @@ def _apply_project_workspace(task_id: str, path: str, _name: str = "") -> None: session["cwd_from_settle"] = False _register_session_cwd(session) - with _session_db(session) as db: - if db is not None: - try: - db.update_session_cwd(session.get("session_key", ""), resolved) - except Exception: - logger.debug("failed to persist project workspace cwd", exc_info=True) - - _persist_session_git_meta(session, resolved) + _persist_session_cwd_and_schedule_git_meta(session, resolved) try: agent = session.get("agent") @@ -6949,9 +6969,9 @@ def _init_session( try: _cwd = _sessions[sid]["cwd"] if hasattr(db, "update_session_cwd"): - db.update_session_cwd(key, _cwd) - # git branch/root probes run off the hot path (see _set_session_cwd). - _persist_session_git_meta(_sessions[sid], _cwd) + _persist_session_cwd_and_schedule_git_meta( + _sessions[sid], _cwd, db=db + ) except Exception: logger.debug( "failed to persist resumed session cwd", exc_info=True From f378a8fb3b71c5e8375ffa19a3d757c429a9452c Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 23:08:34 -0700 Subject: [PATCH 285/376] fix(projects): dedup project_create by primary_path (#75820) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Creating a project whose resolved primary path already belongs to a non-archived project now raises a clear ValueError naming the existing project (create_project) — duplicated projects each seeded an identical copy of the repo subtree, multiplying the duplicate-lane bug per copy. The agent-facing project_create tool is idempotent instead: it re-activates the existing project rather than erroring. allow_duplicate_path=True keeps deliberate duplicates possible. Also updates the legacy non-git lane-id expectation to the branch-style id introduced for #53329. --- hermes_cli/projects_db.py | 41 ++++++++++++++++++++++++ tests/hermes_cli/test_projects_db.py | 43 ++++++++++++++++++++++++++ tests/tui_gateway/test_project_tree.py | 4 ++- tools/project_tools.py | 14 +++++++-- 4 files changed, 98 insertions(+), 4 deletions(-) diff --git a/hermes_cli/projects_db.py b/hermes_cli/projects_db.py index 53bead2227aab..12eb206ddaf1d 100644 --- a/hermes_cli/projects_db.py +++ b/hermes_cli/projects_db.py @@ -319,6 +319,32 @@ def _unique_slug(conn: sqlite3.Connection, candidate: str) -> str: return slug +def _primary_path_key(path: str) -> str: + """Comparison key for primary-path dedup (absolute + case/sep-normalized).""" + return os.path.normcase(_normalize_path(path)) + + +def find_by_primary_path( + conn: sqlite3.Connection, path: str, *, include_archived: bool = False +) -> Optional[Project]: + """The first (oldest) project whose primary path matches ``path``, else None. + + Comparison is separator/case normalized so equivalent Windows spellings of + the same folder do not slip past the dedup check. + """ + key = _primary_path_key(path) + if not key: + return None + for proj in list_projects(conn, include_archived=include_archived): + primary = proj.primary_path or next( + (f.path for f in proj.folders if f.is_primary), + proj.folders[0].path if proj.folders else None, + ) + if primary and _primary_path_key(primary) == key: + return proj + return None + + def create_project( conn: sqlite3.Connection, *, @@ -330,12 +356,19 @@ def create_project( icon: Optional[str] = None, color: Optional[str] = None, board_slug: Optional[str] = None, + allow_duplicate_path: bool = False, ) -> str: """Create a project and return its id. ``folders`` are normalized to absolute paths. If ``primary_path`` is given it is added to the folder set (if not already present) and marked primary; otherwise the first folder becomes primary. + + Duplicate projects pointing at the same folder multiply the sidebar's + per-project repo subtrees (every duplicate renders its own copy of the same + lanes), so a create whose resolved primary path already belongs to a + non-archived project raises ``ValueError`` naming the existing project — + pass ``allow_duplicate_path=True`` to bypass deliberately. """ name = str(name or "").strip() if not name: @@ -357,6 +390,14 @@ def create_project( if primary is None and folder_paths: primary = folder_paths[0] + if primary and not allow_duplicate_path: + existing = find_by_primary_path(conn, primary) + if existing is not None: + raise ValueError( + f"folder already belongs to project '{existing.slug}' ({existing.id}); " + "switch to it instead of creating a duplicate" + ) + with write_txn(conn): unique = _unique_slug(conn, slug_candidate) conn.execute( diff --git a/tests/hermes_cli/test_projects_db.py b/tests/hermes_cli/test_projects_db.py index fc0347eee982b..81a8bb1efdaac 100644 --- a/tests/hermes_cli/test_projects_db.py +++ b/tests/hermes_cli/test_projects_db.py @@ -78,6 +78,49 @@ def test_project_for_path_skips_archived(conn): assert pdb.project_for_path(conn, "/www/app/src").id == pid +def test_create_dedups_by_primary_path(conn): + pid = pdb.create_project(conn, name="GeoTrace", folders=["/www/geotrace"]) + + # Same folder again (any name): refused, existing project named in error. + with pytest.raises(ValueError, match="already belongs to project 'geotrace'"): + pdb.create_project(conn, name="GeoTrace", folders=["/www/geotrace"]) + with pytest.raises(ValueError, match="already belongs"): + pdb.create_project(conn, name="Other Name", primary_path="/www/geotrace") + + # Trailing-separator spelling of the same folder is still a duplicate. + with pytest.raises(ValueError, match="already belongs"): + pdb.create_project(conn, name="GeoTrace", primary_path="/www/geotrace/") + + # Deliberate duplicates stay possible. + dup = pdb.create_project( + conn, name="GeoTrace", folders=["/www/geotrace"], allow_duplicate_path=True + ) + assert dup != pid + assert len(pdb.list_projects(conn)) == 2 + + +def test_create_dedup_ignores_archived_and_other_paths(conn): + pid = pdb.create_project(conn, name="App", folders=["/www/app"]) + pdb.archive_project(conn, pid) + + # Archived project no longer blocks the path. + fresh = pdb.create_project(conn, name="App", folders=["/www/app"]) + assert fresh != pid + + # Different folder is never a collision; folder-less projects don't match. + pdb.create_project(conn, name="Elsewhere", folders=["/www/other"]) + pdb.create_project(conn, name="No Folder") + + +def test_find_by_primary_path(conn): + pid = pdb.create_project(conn, name="App", folders=["/www/app"]) + + assert pdb.find_by_primary_path(conn, "/www/app").id == pid + assert pdb.find_by_primary_path(conn, "/www/app/").id == pid + assert pdb.find_by_primary_path(conn, "/www/nope") is None + assert pdb.find_by_primary_path(conn, "") is None + + diff --git a/tests/tui_gateway/test_project_tree.py b/tests/tui_gateway/test_project_tree.py index 067316740b0e3..bf2c753764707 100644 --- a/tests/tui_gateway/test_project_tree.py +++ b/tests/tui_gateway/test_project_tree.py @@ -295,7 +295,9 @@ def test_non_git_cwd_preserves_legacy_workspace_grouping(): assert project["isAuto"] is True assert project["label"] == "notes" assert project["sessionCount"] == 1 - assert _lane_ids(project) == ["/work/notes"] + # Branch-style lane id (#53329): keying this lane by the raw path used to + # fork a duplicate lane against the live overlay's `::branch::main` id. + assert _lane_ids(project) == ["/work/notes::branch::main"] assert tree["scoped_session_ids"] == [legacy["id"]] diff --git a/tools/project_tools.py b/tools/project_tools.py index 2b52e3144d613..3b4bc70a0fc55 100644 --- a/tools/project_tools.py +++ b/tools/project_tools.py @@ -101,9 +101,17 @@ def project_create(name: str, path: Optional[str] = None, task_id: Optional[str] try: with pdb.connect_closing() as conn: - pid = pdb.create_project(conn, name=name, folders=[folder] if folder else [], primary_path=folder or None) - pdb.set_active(conn, pid) - proj = pdb.get_project(conn, pid) + existing = pdb.find_by_primary_path(conn, folder) if folder else None + if existing is not None: + # Idempotent create: the folder already belongs to a project. + # Re-activating it beats minting a duplicate — duplicated + # projects render N identical sidebar subtrees (#75820). + pdb.set_active(conn, existing.id) + proj = existing + else: + pid = pdb.create_project(conn, name=name, folders=[folder] if folder else [], primary_path=folder or None) + pdb.set_active(conn, pid) + proj = pdb.get_project(conn, pid) except ValueError as exc: return json.dumps({"success": False, "error": str(exc)}) From cab6eb78f9fbddd913084103897b0e960138972c Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 23:09:35 -0700 Subject: [PATCH 286/376] test(projects): widen lane-id derivation regression coverage Cover the kanban ::kanban id, the -wt- suffix raw-path lane, and Windows separator/trailing-slash spellings collapsing to one lane key. --- tests/tui_gateway/test_project_tree.py | 26 ++++++++++++++++++++++++++ 1 file changed, 26 insertions(+) diff --git a/tests/tui_gateway/test_project_tree.py b/tests/tui_gateway/test_project_tree.py index bf2c753764707..feaa00ad1b6f1 100644 --- a/tests/tui_gateway/test_project_tree.py +++ b/tests/tui_gateway/test_project_tree.py @@ -631,3 +631,29 @@ def test_non_git_folder_lane_matches_overlay_scheme(): assert result["lane_key"] == expected, ( f"Expected lane_key={expected!r} but got {result['lane_key']!r}" ) + + +def test_heuristic_lane_ids_for_kanban_and_wt_suffix_are_unchanged(): + """The branch-style id applies ONLY to the plain-folder fallback. + + Kanban worktrees keep the ::kanban id and `-wt-` folders keep + the raw-path lane key so existing worktree lanes don't fork. + """ + kanban = pt._place_by_heuristic("/www/app/.worktrees/t_1a2b3c") + assert kanban is not None + assert kanban["lane_key"] == pt._kanban_lane_id("/www/app") + assert kanban["is_kanban"] is True + + wt = pt._place_by_heuristic("/www/app-wt-feature") + assert wt is not None + assert wt["lane_key"] == "/www/app-wt-feature" + assert wt["lane_label"] == "feature" + assert wt["is_main"] is False + + +def test_equivalent_windows_spellings_derive_one_lane_key(): + """Lane identity must collapse separator/trailing-slash variants (#62165).""" + a = pt._place_by_heuristic("C:/work/notes") + b = pt._place_by_heuristic("C:\\work\\notes\\") + assert a is not None and b is not None + assert pt._lane_key(a["lane_key"]) == pt._lane_key(b["lane_key"]) From e0e4d3ab9d145d9c36370137fd6147fa0c892b77 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 23:09:48 -0700 Subject: [PATCH 287/376] chore(contributors): map salvage-branch contributor emails --- contributors/emails/1265291278@qq.com | 1 + contributors/emails/rsk-731@users.noreply.github.com | 1 + 2 files changed, 2 insertions(+) create mode 100644 contributors/emails/1265291278@qq.com create mode 100644 contributors/emails/rsk-731@users.noreply.github.com diff --git a/contributors/emails/1265291278@qq.com b/contributors/emails/1265291278@qq.com new file mode 100644 index 0000000000000..e441c813ef219 --- /dev/null +++ b/contributors/emails/1265291278@qq.com @@ -0,0 +1 @@ +Inspired-by-Atmosphere diff --git a/contributors/emails/rsk-731@users.noreply.github.com b/contributors/emails/rsk-731@users.noreply.github.com new file mode 100644 index 0000000000000..6b6fd6c7a4d39 --- /dev/null +++ b/contributors/emails/rsk-731@users.noreply.github.com @@ -0,0 +1 @@ +rsk-731 From d16326bb2506e415c1752e0d43beaae01edfe3a2 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 23:44:01 -0700 Subject: [PATCH 288/376] test: align lost-and-found schema pins with git_metadata_generation column The salvaged #76716 adds git_metadata_generation to sessions (54 -> 55 columns). Update the synthetic-rebuild test's pinned widths and row builders to the new current layout. --- tests/hermes_cli/test_session_recovery_lost_and_found.py | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/tests/hermes_cli/test_session_recovery_lost_and_found.py b/tests/hermes_cli/test_session_recovery_lost_and_found.py index 0307eb28ae6eb..5d7bf99abfc1a 100644 --- a/tests/hermes_cli/test_session_recovery_lost_and_found.py +++ b/tests/hermes_cli/test_session_recovery_lost_and_found.py @@ -407,7 +407,7 @@ def session_row(session_id: str, ncols: int) -> list: # Junk that must NOT be classified into canonical tables. insert(3, 300, ["random", "noise", 42]) - insert(54, 301, ["not-a-session-id", "cli"] + [None] * 52) + insert(55, 301, ["not-a-session-id", "cli"] + [None] * 53) insert(23, 302, [None, "sess-x", "not-a-role", "junk"]) finally: conn.close() @@ -428,7 +428,7 @@ def test_classify_lost_and_found_row_sentinels() -> None: ) assert ( classify_lost_and_found_row( - 54, ("20260101_010101_aaa001", "cli") + (None,) * 52 + 55, ("20260101_010101_aaa001", "cli") + (None,) * 53 ) == "sessions" ) @@ -453,7 +453,7 @@ def test_classify_lost_and_found_row_sentinels() -> None: # Junk shapes. assert classify_lost_and_found_row(3, ("random", "noise", 42)) is None assert ( - classify_lost_and_found_row(54, ("not-a-session-id", "cli") + (None,) * 52) + classify_lost_and_found_row(55, ("not-a-session-id", "cli") + (None,) * 53) is None ) assert ( From f080bc3db15914d72a822634afd9818f6f0a539d Mon Sep 17 00:00:00 2001 From: LeonSGP43 Date: Thu, 14 May 2026 18:35:56 +0800 Subject: [PATCH 289/376] fix(tui): steady scrollbar after transcript shrink (cherry picked from commit 40b0e93b12622679a34571fba04e13a27eec8e55) --- ui-tui/src/__tests__/viewportStore.test.ts | 18 ++++++++++++++++++ ui-tui/src/lib/viewportStore.ts | 18 +++++++++++++++--- 2 files changed, 33 insertions(+), 3 deletions(-) diff --git a/ui-tui/src/__tests__/viewportStore.test.ts b/ui-tui/src/__tests__/viewportStore.test.ts index 7a571fb95ad0e..d9b2e662514f4 100644 --- a/ui-tui/src/__tests__/viewportStore.test.ts +++ b/ui-tui/src/__tests__/viewportStore.test.ts @@ -87,4 +87,22 @@ describe('viewportStore', () => { expect(getScrollbarSnapshot(handle as any).top).toBe(10) }) + + it('uses fresh scroll height to clear stale scrollbar non-bottom state after shrink', () => { + const handle = { + getFreshScrollHeight: () => 40, + getScrollHeight: () => 60, + getScrollTop: () => 20, + getViewportHeight: () => 20 + } + + const snap = getScrollbarSnapshot(handle as any) + + expect(snap).toEqual({ + scrollHeight: 40, + top: 20, + viewportHeight: 20 + }) + expect(scrollbarSnapshotKey(snap)).toBe('20:20:40') + }) }) diff --git a/ui-tui/src/lib/viewportStore.ts b/ui-tui/src/lib/viewportStore.ts index 25acbd8bebcd6..ab5dcf3e18644 100644 --- a/ui-tui/src/lib/viewportStore.ts +++ b/ui-tui/src/lib/viewportStore.ts @@ -70,12 +70,24 @@ export function getScrollbarSnapshot(s?: ScrollBoxHandle | null): ScrollbarSnaps } const viewportHeight = Math.max(0, s.getViewportHeight()) - const scrollHeight = Math.max(viewportHeight, s.getScrollHeight()) - const maxTop = Math.max(0, scrollHeight - viewportHeight) + const top = Math.max(0, s.getScrollTop()) + const cachedScrollHeight = Math.max(viewportHeight, s.getScrollHeight()) + let scrollHeight = cachedScrollHeight + let maxTop = Math.max(0, scrollHeight - viewportHeight) + + if (top < maxTop) { + const freshScrollHeight = Math.max(viewportHeight, s.getFreshScrollHeight?.() ?? cachedScrollHeight) + const freshMaxTop = Math.max(0, freshScrollHeight - viewportHeight) + + if (top >= freshMaxTop) { + scrollHeight = freshScrollHeight + maxTop = freshMaxTop + } + } return { scrollHeight, - top: Math.max(0, Math.min(maxTop, s.getScrollTop())), + top: Math.max(0, Math.min(maxTop, top)), viewportHeight } } From 6b39ae490d7bf111f319c6ab181ecf32a5295480 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 23:09:37 -0700 Subject: [PATCH 290/376] fix(desktop): let sidebar wheel scrolling chain past the virtualized list MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit With 25+ sessions the recents list virtualizes into its own nested scroller inside the sidebar's scroll container. Both carried overscroll-contain, so once the inner scroller hit a scroll boundary the wheel gesture was consumed instead of chaining to the outer sidebar scroller — read as a mid-list wheel dead-zone while scrollbar drag kept working. Drop the containment on the inner scroller only; the outer sidebar scroller keeps overscroll-contain so the gesture still never escapes the sidebar. Fixes #84964 --- .../sidebar/virtual-session-list.test.tsx | 25 +++++++++++++++++++ .../app/chat/sidebar/virtual-session-list.tsx | 14 ++++++++--- 2 files changed, 35 insertions(+), 4 deletions(-) diff --git a/apps/desktop/src/app/chat/sidebar/virtual-session-list.test.tsx b/apps/desktop/src/app/chat/sidebar/virtual-session-list.test.tsx index 3dec464df15d1..54d98c150c6a7 100644 --- a/apps/desktop/src/app/chat/sidebar/virtual-session-list.test.tsx +++ b/apps/desktop/src/app/chat/sidebar/virtual-session-list.test.tsx @@ -82,4 +82,29 @@ describe('VirtualSessionList', () => { expect(spacer?.style.paddingTop).toBe('') expect(spacer?.style.paddingBottom).toBe('') }) + + it('lets wheel overscroll chain to the outer sidebar scroller (#84964)', () => { + const { getByTestId } = render( + + ) + + const scroller = getByTestId('divider-Today').parentElement?.parentElement?.parentElement + + // The inner virtualized scroller must NOT contain overscroll: it is nested + // inside the sidebar's own scroll container, and containing it swallowed + // wheel events at the inner scroll boundary — the mid-list wheel dead-zone + // at 25+ sessions. Chaining stays inside the sidebar because the OUTER + // scroller keeps overscroll-contain. + expect(scroller?.className).toContain('overflow-y-auto') + expect(scroller?.className).not.toContain('overscroll-contain') + }) }) diff --git a/apps/desktop/src/app/chat/sidebar/virtual-session-list.tsx b/apps/desktop/src/app/chat/sidebar/virtual-session-list.tsx index 48cbe68396087..5e4e93cca91dd 100644 --- a/apps/desktop/src/app/chat/sidebar/virtual-session-list.tsx +++ b/apps/desktop/src/app/chat/sidebar/virtual-session-list.tsx @@ -155,10 +155,16 @@ export const VirtualSessionList: FC = ({ // fade bar reserves its 4px on every platform but stays invisible until // hover — and the wrapper no longer stacks a second scroller, so the // double-gutter this class change was reaching for is already gone. - className={cn( - 'scrollbar-fade relative min-h-0 flex-1 overflow-x-hidden overflow-y-auto overscroll-contain', - className - )} + // + // No `overscroll-contain` here: this scroller is NESTED inside the + // sidebar's own scroll container (index.tsx SCROLL_Y). Containing + // overscroll on the inner scroller swallowed wheel events at its scroll + // boundaries instead of chaining them to the outer sidebar scroller, + // which read as a wheel dead-zone mid-list once 25+ sessions + // virtualized (#84964) — the scrollbar still dragged, only the wheel + // died. The outer sidebar scroller keeps its own overscroll-contain, so + // the gesture still never escapes the sidebar. + className={cn('scrollbar-fade relative min-h-0 flex-1 overflow-x-hidden overflow-y-auto', className)} ref={scrollerRef} >
From da68ecf4b3c5621865f9bf69a3da43e66015f8fb Mon Sep 17 00:00:00 2001 From: David Metcalfe <80915+DavidMetcalfe@users.noreply.github.com> Date: Sun, 21 Jun 2026 19:47:07 -0700 Subject: [PATCH 291/376] design(kanban): add side-by-side dialog prototype (4 variants) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Reference prototype for the upcoming 'replace window.confirm/prompt/alert in the kanban plugin' work. Self-contained HTML — open in any browser, no build step. Served at docs/design/kanban-dialogs/index.html. Four variants side-by-side, each rendered in the same kanban context: - A (Conservative): direct host ConfirmDialog mapping, minimal chrome - B (Strong-fit, Pro's pick): textarea + SVG icon + inline validation - B-refined (synthesis, recommended): auto-focus, dual-validation, cancellable spinner during PATCH, per-task summaries in bulk-many - C (Divergent): undo toast for non-destructive moves Decision matrix at the bottom of the page. Copy is verbatim from web/src/i18n/en.ts (confirmDone, confirmArchive, confirmBlocked, trash.confirm, completionSummary, etc.). Design pass: Gemini 3.1 Pro (initial brief) + GPT-OSS 120B (cross-vendor review). Not shipped to users — review reference only. --- docs/design/kanban-dialogs/index.html | 903 ++++++++++++++++++++++++++ 1 file changed, 903 insertions(+) create mode 100644 docs/design/kanban-dialogs/index.html diff --git a/docs/design/kanban-dialogs/index.html b/docs/design/kanban-dialogs/index.html new file mode 100644 index 0000000000000..f32f74f5a8e12 --- /dev/null +++ b/docs/design/kanban-dialogs/index.html @@ -0,0 +1,903 @@ + + + + + +Hermes Kanban — Native Dialog Prototypes + + + + +
+

Hermes Kanban — Native Dialog Prototypes

+

+ Four approaches to replacing window.confirm(), + window.prompt(), and window.alert() in + plugins/kanban/dashboard/dist/index.js. Click a page-level + trigger to fire the same flow in every variant simultaneously, or use the + per-variant buttons to fire flows unique to that variant. +

+
+ + + + + + + +
+
+ +
+ + + + +
+
+

Variant A

+
Conservative
+

Direct 1:1 mapping to the host's ConfirmDialog. Single-line input. Minimal chrome.

+
+
+
Click a trigger to preview.
+
+
+ Trade-off: Safest to ship — zero new components, zero new + patterns. Fails the GPT-OSS review on three points: no multi-line + summary, validation triggers a SECOND dialog instead of inline, no bulk + affordance. +
+
+ + + + +
+
+

Variant B

+
Strong-fit (Pro's pick)
+

Textarea + contextual SVG icon + inline validation + pluralized copy. What the design brief recommends.

+
+
+
Click a trigger to preview.
+
+
+ Trade-off: Best baseline, but Pro suggested either + a disabled-button or an inline error — GPT-OSS caught that both + are needed (button disabled AND error visible) for screen-reader users. + Bottom-sheet on mobile is the right move but still has keyboard edge cases. +
+
+ + + + +
+
+

Variant B-refined

+
Synthesis (recommended)
+

All of B's improvements + GPT-OSS fixes: auto-focus, dual-validation, cancellable-spinner, per-task summaries in bulk.

+
+
+
Click a trigger to preview.
+
+
+ Why this is the recommendation: Single contextual icon + (not four) keeps the title readable. Confirm button stays enabled until + textarea has content; the inline error appears on submit-attempted-empty + AND on blur if still empty. Cancel button stays clickable during the + PATCH (only the confirm shows the spinner) so users can abort slow + networks. Bulk-many shows an expandable list with per-task summary + fields. +
+
+ + + + +
+
+

Variant C

+
Divergent: undo toast
+

Skip the modal entirely for non-destructive moves. Optimistic UI + 5s undo in a bottom-right toast.

+
+
+
Click a trigger to preview.
+
+
+
+ Trade-off: Radical speedup for routine moves, but breaks + the required-summary flow (you can't optimistically "complete" a task + that's missing required schema data). Best used as a + complement to B-refined — undo toast for the safe moves, + modal for done/blocked/archive. +
+
+ +
+ +
+

Decision matrix — recommended pick: B-refined

+ + + + + + + + + + + + + + + + + +
CapabilityABB-refinedC
Centered modal (Radix) ✓ ✓ ✓ ✗
Multi-line summary ✗ ✓ ✓ ✗
Single contextual icon ✗ 4 ✓ ✗
Inline validation (no second dialog)✗ ~ ✓ ✗
Disabled button + persistent error ✗ ~ ✓ ✗
Cancellable spinner during PATCH ✗ ✗ ✓ ✓
Bulk-many expandable list ✗ ✗ ✓ ✗
Auto-focus textarea + mobile scroll ✗ ~ ✓ ✗
Toast on success ✗ ✗ ✗ ✓
Toast on error ✗ ✗ opt. ✓
Survives the required-summary flow ✗ ✓ ✓ ✗
+
+ + + + + \ No newline at end of file From 7448e7a5065daf17f33fb2c21349b1cb2ed69d0d Mon Sep 17 00:00:00 2001 From: David Metcalfe <80915+DavidMetcalfe@users.noreply.github.com> Date: Sun, 21 Jun 2026 19:59:09 -0700 Subject: [PATCH 292/376] feat(plugins): expose Dialog/ConfirmDialog/Toast/useToast/useConfirmDelete on plugin SDK MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Additive expansion of window.__HERMES_PLUGIN_SDK__. Plugins can now render host-styled dialogs, confirmations, and toasts instead of falling back to window.alert/confirm/prompt. New components: Dialog, DialogClose, DialogContent, DialogDescription, DialogFooter, DialogHeader, DialogTitle, ConfirmDialog, Toast. New hooks: useToast (replaces showToast/toast pair), useConfirmDelete (single-id delete-confirm state machine). SDK_CONTRACT_VERSION unchanged at 1.1.0 — additive surface per sdk.d.ts:23-25 (no major bump required). Consumer: kanban plugin's 'replace native dialogs' work, see issue #50547. A reference prototype showing 4 design variants is committed at docs/design/kanban-dialogs/index.html. Adds web/src/plugins/registry.test.ts (3 vitest cases) that smoke-test the new keys are wired and that the version constant is unchanged. --- web/src/plugins/registry.test.ts | 54 ++++++++++++++++++++++++++++++++ web/src/plugins/registry.ts | 24 ++++++++++++++ web/src/plugins/sdk.d.ts | 24 ++++++++++++++ 3 files changed, 102 insertions(+) create mode 100644 web/src/plugins/registry.test.ts diff --git a/web/src/plugins/registry.test.ts b/web/src/plugins/registry.test.ts new file mode 100644 index 0000000000000..8fe064785824d --- /dev/null +++ b/web/src/plugins/registry.test.ts @@ -0,0 +1,54 @@ +/** + * Smoke test for the plugin SDK surface additions. + * + * Verifies that `exposePluginSDK()` writes the new dialog/toast primitives to + * `window.__HERMES_PLUGIN_SDK__`. Each new key is checked individually so a + * regression in one helper doesn't mask the others. + * + * Companion to the additive PR "feat(plugins): expose Dialog/ConfirmDialog/ + * Toast/useToast/useConfirmDelete on the plugin SDK". See issue #50547. + */ +import { beforeEach, describe, expect, it } from "vitest"; +import { exposePluginSDK } from "./registry"; + +describe("plugin SDK dialog/toast surface", () => { + beforeEach(() => { + // Reset window between tests so exposePluginSDK() writes fresh. + (globalThis as any).window = { + __HERMES_PLUGINS__: undefined, + __HERMES_PLUGIN_SDK__: undefined, + }; + }); + + it("exposes Dialog + subcomponents on components", () => { + exposePluginSDK(); + const sdk = (globalThis as any).window.__HERMES_PLUGIN_SDK__; + expect(sdk.components.Dialog).toBeDefined(); + expect(sdk.components.DialogContent).toBeDefined(); + expect(sdk.components.DialogHeader).toBeDefined(); + expect(sdk.components.DialogTitle).toBeDefined(); + expect(sdk.components.DialogDescription).toBeDefined(); + expect(sdk.components.DialogFooter).toBeDefined(); + expect(sdk.components.DialogClose).toBeDefined(); + expect(sdk.components.ConfirmDialog).toBeDefined(); + expect(sdk.components.Toast).toBeDefined(); + }); + + it("exposes useToast and useConfirmDelete on hooks", () => { + exposePluginSDK(); + const sdk = (globalThis as any).window.__HERMES_PLUGIN_SDK__; + expect(typeof sdk.hooks.useToast).toBe("function"); + expect(typeof sdk.hooks.useConfirmDelete).toBe("function"); + // Original React hooks still present (no accidental removal). + expect(typeof sdk.hooks.useState).toBe("function"); + expect(typeof sdk.hooks.useCallback).toBe("function"); + }); + + it("does not bump SDK_CONTRACT_VERSION (additive change)", () => { + exposePluginSDK(); + const sdk = (globalThis as any).window.__HERMES_PLUGIN_SDK__; + // Pre-existing version per registry.ts:98. This test fails if a future + // PR accidentally bumps the major for an additive surface change. + expect(sdk.sdkVersion).toBe("1.1.0"); + }); +}); \ No newline at end of file diff --git a/web/src/plugins/registry.ts b/web/src/plugins/registry.ts index 392c536d0adb1..df7c77d8d06ce 100644 --- a/web/src/plugins/registry.ts +++ b/web/src/plugins/registry.ts @@ -22,6 +22,14 @@ import { cn, timeAgo, isoTimeAgo } from "@/lib/utils"; import { Badge } from "@nous-research/ui/ui/components/badge"; import { Button } from "@nous-research/ui/ui/components/button"; import { Checkbox } from "@nous-research/ui/ui/components/checkbox"; +import { ConfirmDialog } from "@nous-research/ui/ui/components/confirm-dialog"; +import { + Dialog, DialogClose, DialogContent, DialogDescription, + DialogFooter, DialogHeader, DialogTitle, +} from "@nous-research/ui/ui/components/dialog"; +import { Toast } from "@nous-research/ui/ui/components/toast"; +import { useConfirmDelete } from "@nous-research/ui/hooks/use-confirm-delete"; +import { useToast } from "@nous-research/ui/hooks/use-toast"; import { Select, SelectOption } from "@nous-research/ui/ui/components/select"; import { Card, CardHeader, CardTitle, CardContent } from "@nous-research/ui/ui/components/card"; import { Input } from "@nous-research/ui/ui/components/input"; @@ -121,6 +129,13 @@ export function exposePluginSDK() { useRef, useContext, createContext, + // useToast returns { showToast, toast } where toast is the current + // visible toast (or null) and showToast(message, 'success'|'error') + // replaces it. useConfirmDelete({ onDelete }) returns the state + // machine (requestDelete / confirm / cancel / isOpen / isDeleting / + // pendingId) for single-id delete confirmations. + useToast, + useConfirmDelete, }, // Hermes API client @@ -148,6 +163,14 @@ export function exposePluginSDK() { Badge, Button, Checkbox, + ConfirmDialog, + Dialog, + DialogClose, + DialogContent, + DialogDescription, + DialogFooter, + DialogHeader, + DialogTitle, Input, Label, Select, @@ -156,6 +179,7 @@ export function exposePluginSDK() { Tabs, TabsList, TabsTrigger, + Toast, PluginSlot, }, diff --git a/web/src/plugins/sdk.d.ts b/web/src/plugins/sdk.d.ts index c55b855ab822e..db31b73e516e6 100644 --- a/web/src/plugins/sdk.d.ts +++ b/web/src/plugins/sdk.d.ts @@ -105,6 +105,30 @@ export interface HermesPluginSDK { useRef: typeof import("react").useRef; useContext: typeof import("react").useContext; createContext: typeof import("react").createContext; + /** + * Toast feedback. Returns ``{ showToast, toast }`` where ``toast`` is the + * current visible toast (or null) and ``showToast(message, 'success' | 'error')`` + * replaces it. Mount the ``Toast`` component (also on ``components``) somewhere + * in your plugin's tree to render the visible toast. + */ + useToast: () => { + showToast: (message: string, type: "success" | "error") => void; + toast: { message: string; type: "success" | "error" } | null; + }; + /** + * Single-id delete-confirm state machine. Pass ``onDelete(id)`` (an async + * function). Returns ``{ requestDelete, confirm, cancel, isOpen, isDeleting, + * pendingId }``. Pair with the ``ConfirmDialog`` primitive (passed as + * ``open={isOpen}`` etc.) or any custom dialog. + */ + useConfirmDelete: (opts: { onDelete: (id: TId) => Promise }) => { + requestDelete: (id: TId) => void; + confirm: () => Promise; + cancel: () => void; + isOpen: boolean; + isDeleting: boolean; + pendingId: TId | null; + }; }; /** From 08f32a63351f32aab5882f6a8d81a570beceb113 Mon Sep 17 00:00:00 2001 From: David Metcalfe <80915+DavidMetcalfe@users.noreply.github.com> Date: Sun, 21 Jun 2026 21:55:20 -0700 Subject: [PATCH 293/376] fix(kanban): replace native browser dialogs with in-app ConfirmDialog MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Migrates 8 of 12 native dialog call sites in the kanban dashboard plugin to the SDK's ConfirmDialog primitive (added in PR #50550): - moveTask, moveSelected, applyBulk, deleteTask, deleteSelected, archiveBoard, removeAttachment, doPatch The 4 remaining carve-outs (window.prompt for completion summary, window.alert for missing summary, cli_hint clipboard fallback) are documented inline — the host's ConfirmDialog hardcodes onClick → unmount, preventing the keep-open-across-validation behavior the completion-summary form needs. Followup: upstream a `disabled` prop to ConfirmDialog and rebuild the completion body using host Dialog components. New architecture: - useKanbanDialogs(t) — Promise-based dialog state machine at KanbanPage scope. request({kind, ...}) returns {confirmed, summary?}. - KanbanDialog component — renders ConfirmDialog from SDK for kind=confirm. - performMoveTask(taskId, newStatus, count, summary) — extracted shared dispatch path for single + bulk moves (optimistic UI + PATCH/POST + error recovery). - requestDialog prop threading — KanbanPage → BoardSwitcher, TaskDrawer → TaskDetail → doPatch/AttachmentsSection. Every call site has a defensive fallback to window.confirm if the prop is missing (verified by test_dashboard_done_actions_prompt_for_completion_summary counting the cancel guards + destructive:true markers in the bundle). New host i18n keys (web/src/i18n/en.ts + types.ts): - kanban.confirmDoneMany / confirmArchiveMany / confirmBlockedMany - kanban.trash.confirmTitle / confirmManyTitle Tests: - Replaced bundle-string-only completion-summary test with behavioral coverage: bundle cancel-guard count + destructive marker count, plus backend tests that confirm cancel preserves old status and confirm dispatches the expected PATCH/DELETE body. - Removed the SDK_CONTRACT_VERSION snapshot test from web/src/plugins/registry.test.ts (forbidden by AGENTS.md "Don't write change-detector tests"; the two remaining tests in that file already cover the new SDK surface behaviorally). Closes #50547 (consumers of #50550). Cross-vendor re-review: Gemini 3.5 Flash + GPT-OSS 120B (both SHOULD-FIX, no remaining BLOCKERs after these fixes). --- plugins/kanban/dashboard/dist/index.js | 615 +++++++++++++----- tests/plugins/test_kanban_dashboard_plugin.py | 288 ++++++++ web/src/i18n/en.ts | 10 + web/src/i18n/types.ts | 9 + web/src/plugins/registry.test.ts | 8 - 5 files changed, 772 insertions(+), 158 deletions(-) diff --git a/plugins/kanban/dashboard/dist/index.js b/plugins/kanban/dashboard/dist/index.js index f15850f73493f..dc582a7660256 100644 --- a/plugins/kanban/dashboard/dist/index.js +++ b/plugins/kanban/dashboard/dist/index.js @@ -116,6 +116,13 @@ archived: "Archive this task? It disappears from the default board view.", blocked: "Mark this task as blocked? The worker's claim is released.", }; + // Pluralized variants used by getDestructiveConfirm() when count > 1. + // Each entry may use {n} as a placeholder for the count. + const FALLBACK_DESTRUCTIVE_MANY = { + done: "Mark {n} tasks as done? The workers' claims are released and dependent children become ready.", + archived: "Archive {n} tasks? They disappear from the default board view.", + blocked: "Mark {n} tasks as blocked? The workers' claims are released.", + }; const FALLBACK_DIAGNOSTIC_EVENT_LABELS = { completion_blocked_hallucination: "⚠ Completion blocked — phantom card ids", suspected_hallucinated_references: "⚠ Prose referenced phantom card ids", @@ -142,9 +149,18 @@ function getColumnHelp(t, status) { return tx(t, "columnHelp." + status, FALLBACK_COLUMN_HELP[status] || ""); } - function getDestructiveConfirm(t, status) { + function getDestructiveConfirm(t, status, count) { const key = DESTRUCTIVE_KEYS[status]; if (!key) return null; + // For bulk operations, use the *Many variant of the i18n key so the + // copy pluralizes correctly ("Mark 3 tasks as done?" instead of + // "Mark this task as done?"). Falls back to the singular English + // string if a translation for the *Many key isn't shipped. + if (count && count > 1) { + const manyKey = key + "Many"; + const manyFallback = FALLBACK_DESTRUCTIVE_MANY[status] || FALLBACK_DESTRUCTIVE[status]; + return tx(t, manyKey, manyFallback, { n: count }); + } return tx(t, key, FALLBACK_DESTRUCTIVE[status]); } function getDiagnosticEventLabel(t, kind) { @@ -174,25 +190,75 @@ return p.phantom_cards || p.phantom_refs || []; } - // Takes an optional `t` so the prompt/alert text is localised. Callers - // outside React components can pass null and fall through to English. - function withCompletionSummary(patch, count, t) { - if (!patch || patch.status !== "done") return patch; - const label = count && count > 1 ? `${count} selected task(s)` : "this task"; - const value = window.prompt( - tx(t, "completionSummary", - "Completion summary for {label}. This is stored as the task result.", - { label: label }), - "", - ); - if (value === null) return null; - const summary = value.trim(); - if (!summary) { - window.alert(tx(t, "completionSummaryRequired", - "Completion summary is required before marking a task done.")); - return null; - } - return Object.assign({}, patch, { result: summary, summary }); + // Helpers for the dialog state machine used by `useKanbanDialogs` below. + // The dialog API is Promise-based so call sites can preserve their + // synchronous-ish flow: ``await kanbanDialogs.request(...)`` and then + // continue with the optimistic UI + PATCH. See #50547. + function dialogLabelForCount(count, t) { + return count && count > 1 ? tx(t, "selectedTasks", "{n} selected tasks", { n: count }) : tx(t, "thisTask", "this task"); + } + + /** + * Hook owning the kanban plugin's modal dialog state. Returns + * - `request(req)` — imperative API. Resolves to + * `{ confirmed: false }` if the user cancels, or + * `{ confirmed: true, summary?: string }` if they confirm. + * - `dialogState` — current dialog descriptor for rendering, or null. + * - `dialogProps` — onConfirm/onCancel handlers bound to the current + * request. + * + * `req` shapes: + * { kind: "confirm", title, description, confirmLabel, destructive } + * + * The "completion" kind (textarea prompt) is deferred: the host's + * ConfirmDialog hardcodes onClick → unmount, preventing validation- + * state retention. See KanbanDialogs doc comment. + */ + function useKanbanDialogs(t) { + const [dialogState, setDialogState] = React.useState(null); + const resolverRef = React.useRef(null); + + const request = React.useCallback(function (req) { + return new Promise(function (resolve) { + resolverRef.current = resolve; + setDialogState(req); + }); + }, []); + + const close = React.useCallback(function (confirmed, extras) { + const resolve = resolverRef.current; + resolverRef.current = null; + setDialogState(null); + if (resolve) { + resolve(Object.assign({ confirmed: confirmed }, extras || {})); + } + }, []); + + const onConfirm = React.useCallback(function (maybeSummary) { + close(true, maybeSummary ? { summary: maybeSummary } : null); + }, [close]); + const onCancel = React.useCallback(function () { close(false, null); }, [close]); + + // Wrap the ConfirmDialog props so call sites can hand them straight + // to . Title/description/confirmLabel are + // sourced from the current dialog state. For "completion" the dialog + // body (textarea + dual-validation) is rendered separately. + const dialogProps = React.useMemo(function () { + if (!dialogState) return null; + return { + open: true, + title: dialogState.title || "", + description: dialogState.description, + confirmLabel: dialogState.confirmLabel || (dialogState.kind === "completion" + ? tx(t, "confirm", "Confirm") + : tx(t, "ok", "OK")), + destructive: !!dialogState.destructive, + onConfirm: function () { onConfirm(); }, + onCancel: onCancel, + }; + }, [dialogState, t, onConfirm, onCancel]); + + return { dialogState: dialogState, dialogProps: dialogProps, request: request }; } const API = "/api/plugins/kanban"; @@ -503,12 +569,40 @@ } } + // ------------------------------------------------------------------------- + // Dialog renderer + // ------------------------------------------------------------------------- + + /** + * Single component that owns the kanban plugin's modal dialog UI. Renders + * whichever dialog `useKanbanDialogs` is currently requesting, or nothing + * if no dialog is open. + * + * Currently supports one dialog kind: + * - "confirm" → standard ConfirmDialog (title + description + buttons) + * + * The "completion" kind (Mark Done → textarea prompt) is not yet wired + * because the host's ConfirmDialog hardcodes `onClick → unmount`, which + * prevents keeping the dialog open across a validation failure. See + * issue #50547 followups. Completion summaries triggered from the + * side-drawer use a documented carve-out (`withCompletionSummary` in + * TaskDetail) until that lands. + */ + function KanbanDialogs(props) { + const { dialogProps, dialogState } = props; + if (!dialogState || !dialogProps) return null; + const ConfirmDialog = SDK.components.ConfirmDialog; + if (!ConfirmDialog) return null; + return h(ConfirmDialog, dialogProps); + } + // ------------------------------------------------------------------------- // Root page // ------------------------------------------------------------------------- function KanbanPage() { const { t } = useI18n(); + const kanbanDialogs = useKanbanDialogs(t); const [board, setBoard] = useState(() => readSelectedBoard() || null); const [boardList, setBoardList] = useState([]); // [{slug, name, counts, ...}] const [showNewBoard, setShowNewBoard] = useState(false); @@ -719,17 +813,67 @@ }, [boardData, tenantFilter, assigneeFilter, search]); // --- actions ------------------------------------------------------------ - const moveTask = useCallback(function (taskId, newStatus) { - const confirmMsg = getDestructiveConfirm(t, newStatus); - if (confirmMsg && !window.confirm(confirmMsg)) return; - const patch = withCompletionSummary({ status: newStatus }, 1, t); - if (!patch) return; + // Performs the actual move (optimistic UI + PATCH) once any required + // confirmation and/or completion summary has been collected by the + // caller. Extracted so moveTask / moveSelected / applyBulk can all + // share the same dispatch path regardless of how confirmation was + // collected (synchronous window.confirm in the original code, async + // dialog via useKanbanDialogs now). + // taskId — required when count <= 1 (single-task PATCH endpoint) + // — ignored when count > 1 (bulk endpoint uses selectedIds) + // summary — completion summary string, or null/undefined to skip + const performMoveTask = useCallback(function (taskId, newStatus, count, summary) { + const patch = { status: newStatus }; + const finalPatch = summary + ? Object.assign({}, patch, { result: summary, summary: summary }) + : patch; + if (count > 1) { + // Bulk path: optimistic UI prepends all moved tasks to dest column. + setBoardData(function (b) { + if (!b) return b; + const moved = []; + const columns = b.columns.map(function (col) { + const kept = []; + for (const tk of col.tasks) { + if (selectedIds.has(tk.id)) moved.push(Object.assign({}, tk, { status: newStatus })); + else kept.push(tk); + } + return Object.assign({}, col, { tasks: kept }); + }); + const dest = columns.find(function (c) { return c.name === newStatus; }); + if (dest) dest.tasks = moved.concat(dest.tasks); + return Object.assign({}, b, { columns }); + }); + const ids = Array.from(selectedIds); + SDK.fetchJSON(withBoard(`${API}/tasks/bulk`, board), { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify(Object.assign({ ids: ids }, finalPatch)), + }).then(function (res) { + const failed = (res.results || []).filter(function (r) { return !r.ok; }); + if (failed.length > 0) { + setError(`Bulk move: ${failed.length} of ${res.results.length} failed`); + setFailedIds(new Set(failed.map(function (f) { return f.id; }))); + } else { + setFailedIds(new Set()); + } + setSelectedIds(new Set()); + setLastSelectedId(null); + loadBoard(); + }).catch(function (err) { + setError(`Move failed: ${err.message || err}`); + setFailedIds(new Set(selectedIds)); + loadBoard(); + }); + return; + } + // Single-task path. setBoardData(function (b) { if (!b) return b; let moved = null; const columns = b.columns.map(function (col) { - const next = col.tasks.filter(function (t) { - if (t.id === taskId) { moved = Object.assign({}, t, { status: newStatus }); return false; } + const next = col.tasks.filter(function (tk) { + if (tk.id === taskId) { moved = Object.assign({}, tk, { status: newStatus }); return false; } return true; }); return Object.assign({}, col, { tasks: next }); @@ -743,12 +887,74 @@ SDK.fetchJSON(withBoard(`${API}/tasks/${encodeURIComponent(taskId)}`, board), { method: "PATCH", headers: { "Content-Type": "application/json" }, - body: JSON.stringify(patch), + body: JSON.stringify(finalPatch), }).catch(function (err) { setError(tx(t, "moveFailed", "Move failed: ") + parseApiErrorMessage(err)); loadBoard(); }); - }, [loadBoard, board, t]); + }, [loadBoard, board, t, selectedIds]); + + // Pre-dispatch dialog step for both moveTask and moveSelected. Drives + // the new in-app ConfirmDialog instead of window.confirm. The flow: + // 1. If newStatus is destructive (done/archived/blocked), open + // a confirm dialog. + // 2. If newStatus is "done", additionally open a completion-summary + // dialog (chained via Promise). + // 3. On confirm of all steps, call performMoveTask. + // 4. On cancel anywhere, do nothing. + const requestMoveConfirm = useCallback(function (newStatus, count) { + const confirmMsg = getDestructiveConfirm(t, newStatus, count); + if (!confirmMsg) return Promise.resolve({ confirmed: true }); + return kanbanDialogs.request({ + kind: "confirm", + title: tx(t, "confirmStatusTitle." + newStatus, "Confirm status change"), + description: confirmMsg, + confirmLabel: tx(t, "confirmStatusLabel." + newStatus, "Confirm"), + destructive: true, + }); + }, [kanbanDialogs, t]); + + const requestCompletionSummary = useCallback(function (count) { + const label = dialogLabelForCount(count, t); + // Uses window.prompt as a documented carve-out — the host's + // ConfirmDialog hardcodes onClick → unmount (confirmedRef + Radix + // AlertDialogAction), making it impossible to keep a dialog open + // across a validation failure. Once ConfirmDialog grows a + // disabled prop upstream, this switches to a Dialog-based + // completion body (see KanbanDialogs doc comment). + var summary = window.prompt( + tx(t, "completionSummary", + "Completion summary for {label}. This is stored as the task result.", + { label: label }), + "", + ); + if (summary === null) return Promise.resolve({ confirmed: false }); + summary = summary.trim(); + if (!summary) { + window.alert(tx(t, "completionSummaryRequired", + "Completion summary is required before marking a task done.")); + return Promise.resolve({ confirmed: false }); + } + return Promise.resolve({ confirmed: true, summary: summary }); + }, [t]); + + // Single-task card move. Drives confirmation + completion summary + // dialogs via the hook, then dispatches via performMoveTask. + const moveTask = useCallback(function (taskId, newStatus) { + requestMoveConfirm(newStatus, 1) + .then(function (r1) { + if (!r1.confirmed) return null; + if (newStatus !== "done") { + performMoveTask(taskId, newStatus, 1, null); + return null; + } + return requestCompletionSummary(1).then(function (r2) { + if (!r2.confirmed) return null; + performMoveTask(taskId, newStatus, 1, r2.summary || null); + }); + }) + .catch(function () { /* dialog cancelled */ }); + }, [requestMoveConfirm, requestCompletionSummary, performMoveTask]); const clearSelected = useCallback(function () { setSelectedIds(new Set()); @@ -756,49 +962,23 @@ setFailedIds(new Set()); }, []); const moveSelected = useCallback(function (newStatus) { - const confirmMsg = DESTRUCTIVE_TRANSITIONS[newStatus]; - if (confirmMsg && !window.confirm(confirmMsg)) return; if (selectedIds.size === 0) return; - const patch = withCompletionSummary({ status: newStatus }, selectedIds.size); - if (!patch) return; - const ids = Array.from(selectedIds); - // Optimistic UI: remove selected from all columns and prepend to target. - setBoardData(function (b) { - if (!b) return b; - const moved = []; - const columns = b.columns.map(function (col) { - const kept = []; - for (const t of col.tasks) { - if (selectedIds.has(t.id)) moved.push(Object.assign({}, t, { status: newStatus })); - else kept.push(t); + const count = selectedIds.size; + const taskId = Array.from(selectedIds)[0]; // representative id for performMoveTask's single-task branch + requestMoveConfirm(newStatus, count) + .then(function (r1) { + if (!r1.confirmed) return null; + if (newStatus !== "done") { + performMoveTask(taskId, newStatus, count, null); + return null; } - return Object.assign({}, col, { tasks: kept }); - }); - const dest = columns.find(function (c) { return c.name === newStatus; }); - if (dest) dest.tasks = moved.concat(dest.tasks); - return Object.assign({}, b, { columns }); - }); - SDK.fetchJSON(withBoard(`${API}/tasks/bulk`, board), { - method: "POST", - headers: { "Content-Type": "application/json" }, - body: JSON.stringify(Object.assign({ ids }, patch)), - }).then(function (res) { - const failed = (res.results || []).filter(function (r) { return !r.ok; }); - if (failed.length > 0) { - setError(`Bulk move: ${failed.length} of ${res.results.length} failed`); - setFailedIds(new Set(failed.map(function (f) { return f.id; }))); - } else { - setFailedIds(new Set()); - } - setSelectedIds(new Set()); - setLastSelectedId(null); - loadBoard(); - }).catch(function (err) { - setError(`Move failed: ${err.message || err}`); - setFailedIds(new Set(selectedIds)); - loadBoard(); - }); - }, [selectedIds, loadBoard, board]); + return requestCompletionSummary(count).then(function (r2) { + if (!r2.confirmed) return null; + performMoveTask(taskId, newStatus, count, r2.summary || null); + }); + }) + .catch(function () { /* dialog cancelled */ }); + }, [selectedIds, requestMoveConfirm, requestCompletionSummary, performMoveTask]); const createTask = useCallback(function (body) { return SDK.fetchJSON(withBoard(`${API}/tasks`, board), { @@ -895,53 +1075,67 @@ const applyBulk = useCallback(function (patch, confirmMsg) { if (selectedIds.size === 0) return; - if (confirmMsg && !window.confirm(confirmMsg)) return; - const finalPatch = withCompletionSummary(patch, selectedIds.size, t); - if (!finalPatch) return; - const body = Object.assign({ ids: Array.from(selectedIds) }, finalPatch); - // Optimistic UI for status moves (same pattern as moveSelected). - if (finalPatch.status) { - setBoardData(function (b) { - if (!b) return b; - const moved = []; - const columns = b.columns.map(function (col) { - const kept = []; - for (const t of col.tasks) { - if (selectedIds.has(t.id)) moved.push(Object.assign({}, t, { status: finalPatch.status })); - else kept.push(t); + const count = selectedIds.size; + const run = function () { + const finalPatch = patch; + const body = Object.assign({ ids: Array.from(selectedIds) }, finalPatch); + // Optimistic UI for status moves (same pattern as moveSelected). + if (finalPatch.status) { + setBoardData(function (b) { + if (!b) return b; + const moved = []; + const columns = b.columns.map(function (col) { + const kept = []; + for (const t of col.tasks) { + if (selectedIds.has(t.id)) moved.push(Object.assign({}, t, { status: finalPatch.status })); + else kept.push(t); + } + return Object.assign({}, col, { tasks: kept }); + }); + const dest = columns.find(function (c) { return c.name === finalPatch.status; }); + if (dest) dest.tasks = moved.concat(dest.tasks); + return Object.assign({}, b, { columns }); + }); + } + SDK.fetchJSON(withBoard(`${API}/tasks/bulk`, board), { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify(body), + }) + .then(function (res) { + const failed = (res.results || []).filter(function (r) { return !r.ok; }); + if (failed.length > 0) { + setError(tx(t, "bulkFailed", "Bulk: ") + + `${failed.length} of ${res.results.length} failed: ` + + failed.slice(0, 3).map(function (f) { return `${f.id} (${f.error})`; }).join("; ")); + setFailedIds(new Set(failed.map(function (f) { return f.id; }))); + } else { + setFailedIds(new Set()); } - return Object.assign({}, col, { tasks: kept }); + setSelectedIds(new Set()); + setLastSelectedId(null); + loadBoard(); + }) + .catch(function (e) { + setError(String(e.message || e)); + setFailedIds(new Set(selectedIds)); + loadBoard(); }); - const dest = columns.find(function (c) { return c.name === finalPatch.status; }); - if (dest) dest.tasks = moved.concat(dest.tasks); - return Object.assign({}, b, { columns }); - }); + }; + if (!confirmMsg) { + run(); + return; } - SDK.fetchJSON(withBoard(`${API}/tasks/bulk`, board), { - method: "POST", - headers: { "Content-Type": "application/json" }, - body: JSON.stringify(body), - }) - .then(function (res) { - const failed = (res.results || []).filter(function (r) { return !r.ok; }); - if (failed.length > 0) { - setError(tx(t, "bulkFailed", "Bulk: ") + - `${failed.length} of ${res.results.length} failed: ` + - failed.slice(0, 3).map(function (f) { return `${f.id} (${f.error})`; }).join("; ")); - setFailedIds(new Set(failed.map(function (f) { return f.id; }))); - } else { - setFailedIds(new Set()); - } - setSelectedIds(new Set()); - setLastSelectedId(null); - loadBoard(); - }) - .catch(function (e) { - setError(String(e.message || e)); - setFailedIds(new Set(selectedIds)); - loadBoard(); - }); - }, [selectedIds, loadBoard, board, t]); + kanbanDialogs.request({ + kind: "confirm", + title: tx(t, "bulkConfirmTitle", "Apply bulk change"), + description: confirmMsg, + confirmLabel: tx(t, "apply", "Apply"), + destructive: false, + }).then(function (r) { + if (r.confirmed) run(); + }).catch(function () { /* cancelled */ }); + }, [selectedIds, loadBoard, board, t, kanbanDialogs]); // --- board switching ---------------------------------------------------- const switchBoard = useCallback(function (nextSlug) { @@ -1000,30 +1194,46 @@ }, [board, loadBoardList, switchBoard]); const deleteTask = useCallback(function (taskId) { - if (!window.confirm(tx(t, "trash.confirm", FALLBACK_TRASH.confirm))) return Promise.resolve(); - return SDK.fetchJSON(`${API}/tasks/${encodeURIComponent(taskId)}`, { - method: "DELETE", - }).then(function () { - loadBoard(); - setSelectedIds(function (prev) { - const next = new Set(prev); - next.delete(taskId); - return next; - }); - }).catch(function (e) { setError(String(e.message || e)); }); - }, [board, loadBoard, t]); + return kanbanDialogs.request({ + kind: "confirm", + title: tx(t, "trash.confirmTitle", "Delete task?"), + description: tx(t, "trash.confirm", FALLBACK_TRASH.confirm), + confirmLabel: tx(t, "common.delete", "Delete"), + destructive: true, + }).then(function (r) { + if (!r.confirmed) return null; + return SDK.fetchJSON(`${API}/tasks/${encodeURIComponent(taskId)}`, { + method: "DELETE", + }).then(function () { + loadBoard(); + setSelectedIds(function (prev) { + const next = new Set(prev); + next.delete(taskId); + return next; + }); + }).catch(function (e) { setError(String(e.message || e)); }); + }).catch(function () { /* cancelled */ }); + }, [board, loadBoard, t, kanbanDialogs]); const deleteSelected = useCallback(function (count) { if (selectedIds.size === 0) return Promise.resolve(); - if (!window.confirm(tx(t, "trash.confirmMany", "Permanently delete {n} selected tasks? This cannot be undone.", { n: count }))) return Promise.resolve(); - const ids = Array.from(selectedIds); - setSelectedIds(new Set()); - return Promise.all(ids.map(function (id) { - return SDK.fetchJSON(`${API}/tasks/${encodeURIComponent(id)}`, { method: "DELETE" }); - })).then(function () { - loadBoard(); - }).catch(function (e) { setError(String(e.message || e)); }); - }, [selectedIds, board, loadBoard, t]); + kanbanDialogs.request({ + kind: "confirm", + title: tx(t, "trash.confirmManyTitle", "Delete {n} tasks?", { n: count }), + description: tx(t, "trash.confirmMany", "Permanently delete {n} selected tasks? This cannot be undone.", { n: count }), + confirmLabel: tx(t, "common.delete", "Delete"), + destructive: true, + }).then(function (r) { + if (!r.confirmed) return null; + const ids = Array.from(selectedIds); + setSelectedIds(new Set()); + return Promise.all(ids.map(function (id) { + return SDK.fetchJSON(`${API}/tasks/${encodeURIComponent(id)}`, { method: "DELETE" }); + })).then(function () { + loadBoard(); + }).catch(function (e) { setError(String(e.message || e)); }); + }).catch(function () { /* cancelled */ }); + }, [selectedIds, board, loadBoard, t, kanbanDialogs]); // --- render ------------------------------------------------------------- if (loading && !boardData) { @@ -1054,6 +1264,7 @@ onNewClick: function () { setShowNewBoard(true); }, onSettingsClick: function () { setShowBoardSettings(true); }, onDeleteBoard: deleteBoard, + requestDialog: function (req) { return kanbanDialogs.request(req); }, }), showNewBoard ? h(NewBoardDialog, { onCancel: function () { setShowNewBoard(false); }, @@ -1097,6 +1308,10 @@ onDelete: deleteSelected, }) : null, error ? h("div", { className: "text-xs text-destructive px-2" }, error) : null, + h(KanbanDialogs, { + dialogProps: kanbanDialogs.dialogProps, + dialogState: kanbanDialogs.dialogState, + }), h(BoardColumns, { board: filteredBoard, boardMeta: boardList.find(function (item) { return item.slug === board; }) || null, @@ -1112,6 +1327,7 @@ onMove: moveTask, onMoveSelected: moveSelected, onDelete: deleteTask, + onDeleteSelected: deleteSelected, onOpen: setSelectedTaskId, onCreate: createTask, allTasks: boardData.columns.reduce(function (acc, c) { return acc.concat(c.tasks); }, []), @@ -1126,6 +1342,11 @@ allTasks: boardData.columns.reduce(function (acc, c) { return acc.concat(c.tasks); }, []), assignees: (boardData && boardData.assignees) || [], eventTick: taskEventTick[selectedTaskId] || 0, + // Hook for the side-drawer's doPatch to use the same in-app + // dialog machinery as the column-card flow. TaskDetail also + // owns its own kanbanDialogs so the dialog portal mounts in + // its tree; we expose requestDialog as the imperative API. + requestDialog: function (req) { return kanbanDialogs.request(req); }, }) : null, ), ); @@ -1319,7 +1540,18 @@ if (busy) return; if (action.kind === "cli_hint") { const cmd = (action.payload && action.payload.command) || action.label; - const fallback = function () { window.prompt("Copy this command:", cmd); }; + const fallback = function () { + // The clipboard API is unavailable in this context. The native + // window.prompt is acceptable here because: + // (a) The success path doesn't open a dialog at all (just + // sets `copiedKey` for 2 seconds), and + // (b) the fallback only fires when the browser blocks + // navigator.clipboard, which is rare. + // Documented carve-out — see issue #50547 followups for the + // dedicated copyFallback dialog body that will replace this + // once ConfirmDialog grows a `disabled` prop upstream. + window.prompt("Copy this command:", cmd); + }; try { const p = navigator.clipboard && navigator.clipboard.writeText(cmd); if (p && p.then) { @@ -1913,7 +2145,20 @@ const msg = tx(t, "archiveBoardConfirm", "Archive board '{name}'? It will be moved to boards/_archived/ so you can recover it later. Tasks on this board will no longer appear anywhere in the UI.", { name: currentName }); - if (window.confirm(msg)) props.onDeleteBoard(props.board); + // Prefer the in-app dialog flow if the host wired one. + if (props.requestDialog) { + props.requestDialog({ + kind: "confirm", + title: tx(t, "archiveBoardTitle", "Archive this board"), + description: msg, + confirmLabel: tx(t, "archive", "Archive"), + destructive: true, + }).then(function (r) { + if (r.confirmed) props.onDeleteBoard(props.board); + }).catch(function () { /* cancelled */ }); + } else if (window.confirm(msg)) { + props.onDeleteBoard(props.board); + } }, size: "sm", className: "h-8", @@ -2403,9 +2648,16 @@ const taskId = e.dataTransfer.getData(MIME_TASK); if (!taskId) return; if (props.selectedIds && props.selectedIds.has(taskId) && props.selectedIds.size > 1) { - if (window.confirm(tx(t, "trash.confirmMany", "Permanently delete {n} selected tasks? This cannot be undone.", { n: props.selectedIds.size }))) { - const ids = Array.from(props.selectedIds); - Promise.all(ids.map(function (id) { return props.onDelete(id); })).catch(function () {}); + // Delegate to the bulk-delete path on the parent so we use a + // single in-app confirmation modal. Falling back to the per-id + // onDelete path (which would prompt N times) is preserved for + // hosts that haven't wired onDeleteSelected. + if (props.onDeleteSelected) { + props.onDeleteSelected(props.selectedIds.size); + } else { + Promise.all( + Array.from(props.selectedIds).map(function (id) { return props.onDelete(id); }) + ).catch(function () {}); } } else { props.onDelete(taskId); @@ -2565,6 +2817,7 @@ draggingTaskId: props.draggingTaskId, selectedIds: props.selectedIds, onDelete: props.onDelete, + onDeleteSelected: props.onDeleteSelected, }), ); } @@ -3247,21 +3500,70 @@ .catch(function (e) { setUploadErr(String(e.message || e)); }); }; + // doPatch is invoked by the side-drawer's StatusActions (block / unblock + // / complete / archive), PriorityEditor, AssigneeEditor, etc. Two + // requirements differ from the column-card drag path: + // + // 1. Confirmation: this happens via the in-app dialog flow exposed + // on `props` by the parent (KanbanPage passes a `requestDialog` + // function down). Falls back to a native window.confirm if the + // parent didn't wire one up. + // + // 2. Completion summary for status=done: until ConfirmDialog grows a + // `disabled` prop upstream (see #50547 followups), we keep the + // prompt + alert as a documented carve-out for this single call + // site. The prompt body, validation copy, and requirement are + // unchanged from the pre-migration implementation. const doPatch = function (patch, opts) { + if (opts && opts.confirm && props.requestDialog) { + return props.requestDialog({ + kind: "confirm", + title: opts.confirmTitle || tx(t, "confirmTitle", "Confirm change"), + description: opts.confirm, + confirmLabel: opts.confirmLabel || tx(t, "common.confirm", "Confirm"), + destructive: !!opts.destructive, + }).then(function (r) { + if (!r.confirmed) return null; + return applyPatch(patch); + }); + } if (opts && opts.confirm && !window.confirm(opts.confirm)) { return Promise.resolve(); } - const finalPatch = withCompletionSummary(patch, 1); - if (!finalPatch) return Promise.resolve(); - setPatchErr(null); - return SDK.fetchJSON(withBoard(`${API}/tasks/${encodeURIComponent(props.taskId)}`, boardSlug), { - method: "PATCH", - headers: { "Content-Type": "application/json" }, - body: JSON.stringify(finalPatch), - }).then(function () { load(); props.onRefresh(); }) - .catch(function (e) { setPatchErr(parseApiErrorMessage(e)); }); + return applyPatch(patch); + + function applyPatch(patch) { + const finalPatch = withCompletionSummary(patch); + if (!finalPatch) return Promise.resolve(); + setPatchErr(null); + return SDK.fetchJSON(withBoard(`${API}/tasks/${encodeURIComponent(props.taskId)}`, boardSlug), { + method: "PATCH", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify(finalPatch), + }).then(function () { load(); props.onRefresh(); }) + .catch(function (e) { setPatchErr(parseApiErrorMessage(e)); }); + } }; + // Local completion-summary prompt used only by doPatch above. + // Documented carve-out — see the doPatch comment. + function withCompletionSummary(patch) { + if (!patch || patch.status !== "done") return patch; + const value = window.prompt( + tx(t, "completionSummary", + "Completion summary for this task. This is stored as the task result."), + "", + ); + if (value === null) return null; + const summary = value.trim(); + if (!summary) { + window.alert(tx(t, "completionSummaryRequired", + "Completion summary is required before marking a task done.")); + return null; + } + return Object.assign({}, patch, { result: summary, summary: summary }); + } + // Triage specifier — calls the auxiliary LLM to flesh out a rough // idea in the Triage column into a concrete spec (title + body with // goal, approach, acceptance criteria) and promotes it to todo. @@ -3412,6 +3714,7 @@ props.onClose(); if (props.onOpenTask) props.onOpenTask(taskId); }, + requestDialog: props.requestDialog, }) : null, data ? h("div", { className: "hermes-kanban-drawer-comment-foot" }, h("div", { @@ -3541,7 +3844,18 @@ className: "hermes-kanban-drawer-close", title: tx(i18n, "removeAttachment", "Remove attachment"), onClick: function () { - if (window.confirm(tx(i18n, "confirmRemoveAttachment", + if (props.requestDialog) { + props.requestDialog({ + kind: "confirm", + title: tx(i18n, "removeAttachment", "Remove attachment"), + description: tx(i18n, "confirmRemoveAttachment", + "Remove this attachment?"), + confirmLabel: tx(i18n, "common.delete", "Delete"), + destructive: true, + }).then(function (r) { + if (r.confirmed && props.onDelete) props.onDelete(a.id); + }).catch(function () { /* cancelled */ }); + } else if (window.confirm(tx(i18n, "confirmRemoveAttachment", "Remove this attachment?"))) { if (props.onDelete) props.onDelete(a.id); } @@ -3695,6 +4009,7 @@ uploadBusy: props.uploadBusy, uploadErr: props.uploadErr, i18n: i18n, + requestDialog: props.requestDialog, }), h("div", { className: "hermes-kanban-section" }, h("div", { className: "hermes-kanban-section-head" }, diff --git a/tests/plugins/test_kanban_dashboard_plugin.py b/tests/plugins/test_kanban_dashboard_plugin.py index 324207a4711fe..4ae35cb9463c2 100644 --- a/tests/plugins/test_kanban_dashboard_plugin.py +++ b/tests/plugins/test_kanban_dashboard_plugin.py @@ -655,6 +655,294 @@ def test_bulk_review_assignment_preserves_implementer_provenance(client): assert event.payload["reviewer"] == "reviewer" +def test_bulk_status_done_forwards_completion_summary(client): + a = client.post("/api/plugins/kanban/tasks", json={"title": "a"}).json()["task"] + b = client.post("/api/plugins/kanban/tasks", json={"title": "b"}).json()["task"] + + r = client.post( + "/api/plugins/kanban/tasks/bulk", + json={ + "ids": [a["id"], b["id"]], + "status": "done", + "result": "DECIDED: ship it", + "summary": "DECIDED: ship it", + "metadata": {"source": "dashboard"}, + }, + ) + + assert r.status_code == 200 + assert all(r["ok"] for r in r.json()["results"]) + conn = kb.connect() + try: + for tid in (a["id"], b["id"]): + task = kb.get_task(conn, tid) + run = kb.latest_run(conn, tid) + assert task.status == "done" + assert task.result == "DECIDED: ship it" + assert run.summary == "DECIDED: ship it" + assert run.metadata == {"source": "dashboard"} + finally: + conn.close() + + +def test_bulk_status_running_rejected(client): + """Bulk updates must match single-task PATCH: direct 'running' is invalid.""" + t = client.post("/api/plugins/kanban/tasks", json={"title": "x"}).json()["task"] + + r = client.post( + "/api/plugins/kanban/tasks/bulk", + json={"ids": [t["id"]], "status": "running"}, + ) + + assert r.status_code == 200 + results = r.json()["results"] + assert len(results) == 1 + assert results[0]["id"] == t["id"] + assert results[0]["ok"] is False + assert "running" in results[0]["error"] + + board = client.get("/api/plugins/kanban/board").json() + statuses = { + tt["id"]: col["name"] + for col in board["columns"] + for tt in col["tasks"] + } + assert statuses.get(t["id"]) != "running" + + +def test_dashboard_done_actions_prompt_for_completion_summary(): + """Behavioral coverage for the migrated ``requestDialog`` flow. + + Replaces the prior bundle-string-only assertion (which only proved the + rename landed). The dialog state machine at + ``plugins/kanban/dashboard/dist/index.js`` resolves with + ``{confirmed: true|false, summary?}``. Each migrated call site must + gate the dispatch on the resolved ``confirmed`` flag. This test + asserts that contract at two layers: + + 1. **Bundle cancel guards**: every migrated site gates on ``r.confirmed`` + (or its subscripted alias ``r1.confirmed``/``r2.confirmed``) before + dispatching. We verify by counting the cancel-guard patterns + + cross-referencing against the 8 migrated sites listed in the PR + description. + 2. **Visual affordance**: every destructive ``requestDialog`` call marks + ``destructive: true`` so the host renders the destructive variant. + + The dispatch path itself (PATCH/DELETE actually firing on confirm, not + on cancel) is covered by the backend behavioral tests + ``test_dashboard_confirm_dispatches_expected_*`` and + ``test_dashboard_cancel_keeps_task_in_old_status`` below — together + they pin the contract end-to-end. + """ + + repo_root = Path(__file__).resolve().parents[2] + js = (repo_root / "plugins" / "kanban" / "dashboard" / "dist" / "index.js").read_text() + + import re + + # Match ``if (!r.confirmed)``, ``if (!r1.confirmed)``, ``if (r.confirmed)`` + # (positive-form gate). The bundle uses both polarities: + # - negative ``if (!r.confirmed) return null;`` in dialog flow bodies + # - positive ``if (r.confirmed) props.onDeleteBoard(...);`` in JSX handlers + cancel_guard_pattern = re.compile( + r"if\s*\(\s*!?\s*r\d?\.confirmed\s*\)", + re.IGNORECASE, + ) + guards = cancel_guard_pattern.findall(js) + # 8 migrated sites per the PR description: + # moveTask (1), moveSelected (1), applyBulk (1), deleteTask (1), + # deleteSelected (1), archiveBoard (1), removeAttachment (1), doPatch (1). + # Plus performMoveTask callers (moveTask/moveSelected each have + # ``r1.confirmed`` + ``r2.confirmed`` for the two-stage flow) → up to + # 10 guards. Loose lower bound to avoid brittleness. + assert len(guards) >= 8, ( + f"expected >= 8 `if (r?.confirmed)` cancel guards in bundle (one " + f"per migrated site, plus extras for two-stage flows); found {len(guards)}" + ) + + # Visual affordance: every destructive requestDialog call must mark + # ``destructive: true`` so the host renders the destructive variant. + # deleteTask, deleteSelected, archiveBoard → at least 3. + destructive_call_count = js.count("destructive: true") + assert destructive_call_count >= 3, ( + f"expected >= 3 `destructive: true` requestDialog calls (single " + f"delete, bulk delete, archive-board); found {destructive_call_count}" + ) + + +def test_dashboard_cancel_keeps_task_in_old_status(client): + """Behavioral: the cancel branch of the dispatch path (no PATCH/DELETE + issued) must leave the task in its previous status. The cancel guard + lives in the bundle; this test pins the backend contract that the guard + relies on. + """ + t = client.post("/api/plugins/kanban/tasks", + json={"title": "x"}).json()["task"] + # Tasks land in ``ready`` by default. No PATCH issued — simulating the + # cancel branch in the bundle. + assert t["status"] == "ready" + r = client.get(f"/api/plugins/kanban/tasks/{t['id']}") + assert r.json()["task"]["status"] == "ready" + + +def test_dashboard_confirm_dispatches_expected_patch_body(client): + """Behavioral: the PATCH body shape the bundle produces on confirm + (status + result + summary) must be accepted by the backend without + rejection. The backend stores ``result`` as the human-readable + completion summary (the bundle comments confirm ``summary`` is sent + duplicatively so the backend can store the value under its preferred + key while the wire format remains explicit). + This is the contract the bundle's performMoveTask relies on. + """ + t = client.post("/api/plugins/kanban/tasks", + json={"title": "x"}).json()["task"] + # Bundle's performMoveTask on confirm with a summary produces: + # { status, result: summary, summary: summary } + r = client.patch( + f"/api/plugins/kanban/tasks/{t['id']}", + json={"status": "done", "result": "shipped", "summary": "shipped"}, + ) + assert r.status_code == 200, r.text + body = r.json()["task"] + assert body["status"] == "done" + assert body.get("result") == "shipped" + + +def test_dashboard_confirm_dispatches_expected_delete(client): + """Behavioral: the DELETE call the bundle issues on confirm + (``fetchJSON(`${API}/tasks/${id}`, { method: 'DELETE' })``) must + succeed and remove the task. + """ + t = client.post("/api/plugins/kanban/tasks", + json={"title": "x"}).json()["task"] + r = client.delete(f"/api/plugins/kanban/tasks/{t['id']}") + assert r.status_code == 200, r.text + # 404 on the now-deleted task confirms removal. + r2 = client.get(f"/api/plugins/kanban/tasks/{t['id']}") + assert r2.status_code == 404 + + +def test_dashboard_surfaces_ready_blocked_error_inline(): + """Regression for #26744: failed status transitions must be surfaced + inline, not swallowed. The drag/drop banner and the drawer's action + row each render the parsed API ``detail`` so operators see *why* + their click did nothing. + """ + repo_root = Path(__file__).resolve().parents[2] + bundle = ( + repo_root / "plugins" / "kanban" / "dashboard" / "dist" / "index.js" + ).read_text() + + # Helper that strips ``"409: {\"detail\":\"…\"}"`` down to the + # human-readable message before it lands in any banner. + assert "function parseApiErrorMessage(err)" in bundle + assert "parsed.detail" in bundle + + # Drag/drop banner now uses the parsed message instead of raw + # ``err.message`` so it no longer leaks HTTP plumbing. + assert "setError(tx(t, \"moveFailed\", \"Move failed: \") + parseApiErrorMessage(err))" in bundle + + # Drawer action row has its own visible error surface and clears it + # on success/refresh so stale failures don't follow the operator + # around. + assert "const [patchErr, setPatchErr] = useState(null);" in bundle + assert "setPatchErr(parseApiErrorMessage(e))" in bundle + assert "setPatchErr(null)" in bundle + + +def test_dashboard_dependency_selects_use_value_change_handler(): + """Regression for the dependency selects in the task drawer: the + add-parent / add-child dropdowns must wire through the shared + selectChangeHandler helper so their value actually lands on the + underlying React state. Salvaged from #20019 @LeonSGP43. + """ + repo_root = Path(__file__).resolve().parents[2] + bundle = ( + repo_root / "plugins" / "kanban" / "dashboard" / "dist" / "index.js" + ).read_text() + + parent_select = ( + 'value: newParent,\n' + ' className: "h-7 text-xs flex-1",\n' + ' }, selectChangeHandler(setNewParent))' + ) + child_select = ( + 'value: newChild,\n' + ' className: "h-7 text-xs flex-1",\n' + ' }, selectChangeHandler(setNewChild))' + ) + + assert parent_select in bundle + assert child_select in bundle + + +def test_bulk_archive(client): + a = client.post("/api/plugins/kanban/tasks", json={"title": "a"}).json()["task"] + b = client.post("/api/plugins/kanban/tasks", json={"title": "b"}).json()["task"] + r = client.post("/api/plugins/kanban/tasks/bulk", + json={"ids": [a["id"], b["id"]], "archive": True}) + assert r.status_code == 200 + assert all(r["ok"] for r in r.json()["results"]) + # Default board (archived hidden) — both gone. + board = client.get("/api/plugins/kanban/board").json() + ids = {t["id"] for col in board["columns"] for t in col["tasks"]} + assert a["id"] not in ids + assert b["id"] not in ids + + +def test_bulk_reassign(client): + a = client.post("/api/plugins/kanban/tasks", + json={"title": "a", "assignee": "old"}).json()["task"] + b = client.post("/api/plugins/kanban/tasks", + json={"title": "b", "assignee": "old"}).json()["task"] + r = client.post("/api/plugins/kanban/tasks/bulk", + json={"ids": [a["id"], b["id"]], "assignee": "new"}) + assert r.status_code == 200 + for tid in (a["id"], b["id"]): + t = client.get(f"/api/plugins/kanban/tasks/{tid}").json()["task"] + assert t["assignee"] == "new" + + +def test_bulk_unassign_via_empty_string(client): + a = client.post("/api/plugins/kanban/tasks", + json={"title": "a", "assignee": "x"}).json()["task"] + r = client.post("/api/plugins/kanban/tasks/bulk", + json={"ids": [a["id"]], "assignee": ""}) + assert r.status_code == 200 + t = client.get(f"/api/plugins/kanban/tasks/{a['id']}").json()["task"] + assert t["assignee"] is None + + +def test_bulk_partial_failure_doesnt_abort_siblings(client): + """One bad id in the middle of a batch must not prevent others from + applying.""" + a = client.post("/api/plugins/kanban/tasks", json={"title": "a"}).json()["task"] + c2 = client.post("/api/plugins/kanban/tasks", json={"title": "c"}).json()["task"] + r = client.post("/api/plugins/kanban/tasks/bulk", + json={"ids": [a["id"], "bogus-id", c2["id"]], "priority": 7}) + assert r.status_code == 200 + results = r.json()["results"] + assert len(results) == 3 + ok_ids = {r["id"] for r in results if r["ok"]} + assert a["id"] in ok_ids + assert c2["id"] in ok_ids + assert any(not r["ok"] and r["id"] == "bogus-id" for r in results) + # Good siblings actually got the priority bump. + for tid in (a["id"], c2["id"]): + t = client.get(f"/api/plugins/kanban/tasks/{tid}").json()["task"] + assert t["priority"] == 7 + + +def test_bulk_empty_ids_400(client): + r = client.post("/api/plugins/kanban/tasks/bulk", json={"ids": []}) + assert r.status_code == 400 + + +# --------------------------------------------------------------------------- +# /config endpoint +# --------------------------------------------------------------------------- + + # --------------------------------------------------------------------------- # /config endpoint # --------------------------------------------------------------------------- diff --git a/web/src/i18n/en.ts b/web/src/i18n/en.ts index 9c85ad7b2012b..603e2638a89a0 100644 --- a/web/src/i18n/en.ts +++ b/web/src/i18n/en.ts @@ -827,6 +827,12 @@ export const en: Translations = { "Mark this task as blocked? The worker's claim is released.", confirmScheduled: "Move this task to Scheduled? Use this for known time delays rather than human blockers.", + confirmDoneMany: + "Mark {n} tasks as done? The workers' claims are released and dependent children become ready.", + confirmArchiveMany: + "Archive {n} tasks? They disappear from the default board view.", + confirmBlockedMany: + "Mark {n} tasks as blocked? The workers' claims are released.", completionSummary: "Completion summary for {label}. This is stored as the task result.", completionSummaryRequired: @@ -864,5 +870,9 @@ export const en: Translations = { "Comments reach the worker on its next run or kanban_show() — no need to block the task first.", commentHintTitle: "Comments are the channel for talking to a task's worker. They land on the thread immediately — no need to block the task first. A running worker picks the thread up on its next kanban_show() or respawn; blocking is only for when you want the worker to STOP and wait for your input.", + trash: { + confirmTitle: "Delete task?", + confirmManyTitle: "Delete {n} tasks?", + }, }, }; diff --git a/web/src/i18n/types.ts b/web/src/i18n/types.ts index f3d1d599654fb..0fee160c5bf65 100644 --- a/web/src/i18n/types.ts +++ b/web/src/i18n/types.ts @@ -828,6 +828,9 @@ export interface Translations { confirmArchive: string; confirmBlocked: string; confirmScheduled?: string; + confirmDoneMany: string; + confirmArchiveMany: string; + confirmBlockedMany: string; completionSummary: string; completionSummaryRequired: string; triagePlaceholder: string; @@ -861,5 +864,11 @@ export interface Translations { saving?: string; commentHint?: string; commentHintTitle?: string; + // Optional in-app confirm-dialog strings for the trash/delete flow; + // non-English locales fall back to the English literals in the bundle. + trash?: { + confirmTitle?: string; + confirmManyTitle?: string; + }; }; } diff --git a/web/src/plugins/registry.test.ts b/web/src/plugins/registry.test.ts index 8fe064785824d..a45a318433c9a 100644 --- a/web/src/plugins/registry.test.ts +++ b/web/src/plugins/registry.test.ts @@ -43,12 +43,4 @@ describe("plugin SDK dialog/toast surface", () => { expect(typeof sdk.hooks.useState).toBe("function"); expect(typeof sdk.hooks.useCallback).toBe("function"); }); - - it("does not bump SDK_CONTRACT_VERSION (additive change)", () => { - exposePluginSDK(); - const sdk = (globalThis as any).window.__HERMES_PLUGIN_SDK__; - // Pre-existing version per registry.ts:98. This test fails if a future - // PR accidentally bumps the major for an additive surface change. - expect(sdk.sdkVersion).toBe("1.1.0"); - }); }); \ No newline at end of file From 77f35add0cc4bcbc20f67cbf937b79c8a8dfa4a7 Mon Sep 17 00:00:00 2001 From: YuYigeng <165616139+YuYigeng@users.noreply.github.com> Date: Sun, 26 Jul 2026 23:41:12 +0800 Subject: [PATCH 294/376] fix(tui): honor destructive slash confirmation config --- .../src/__tests__/createSlashHandler.test.ts | 33 +++++++++++++ ui-tui/src/__tests__/useConfigSync.test.ts | 48 +++++++++++++++++++ ui-tui/src/app/interfaces.ts | 1 + ui-tui/src/app/slash/commands/core.ts | 2 +- ui-tui/src/app/uiStore.ts | 1 + ui-tui/src/app/useConfigSync.ts | 5 ++ ui-tui/src/gatewayTypes.ts | 6 +++ 7 files changed, 95 insertions(+), 1 deletion(-) diff --git a/ui-tui/src/__tests__/createSlashHandler.test.ts b/ui-tui/src/__tests__/createSlashHandler.test.ts index 4fa7ff2dca88d..6afd025587472 100644 --- a/ui-tui/src/__tests__/createSlashHandler.test.ts +++ b/ui-tui/src/__tests__/createSlashHandler.test.ts @@ -482,6 +482,19 @@ describe('createSlashHandler', () => { expect(ctx.gateway.rpc).not.toHaveBeenCalled() }) + it.each([ + ['/new sprint planning', 'new session started', 'sprint planning'], + ['/clear', undefined, undefined] + ])('skips the confirmation for %s when config disables it', (command, message, title) => { + patchUiState({ destructiveSlashConfirm: false }) + const ctx = buildCtx() + + expect(createSlashHandler(ctx)(command)).toBe(true) + + expect(getOverlayState().confirm).toBeNull() + expect(ctx.session.newSession).toHaveBeenCalledWith(message, title) + }) + it('routes the /reset catalog alias through the local fresh-session lifecycle', () => { const ctx = buildCtx({ local: { @@ -501,6 +514,26 @@ describe('createSlashHandler', () => { expect(ctx.gateway.gw.request).not.toHaveBeenCalled() }) + it('skips the confirmation for the /reset alias when config disables it', () => { + patchUiState({ destructiveSlashConfirm: false }) + + const ctx = buildCtx({ + local: { + catalog: { + canon: { + '/new': '/new', + '/reset': '/new' + } + } + } + }) + + expect(createSlashHandler(ctx)('/reset')).toBe(true) + + expect(getOverlayState().confirm).toBeNull() + expect(ctx.session.newSession).toHaveBeenCalledWith('new session started', undefined) + }) + it('keeps visible scrollback when branching a TUI session', async () => { patchUiState({ sid: 'sid-parent' }) const rpc = vi.fn(() => Promise.resolve({ session_id: 'sid-branch', title: 'branch title' })) diff --git a/ui-tui/src/__tests__/useConfigSync.test.ts b/ui-tui/src/__tests__/useConfigSync.test.ts index 9191b26d70ba0..569e10ea1fab0 100644 --- a/ui-tui/src/__tests__/useConfigSync.test.ts +++ b/ui-tui/src/__tests__/useConfigSync.test.ts @@ -47,6 +47,54 @@ describe('applyDisplay', () => { expect(s.streaming).toBe(false) }) + it('hydrates the destructive slash confirmation policy from approvals', () => { + const setBell = vi.fn() + + applyDisplay( + { + config: { + approvals: { destructive_slash_confirm: false }, + display: {} + } + }, + setBell + ) + + expect($uiState.get().destructiveSlashConfirm).toBe(false) + + applyDisplay( + { + config: { + approvals: { destructive_slash_confirm: true }, + display: {} + } + }, + setBell + ) + + expect($uiState.get().destructiveSlashConfirm).toBe(true) + }) + + it('defaults destructive slash confirmation on and preserves it across config RPC failure', () => { + const setBell = vi.fn() + + applyDisplay({ config: { display: {} } }, setBell) + expect($uiState.get().destructiveSlashConfirm).toBe(true) + + applyDisplay( + { + config: { + approvals: { destructive_slash_confirm: false }, + display: {} + } + }, + setBell + ) + applyDisplay(null, setBell) + + expect($uiState.get().destructiveSlashConfirm).toBe(false) + }) + it('coerces legacy true + "on" alias to top', () => { const setBell = vi.fn() diff --git a/ui-tui/src/app/interfaces.ts b/ui-tui/src/app/interfaces.ts index 4583ccfe608c0..e8a1a1d9ffa47 100644 --- a/ui-tui/src/app/interfaces.ts +++ b/ui-tui/src/app/interfaces.ts @@ -322,6 +322,7 @@ export interface UiState { busy: boolean busyInputMode: BusyInputMode compact: boolean + destructiveSlashConfirm: boolean detailsMode: DetailsMode detailsModeCommandOverride: boolean // Focus view (/focus) — display-only reduced-output mode. Drives the diff --git a/ui-tui/src/app/slash/commands/core.ts b/ui-tui/src/app/slash/commands/core.ts index 62f64ccaffc2c..794457a16866c 100644 --- a/ui-tui/src/app/slash/commands/core.ts +++ b/ui-tui/src/app/slash/commands/core.ts @@ -199,7 +199,7 @@ export const coreCommands: SlashCommand[] = [ ctx.session.newSession(isNew ? 'new session started' : undefined, requestedTitle || undefined) } - if (NO_CONFIRM_DESTRUCTIVE) { + if (NO_CONFIRM_DESTRUCTIVE || !ctx.ui.destructiveSlashConfirm) { return commit() } diff --git a/ui-tui/src/app/uiStore.ts b/ui-tui/src/app/uiStore.ts index fe7a6674da3b1..e43b38e5013f9 100644 --- a/ui-tui/src/app/uiStore.ts +++ b/ui-tui/src/app/uiStore.ts @@ -14,6 +14,7 @@ const buildUiState = (): UiState => ({ busy: false, busyInputMode: 'queue', compact: false, + destructiveSlashConfirm: true, detailsMode: 'collapsed', detailsModeCommandOverride: false, focusView: false, diff --git a/ui-tui/src/app/useConfigSync.ts b/ui-tui/src/app/useConfigSync.ts index e8dd8b1334c6a..9509f9b44d41d 100644 --- a/ui-tui/src/app/useConfigSync.ts +++ b/ui-tui/src/app/useConfigSync.ts @@ -253,6 +253,7 @@ export const applyDisplay = ( setVoiceRecordKey?: (v: ParsedVoiceRecordKey) => void ) => { const d = cfg?.config?.display ?? {} + const approvals = cfg?.config?.approvals setBell(!!d.bell_on_complete) @@ -273,6 +274,10 @@ export const applyDisplay = ( battery: !!d.battery, busyInputMode: normalizeBusyInputMode(d.busy_input_mode), compact: !!d.tui_compact, + // Fail safe: only YAML boolean false disables the prompt. A transient + // config RPC failure (cfg=null) preserves the last known policy instead + // of silently changing approval behavior until the next successful poll. + ...(cfg ? { destructiveSlashConfirm: approvals?.destructive_slash_confirm !== false } : {}), detailsMode: resolveDetailsMode(d), detailsModeCommandOverride: false, focusView: !!d.focus_view, diff --git a/ui-tui/src/gatewayTypes.ts b/ui-tui/src/gatewayTypes.ts index d5ab544ab5484..a701b45fd55b0 100644 --- a/ui-tui/src/gatewayTypes.ts +++ b/ui-tui/src/gatewayTypes.ts @@ -119,8 +119,14 @@ export interface ConfigVoiceConfig { submit_mode?: unknown } +export interface ConfigApprovalsConfig { + // Raw config value: only the explicit boolean false disables the safety gate. + destructive_slash_confirm?: unknown +} + export interface ConfigFullResponse { config?: { + approvals?: ConfigApprovalsConfig display?: ConfigDisplayConfig voice?: ConfigVoiceConfig paste_collapse_threshold?: number From 7d34d7d8d19249091466d9218f0adce1b71563dd Mon Sep 17 00:00:00 2001 From: YuYigeng <165616139+YuYigeng@users.noreply.github.com> Date: Fri, 31 Jul 2026 00:11:06 +0800 Subject: [PATCH 295/376] docs(tui): clarify destructive confirm controls --- hermes_cli/config_defaults.py | 5 +++-- website/docs/user-guide/security.md | 2 +- 2 files changed, 4 insertions(+), 3 deletions(-) diff --git a/hermes_cli/config_defaults.py b/hermes_cli/config_defaults.py index 0f29f5059efb4..c4bc6386b0752 100644 --- a/hermes_cli/config_defaults.py +++ b/hermes_cli/config_defaults.py @@ -2208,8 +2208,9 @@ # through tools.slash_confirm — native yes/no buttons on Telegram, # Discord, and Slack; text fallback elsewhere. Users click "Always # Approve" to silence the prompt permanently; that flips this key to - # false. TUI has its own modal overlay (HERMES_TUI_NO_CONFIRM=1 to - # opt out there). + # false. TUI also honors this setting for its /clear, /new, and /reset + # modal; HERMES_TUI_NO_CONFIRM=1 force-skips that modal regardless of + # the configured value. "destructive_slash_confirm": True, }, diff --git a/website/docs/user-guide/security.md b/website/docs/user-guide/security.md index 9818ffa9f4f0f..d4bc6b3cd468f 100644 --- a/website/docs/user-guide/security.md +++ b/website/docs/user-guide/security.md @@ -46,7 +46,7 @@ The full set of keys: | `timeout` | `300` | Seconds Hermes waits for an approval reply before timing out. | | `cron_mode` | `deny` | How [cron jobs](./features/cron.md) behave headlessly when they trigger a dangerous-command prompt. `deny` blocks the command (the agent must find another path); `approve` auto-approves everything in cron context. | | `mcp_reload_confirm` | `true` | When true, `/reload-mcp` asks before rebuilding the MCP tool set. Rebuilding invalidates the provider prompt cache (tool schemas live in the system prompt), so the next message re-sends full input tokens. Users who click **Always Approve** flip this key to `false`. | -| `destructive_slash_confirm` | `true` | When true, destructive session slash commands (`/clear`, `/new`, `/reset`, `/undo`) prompt before discarding conversation state. Three-option dialog (Approve Once / Always Approve / Cancel) routed through native yes/no buttons on Telegram, Discord, and Slack; text fallback elsewhere. Users who click **Always Approve** flip this key to `false`. TUI uses its own modal overlay (set `HERMES_TUI_NO_CONFIRM=1` to opt out there). | +| `destructive_slash_confirm` | `true` | When true, destructive session slash commands (`/clear`, `/new`, `/reset`, `/undo`) prompt before discarding conversation state. Three-option dialog (Approve Once / Always Approve / Cancel) routed through native yes/no buttons on Telegram, Discord, and Slack; text fallback elsewhere. Users who click **Always Approve** flip this key to `false`. The TUI also honors this setting for its `/clear`, `/new`, and `/reset` modal; `HERMES_TUI_NO_CONFIRM=1` force-skips that modal regardless of the configured value. | | Mode | Behavior | |------|----------| From 8e35ff0a628e38dbfab0483c5e0c2cdd60da8baa Mon Sep 17 00:00:00 2001 From: PRATHAMESH75 Date: Tue, 28 Jul 2026 18:34:20 +0530 Subject: [PATCH 296/376] fix(desktop): confirm before clearing the entire Enabled Toolsets list (#73319) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Config settings auto-save on a 550ms debounce with no undo. The 'Enabled Toolsets' list is rendered by the generic ConfigField with no destructive-change guard, so a stray select-all + Backspace (or any edit that empties the list) is persisted the moment Settings closes — silently disabling memory, terminal, web search, delegation, and most tools. Recovery required CLI intervention. Guard the one destructive transition: when the enabled-toolsets list goes from non-empty to empty, window.confirm() before applying it (the same pattern env-var removal already uses in toolset-config-panel.tsx). Every other edit passes through untouched. The decision is a pure helper (clearsEnabledToolsets) so it is unit tested directly rather than through a full settings render. New i18n key toolsetsWipeConfirm added to en + zh; partial locales inherit the English string via defineLocale fallback. Fixes #73319 --- .../src/app/settings/config-settings.tsx | 11 ++++- apps/desktop/src/app/settings/helpers.test.ts | 41 +++++++++++++++++++ apps/desktop/src/app/settings/helpers.ts | 25 +++++++++++ apps/desktop/src/i18n/en.ts | 2 + apps/desktop/src/i18n/types.ts | 1 + apps/desktop/src/i18n/zh.ts | 1 + 6 files changed, 80 insertions(+), 1 deletion(-) diff --git a/apps/desktop/src/app/settings/config-settings.tsx b/apps/desktop/src/app/settings/config-settings.tsx index 45e0ff5f15650..1dc4735797375 100644 --- a/apps/desktop/src/app/settings/config-settings.tsx +++ b/apps/desktop/src/app/settings/config-settings.tsx @@ -28,7 +28,7 @@ import { useOnProfileSwitch } from '../hooks/use-on-profile-switch' import { PanelEmpty } from '../overlays/panel' import { ConfigField } from './config-field' -import { enumOptionsFor, getNested, isExternalMemoryProvider, sectionFieldEntries, setNested } from './helpers' +import { clearsEnabledToolsets, enumOptionsFor, getNested, isExternalMemoryProvider, sectionFieldEntries, setNested } from './helpers' import { MemoryConnect } from './memory/connect' import { ProviderConfigPanel } from './memory/provider-config-panel' import { ModelSettings, ModelSettingsSkeleton } from './model-settings' @@ -183,6 +183,15 @@ export function ConfigSettings({ }, [config, onConfigSaved, saveVersion]) const updateConfig = (next: HermesConfigRecord) => { + // Guard the single most destructive config edit: clearing the entire + // "Enabled Toolsets" list silently disables memory, terminal, web search, + // delegation, and most tools, and a stray select-all + Backspace can do it. + // Auto-save is debounced with no undo, so confirm a non-empty → empty + // transition before applying it. Every other edit passes through untouched. + if (config && clearsEnabledToolsets(config, next) && !window.confirm(c.toolsetsWipeConfirm)) { + return + } + saveVersionRef.current += 1 setConfig(next) setSaveVersion(saveVersionRef.current) diff --git a/apps/desktop/src/app/settings/helpers.test.ts b/apps/desktop/src/app/settings/helpers.test.ts index cd0a455b3cc75..8e7a300038fb1 100644 --- a/apps/desktop/src/app/settings/helpers.test.ts +++ b/apps/desktop/src/app/settings/helpers.test.ts @@ -5,6 +5,7 @@ import type { HermesConfigRecord } from '@/types/hermes' import { FIELD_DESCRIPTIONS, FIELD_LABELS, SECTIONS } from './constants' import { defineFieldCopy, fieldCopyForSchemaKey, schemaKeyToFieldCopyKey } from './field-copy' import { + clearsEnabledToolsets, enumOptionsFor, getNested, isExternalMemoryProvider, @@ -363,4 +364,44 @@ describe('settings helpers', () => { expect(sectionFieldEntries({}, {}).get('memory') ?? []).toHaveLength(0) }) }) + + describe('clearsEnabledToolsets', () => { + it('flags a non-empty → empty transition', () => { + const prev: HermesConfigRecord = { toolsets: ['memory', 'terminal', 'web_search'] } + const next: HermesConfigRecord = { toolsets: [] } + + expect(clearsEnabledToolsets(prev, next)).toBe(true) + }) + + it('does not flag a non-empty → missing transition (deep-merge preserves the key)', () => { + // PUT /api/config deep-merges the override onto the stored config, so an + // import that omits `toolsets` keeps the existing list — no wipe happens, + // so there is nothing to confirm. + const prev: HermesConfigRecord = { toolsets: ['memory'] } + const next: HermesConfigRecord = {} + + expect(clearsEnabledToolsets(prev, next)).toBe(false) + }) + + it('does not flag when at least one toolset remains', () => { + const prev: HermesConfigRecord = { toolsets: ['memory', 'terminal'] } + const next: HermesConfigRecord = { toolsets: ['memory'] } + + expect(clearsEnabledToolsets(prev, next)).toBe(false) + }) + + it('does not flag when the list was already empty', () => { + const prev: HermesConfigRecord = { toolsets: [] } + const next: HermesConfigRecord = { toolsets: [] } + + expect(clearsEnabledToolsets(prev, next)).toBe(false) + }) + + it('does not flag an unrelated edit that never touched toolsets', () => { + const prev: HermesConfigRecord = { model: 'a', toolsets: ['memory'] } + const next: HermesConfigRecord = { model: 'b', toolsets: ['memory'] } + + expect(clearsEnabledToolsets(prev, next)).toBe(false) + }) + }) }) diff --git a/apps/desktop/src/app/settings/helpers.ts b/apps/desktop/src/app/settings/helpers.ts index e3f0fe14e2a21..8806e5e421b5a 100644 --- a/apps/desktop/src/app/settings/helpers.ts +++ b/apps/desktop/src/app/settings/helpers.ts @@ -97,6 +97,31 @@ export function getNested(obj: HermesConfigRecord, path: string): unknown { return cur } +/** + * True when an edit clears the entire "Enabled Toolsets" list — i.e. the + * previous config had a non-empty toolsets array and the next one is an + * explicit empty array. + * + * A *missing* toolsets key is deliberately NOT a clear: `PUT /api/config` + * deep-merges the override onto the stored config (`_deep_merge` preserves base + * keys absent from the override), so an import that omits `toolsets` leaves the + * existing toolsets intact. Prompting there would warn about a wipe that never + * happens. Only an explicit `[]` actually empties the list. + * + * Clearing every toolset silently disables memory, terminal, web search, + * delegation, and most tools, and config auto-saves with no undo, so callers + * use this to confirm the destructive transition before applying it. Any edit + * that keeps at least one toolset — or that never had one — returns false. + */ +export function clearsEnabledToolsets(prev: HermesConfigRecord, next: HermesConfigRecord): boolean { + const prevToolsets = getNested(prev, 'toolsets') + const nextToolsets = getNested(next, 'toolsets') + const hadToolsets = Array.isArray(prevToolsets) && prevToolsets.length > 0 + const clearsToolsets = Array.isArray(nextToolsets) && nextToolsets.length === 0 + + return hadToolsets && clearsToolsets +} + export function inferFieldSchema(value: unknown): ConfigFieldSchema { if (typeof value === 'boolean') { return { type: 'boolean' } diff --git a/apps/desktop/src/i18n/en.ts b/apps/desktop/src/i18n/en.ts index dfe0374cc5f1e..1e836201ec67c 100644 --- a/apps/desktop/src/i18n/en.ts +++ b/apps/desktop/src/i18n/en.ts @@ -584,6 +584,8 @@ export const en: Translations = { autosaveFailed: 'Autosave failed', imported: 'Config imported', invalidJson: 'Invalid config JSON', + toolsetsWipeConfirm: + 'Remove all enabled toolsets? This disables memory, terminal, web search, delegation, and most other tools until you re-enable them.', keepAwakeTitle: 'Keep computer awake', keepAwakeDesc: 'Stop this machine from sleeping so long or overnight runs keep going. The display can still dim.', attachmentSizeTitle: 'Max preview / image load size', diff --git a/apps/desktop/src/i18n/types.ts b/apps/desktop/src/i18n/types.ts index 68eab864c1822..ce2066bedae70 100644 --- a/apps/desktop/src/i18n/types.ts +++ b/apps/desktop/src/i18n/types.ts @@ -480,6 +480,7 @@ export interface Translations { autosaveFailed: string imported: string invalidJson: string + toolsetsWipeConfirm: string keepAwakeTitle: string keepAwakeDesc: string attachmentSizeTitle: string diff --git a/apps/desktop/src/i18n/zh.ts b/apps/desktop/src/i18n/zh.ts index 2f720b190e2c2..b133f1c6b35ef 100644 --- a/apps/desktop/src/i18n/zh.ts +++ b/apps/desktop/src/i18n/zh.ts @@ -792,6 +792,7 @@ export const zh: Translations = { autosaveFailed: '自动保存失败', imported: '配置已导入', invalidJson: '配置 JSON 无效', + toolsetsWipeConfirm: '确定移除所有已启用的工具集吗?这将禁用记忆、终端、网络搜索、委派以及大多数其他工具,直到你重新启用它们。', keepAwakeTitle: '保持电脑唤醒', keepAwakeDesc: '阻止本机休眠,让长时间或通宵运行继续进行。屏幕仍可变暗。', attachmentSizeTitle: '预览 / 图片加载大小上限', From e6708af1f23821e7b9c962d94745b362bfb343d8 Mon Sep 17 00:00:00 2001 From: Gorde Minchel Date: Thu, 6 Aug 2026 15:28:03 +0800 Subject: [PATCH 297/376] feat(desktop): confirm before deleting a session MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Deleting a session in the desktop app fired instantly on click — the CLI path (hermes sessions delete) asks y/N by default, so one misclick (Archive and Delete sit right next to each other) permanently destroyed a conversation with no dialog and no undo (#61470). Route every delete entry point (sidebar rows, tab menus, the chat header, context menus — all share useSessionActions) through a shared DeleteSessionDialog built on ConfirmDialog. ConfirmDialog gains an onOpenAutoFocus prop so dialogs with no input keep focus off the close button (a11y). Tests: menu delete now asks; cancel keeps the session; Enter confirms; Escape cancels; delete item disabled without onDelete; the same guard applies via SessionContextMenu. --- .../sidebar/session-actions-menu.test.tsx | 132 +++++++++++++++++- .../app/chat/sidebar/session-actions-menu.tsx | 65 ++++++++- .../src/components/ui/confirm-dialog.tsx | 8 +- apps/desktop/src/i18n/ar.ts | 4 + apps/desktop/src/i18n/en.ts | 4 + apps/desktop/src/i18n/ja.ts | 4 + apps/desktop/src/i18n/types.ts | 4 + apps/desktop/src/i18n/zh-hant.ts | 4 + apps/desktop/src/i18n/zh.ts | 4 + 9 files changed, 221 insertions(+), 8 deletions(-) diff --git a/apps/desktop/src/app/chat/sidebar/session-actions-menu.test.tsx b/apps/desktop/src/app/chat/sidebar/session-actions-menu.test.tsx index 82faefb85bf77..45ee1221e86a8 100644 --- a/apps/desktop/src/app/chat/sidebar/session-actions-menu.test.tsx +++ b/apps/desktop/src/app/chat/sidebar/session-actions-menu.test.tsx @@ -2,7 +2,7 @@ import { cleanup, fireEvent, render, screen, waitFor, within } from '@testing-li import { atom } from 'nanostores' import { afterEach, describe, expect, it, vi } from 'vitest' -import { SessionActionsMenu } from './session-actions-menu' +import { SessionActionsMenu, SessionContextMenu } from './session-actions-menu' afterEach(cleanup) @@ -20,7 +20,16 @@ vi.mock('@/hermes', () => ({ renameSession: vi.fn() })) vi.mock('@/i18n', () => ({ useI18n: () => ({ t: { - common: { cancel: 'Cancel', close: 'Close', delete: 'Delete', save: 'Save' }, + common: { + cancel: 'Cancel', + close: 'Close', + confirm: 'Confirm', + delete: 'Delete', + done: 'Done', + loading: 'Loading…', + save: 'Save' + }, + errors: { genericFailure: 'Something went wrong' }, sidebar: { projects: { menuAppearance: 'Appearance', @@ -35,6 +44,10 @@ vi.mock('@/i18n', () => ({ branchFrom: 'Branch from here', copyId: 'Copy ID', copyIdFailed: 'Failed to copy ID', + deleteDesc: (title: string) => `Delete ${title}?`, + deleteTitle: 'Delete session?', + deleting: 'Deleting…', + deleted: 'Session deleted', export: 'Export', hideTabBar: 'Hide tab bar', pin: 'Pin', @@ -140,4 +153,119 @@ describe('SessionActionsMenu', () => { // eslint-disable-next-line no-restricted-globals -- asserting real focus requires the live document expect(document.activeElement).not.toBe(trigger) }) + + it('confirms before deleting — cancel keeps the session, confirm deletes it', async () => { + const onDelete = vi.fn() + render( + + + + ) + + const trigger = screen.getByRole('button', { name: 'Session actions' }) + fireEvent.pointerDown(trigger, { button: 0, pointerType: 'mouse' }) + fireEvent.pointerUp(trigger, { button: 0, pointerType: 'mouse' }) + fireEvent.click(trigger) + + const deleteItem = await screen.findByRole('menuitem', { name: /delete/i }) + fireEvent.click(deleteItem) + + // The confirm dialog is up and names the session being deleted. + expect(await screen.findByRole('dialog')).toBeTruthy() + expect(screen.getByText(/My session/)).toBeTruthy() + + // Cancel: nothing is deleted. + fireEvent.click(screen.getByRole('button', { name: 'Cancel' })) + expect(onDelete).not.toHaveBeenCalled() + + // Re-open the menu and confirm: only now does the delete call fire. + fireEvent.pointerDown(trigger, { button: 0, pointerType: 'mouse' }) + fireEvent.pointerUp(trigger, { button: 0, pointerType: 'mouse' }) + fireEvent.click(trigger) + const deleteItemAgain = await screen.findByRole('menuitem', { name: /delete/i }) + fireEvent.click(deleteItemAgain) + + expect(await screen.findByRole('dialog')).toBeTruthy() + fireEvent.click(screen.getByRole('button', { name: 'Delete' })) + // ConfirmDialog shows a done beat before auto-closing (600ms); awaiting it + // also drains the async run() update inside act(). + expect(await screen.findByText('Session deleted')).toBeTruthy() + expect(onDelete).toHaveBeenCalledTimes(1) + }) + + it('disables the delete item when no onDelete is provided', async () => { + render( + + + + ) + + const trigger = screen.getByRole('button', { name: 'Session actions' }) + fireEvent.pointerDown(trigger, { button: 0, pointerType: 'mouse' }) + fireEvent.pointerUp(trigger, { button: 0, pointerType: 'mouse' }) + fireEvent.click(trigger) + + const deleteItem = await screen.findByRole('menuitem', { name: /delete/i }) + expect(deleteItem.getAttribute('aria-disabled')).toBe('true') + }) + + it('confirms with the Enter key and cancels with Escape', async () => { + const onDelete = vi.fn() + render( + + + + ) + + const trigger = screen.getByRole('button', { name: 'Session actions' }) + fireEvent.pointerDown(trigger, { button: 0, pointerType: 'mouse' }) + fireEvent.pointerUp(trigger, { button: 0, pointerType: 'mouse' }) + fireEvent.click(trigger) + fireEvent.click(await screen.findByRole('menuitem', { name: /delete/i })) + + const dialog = await screen.findByRole('dialog') + expect(dialog).toBeTruthy() + + // Escape cancels: dialog closes, nothing is deleted. + fireEvent.keyDown(window.document, { key: 'Escape' }) + expect(await screen.queryByRole('dialog')).toBeNull() + expect(onDelete).not.toHaveBeenCalled() + + // Re-open and confirm with Enter: the delete call fires. + fireEvent.pointerDown(trigger, { button: 0, pointerType: 'mouse' }) + fireEvent.pointerUp(trigger, { button: 0, pointerType: 'mouse' }) + fireEvent.click(trigger) + fireEvent.click(await screen.findByRole('menuitem', { name: /delete/i })) + fireEvent.keyDown(await screen.findByRole('dialog'), { key: 'Enter' }) + + expect(await screen.findByText('Session deleted')).toBeTruthy() + expect(onDelete).toHaveBeenCalledTimes(1) + }) + + it('routes the same confirm guard through the context menu', async () => { + const onDelete = vi.fn() + render( + + + + ) + + const row = screen.getByRole('button', { name: 'Session row' }) + fireEvent.contextMenu(row) + + fireEvent.click(await screen.findByRole('menuitem', { name: /delete/i })) + expect(await screen.findByRole('dialog')).toBeTruthy() + + fireEvent.click(screen.getByRole('button', { name: 'Delete' })) + expect(await screen.findByText('Session deleted')).toBeTruthy() + expect(onDelete).toHaveBeenCalledTimes(1) + }) }) diff --git a/apps/desktop/src/app/chat/sidebar/session-actions-menu.tsx b/apps/desktop/src/app/chat/sidebar/session-actions-menu.tsx index 9a0e406f4159a..3a9e5b8c10c3d 100644 --- a/apps/desktop/src/app/chat/sidebar/session-actions-menu.tsx +++ b/apps/desktop/src/app/chat/sidebar/session-actions-menu.tsx @@ -20,8 +20,9 @@ import { import { Button } from '@/components/ui/button' import { Codicon } from '@/components/ui/codicon' import { ColorSwatches } from '@/components/ui/color-swatches' +import { ConfirmDialog } from '@/components/ui/confirm-dialog' import { CopyButton } from '@/components/ui/copy-button' -import { Dialog, DialogContent, DialogFooter, DialogHeader, DialogTitle } from '@/components/ui/dialog' +import { Dialog, DialogContent, DialogFooter, DialogHeader, DialogTitle, preventCloseButtonAutoFocus } from '@/components/ui/dialog' import { Input } from '@/components/ui/input' import { renameSession } from '@/hermes' import { useI18n } from '@/i18n' @@ -198,6 +199,7 @@ function useSessionActions({ // action leaves the restore alone (it's the correct behavior for them). Mirrors // the project menu's appearance-popover guard. const suppressCloseFocusRef = useRef(false) + const [deleteOpen, setDeleteOpen] = useState(false) const tiles = useStore($sessionTiles) const selectedStoredSessionId = useStore($selectedStoredSessionId) const isRemote = useStore($connection)?.mode === 'remote' @@ -399,7 +401,15 @@ function useSessionActions({ label: t.common.delete, onSelect: () => { triggerHaptic('warning') - onDelete?.() + + // Deleting is irreversible (the CLI path asks y/N; the desktop used to + // fire instantly on click). Gate it behind an explicit confirm — see + // #61470. The dialog owns the delete call, so every surface that routes + // through this menu (sidebar rows, tab menus, the chat header) gets the + // guard for free. + if (onDelete) { + setDeleteOpen(true) + } }, variant: 'destructive' } @@ -484,7 +494,50 @@ function useSessionActions({ } } - return { onCloseAutoFocus, renameDialog, renderItems } + const deleteDialog = ( + { + onDelete?.() + }} + onOpenChange={setDeleteOpen} + open={deleteOpen} + sessionTitle={title} + /> + ) + + return { deleteDialog, onCloseAutoFocus, renameDialog, renderItems } +} + +interface DeleteSessionDialogProps { + open: boolean + onOpenChange: (open: boolean) => void + onConfirm: () => void + sessionTitle: string +} + +// Thin wrapper over ConfirmDialog — the single choke point for every session +// delete entry point (sidebar rows, tab menus, the chat header). Deleting a +// session is irreversible and the desktop used to fire it instantly on click +// (#61470); this mirrors the CLI's y/N guard. onConfirm is the fire-and-forget +// delete call; ConfirmDialog owns the busy/done beat and Enter-to-confirm. +function DeleteSessionDialog({ open, onOpenChange, onConfirm, sessionTitle }: DeleteSessionDialogProps) { + const { t } = useI18n() + const r = t.sidebar.row + + return ( + onOpenChange(false)} + onConfirm={onConfirm} + onOpenAutoFocus={preventCloseButtonAutoFocus} + open={open} + title={r.deleteTitle} + /> + ) } interface SessionActionsMenuProps @@ -494,7 +547,7 @@ interface SessionActionsMenuProps export function SessionActionsMenu({ children, align = 'end', sideOffset = 6, ...actions }: SessionActionsMenuProps) { const { t } = useI18n() - const { onCloseAutoFocus, renameDialog, renderItems } = useSessionActions(actions) + const { deleteDialog, onCloseAutoFocus, renameDialog, renderItems } = useSessionActions(actions) return ( <> @@ -509,6 +562,7 @@ export function SessionActionsMenu({ children, align = 'end', sideOffset = 6, .. {children} {renameDialog} + {deleteDialog} ) } @@ -519,7 +573,7 @@ interface SessionContextMenuProps extends SessionActions { export function SessionContextMenu({ children, ...actions }: SessionContextMenuProps) { const { t } = useI18n() - const { onCloseAutoFocus, renameDialog, renderItems } = useSessionActions(actions) + const { deleteDialog, onCloseAutoFocus, renameDialog, renderItems } = useSessionActions(actions) return ( <> @@ -532,6 +586,7 @@ export function SessionContextMenu({ children, ...actions }: SessionContextMenuP {children} {renameDialog} + {deleteDialog} ) } diff --git a/apps/desktop/src/components/ui/confirm-dialog.tsx b/apps/desktop/src/components/ui/confirm-dialog.tsx index becb958a986fa..d4d5fc327defe 100644 --- a/apps/desktop/src/components/ui/confirm-dialog.tsx +++ b/apps/desktop/src/components/ui/confirm-dialog.tsx @@ -28,6 +28,10 @@ interface ConfirmDialogProps { destructive?: boolean /** Close as soon as onConfirm resolves — for optimistic actions that finish in the background. */ dismissOnConfirm?: boolean + /** Focus control for dialogs with no input. Pass `preventCloseButtonAutoFocus` + * so opening doesn't land focus on the close/cancel button (which would pop + * its tooltip with no pointer near it). */ + onOpenAutoFocus?: (event: Event) => void } // Shared confirmation dialog: Enter confirms (from anywhere in the dialog), @@ -44,7 +48,8 @@ export function ConfirmDialog({ doneLabel, cancelLabel, destructive = false, - dismissOnConfirm = false + dismissOnConfirm = false, + onOpenAutoFocus }: ConfirmDialogProps) { const { t } = useI18n() const [status, setStatus] = useState<'done' | 'idle' | 'saving'>('idle') @@ -104,6 +109,7 @@ export function ConfirmDialog({ void run() } }} + onOpenAutoFocus={onOpenAutoFocus} > {title} diff --git a/apps/desktop/src/i18n/ar.ts b/apps/desktop/src/i18n/ar.ts index 1c9db2719372f..1397b6f4aa208 100644 --- a/apps/desktop/src/i18n/ar.ts +++ b/apps/desktop/src/i18n/ar.ts @@ -1668,6 +1668,10 @@ export const ar = defineLocale({ renameTitle: 'إعادة تسمية الجلسة', renameDesc: '', untitledPlaceholder: 'جلسة بلا عنوان', + deleteTitle: 'حذف الجلسة؟', + deleteDesc: title => `سيتم حذف «${title}» نهائيًا. لا يمكن التراجع عن هذا الإجراء.`, + deleting: 'جارٍ الحذف…', + deleted: 'تم حذف الجلسة', ageNow: 'الآن', ageDay: 'يوم', ageHour: 'ساعة', diff --git a/apps/desktop/src/i18n/en.ts b/apps/desktop/src/i18n/en.ts index 1e836201ec67c..f5f021424c04c 100644 --- a/apps/desktop/src/i18n/en.ts +++ b/apps/desktop/src/i18n/en.ts @@ -2040,6 +2040,10 @@ export const en: Translations = { renameTitle: 'Rename session', renameDesc: 'Leave empty to clear.', untitledPlaceholder: 'Untitled session', + deleteTitle: 'Delete session?', + deleteDesc: title => `This will permanently delete “${title}”. This cannot be undone.`, + deleting: 'Deleting…', + deleted: 'Session deleted', untitledChat: id => `Chat ${id}`, messageCount: count => `${count} ${count === 1 ? 'message' : 'messages'}`, todoProgress: 'Tasks completed', diff --git a/apps/desktop/src/i18n/ja.ts b/apps/desktop/src/i18n/ja.ts index a59bed41c2f68..db3799c83b2da 100644 --- a/apps/desktop/src/i18n/ja.ts +++ b/apps/desktop/src/i18n/ja.ts @@ -1811,6 +1811,10 @@ export const ja = defineLocale({ renameTitle: 'セッションの名前を変更', renameDesc: '空欄にするとクリアされます。', untitledPlaceholder: '無題のセッション', + deleteTitle: 'セッションを削除しますか?', + deleteDesc: title => `「${title}」を完全に削除します。この操作は元に戻せません。`, + deleting: '削除中…', + deleted: 'セッションを削除しました', untitledChat: id => `セッション ${id}`, ageNow: 'たった今', ageDay: '日', diff --git a/apps/desktop/src/i18n/types.ts b/apps/desktop/src/i18n/types.ts index ce2066bedae70..ac803a37821c6 100644 --- a/apps/desktop/src/i18n/types.ts +++ b/apps/desktop/src/i18n/types.ts @@ -1723,6 +1723,10 @@ export interface Translations { renameTitle: string renameDesc: string untitledPlaceholder: string + deleteTitle: string + deleteDesc: (title: string) => string + deleting: string + deleted: string untitledChat: (id: string) => string messageCount: (count: number) => string todoProgress: string diff --git a/apps/desktop/src/i18n/zh-hant.ts b/apps/desktop/src/i18n/zh-hant.ts index 8efe69eba864d..d3100740451a8 100644 --- a/apps/desktop/src/i18n/zh-hant.ts +++ b/apps/desktop/src/i18n/zh-hant.ts @@ -1753,6 +1753,10 @@ export const zhHant = defineLocale({ renameTitle: '重新命名工作階段', renameDesc: '留空則清除。', untitledPlaceholder: '未命名工作階段', + deleteTitle: '刪除會話?', + deleteDesc: title => `這將永久刪除「${title}」,且無法復原。`, + deleting: '正在刪除…', + deleted: '會話已刪除', untitledChat: id => `工作階段 ${id}`, ageNow: '剛才', ageDay: '天', diff --git a/apps/desktop/src/i18n/zh.ts b/apps/desktop/src/i18n/zh.ts index b133f1c6b35ef..a1b03f9e450c2 100644 --- a/apps/desktop/src/i18n/zh.ts +++ b/apps/desktop/src/i18n/zh.ts @@ -2226,6 +2226,10 @@ export const zh: Translations = { renameTitle: '重命名会话', renameDesc: '留空则清除。', untitledPlaceholder: '无标题会话', + deleteTitle: '删除会话?', + deleteDesc: title => `这将永久删除“${title}”,且无法撤销。`, + deleting: '正在删除…', + deleted: '会话已删除', untitledChat: id => `会话 ${id}`, messageCount: count => `${count} 条消息`, todoProgress: '任务完成度', From 4c24629bc91e09ec8ed3f87d5e6019d11d017d48 Mon Sep 17 00:00:00 2001 From: briandevans <252620095+briandevans@users.noreply.github.com> Date: Sat, 8 Aug 2026 08:29:18 -0700 Subject: [PATCH 298/376] fix(console): skip the checkpoints prune confirmation the console already took MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Hermes Console registers `checkpoints prune`, `clear` and `clear-legacy` as mutating, so it takes a console-level confirmation before dispatching any of them. `_apply_confirmed_defaults` then exists to keep the CLI layer from asking a second time — its docstring says so — but it only force-defaults `clear` and `clear-legacy`. `prune` was left out, even though `cmd_prune` gates its orphan preview on the identical `not args.force` shape. `_capture_output` redirects stdout and stderr but never stdin, so the unskipped `_confirm()` call hits `input()` with no terminal behind it: `EOFError` propagates into `_confirm`, which returns False, and `cmd_prune` prints "Aborted." and returns 1. The console turns that non-zero exit into a ConsoleCommandError, so `checkpoints prune` fails outright for any user who has at least one orphan checkpoint project — after that user already confirmed. When the server does happen to inherit a foreground terminal, the same call instead blocks a console worker thread and eats the operator's keystrokes. Forcing the flag is the documented behavior here rather than a weakening of the recent orphan-allowlist hardening. `orphan_allowlist` binds a deletion to the identities shown in the preview, guarding the window where a workdir disappears while the command waits on `input()`. Under the console there is no preview and no wait, which is exactly the `--force` case the comment on `cmd_prune` describes as "no restriction". --- hermes_cli/console_engine.py | 5 +- tests/hermes_cli/test_console_engine.py | 82 +++++++++++++++++++++++++ 2 files changed, 86 insertions(+), 1 deletion(-) diff --git a/hermes_cli/console_engine.py b/hermes_cli/console_engine.py index 10104b4ece12e..ab644297c0f2c 100644 --- a/hermes_cli/console_engine.py +++ b/hermes_cli/console_engine.py @@ -1254,7 +1254,10 @@ def _apply_confirmed_defaults(args: argparse.Namespace) -> None: setattr(args, attr, True) if getattr(args, "_console_command", None) == "import": setattr(args, "force", True) - if getattr(args, "checkpoints_command", None) in {"clear", "clear-legacy"}: + # Every checkpoints subcommand the console registers as mutating gates its + # own confirmation on --force, so all three belong here. `prune` reaches + # _confirm() for its orphan preview, and the console never redirects stdin. + if getattr(args, "checkpoints_command", None) in {"prune", "clear", "clear-legacy"}: setattr(args, "force", True) if getattr(args, "plugins_action", None) == "install": if not getattr(args, "enable", False) and not getattr(args, "no_enable", False): diff --git a/tests/hermes_cli/test_console_engine.py b/tests/hermes_cli/test_console_engine.py index b99f7108db40a..1b56ef8d4028a 100644 --- a/tests/hermes_cli/test_console_engine.py +++ b/tests/hermes_cli/test_console_engine.py @@ -504,3 +504,85 @@ def test_execute_handler_string_exit_returns_error_not_crash(_isolate_hermes_hom assert result.status == "error" assert result.output + + +_ORPHAN_STORE_STATUS = { + "projects": [ + {"hash": "abc123", "workdir": "/gone/v2-project", "exists": False, "commits": 4}, + ], + "pre_v2_projects": [], +} + + +def _patch_checkpoint_manager(monkeypatch, prune_calls: list) -> None: + """Report one orphan project and record the resulting prune call.""" + import tools.checkpoint_manager as ckpt_mgr + + monkeypatch.setattr(ckpt_mgr, "store_status", lambda *a, **k: _ORPHAN_STORE_STATUS) + + def _fake_prune(**kwargs): + prune_calls.append(kwargs) + return { + "scanned": 1, + "deleted_orphan": 1, + "deleted_stale": 0, + "errors": 0, + "bytes_freed": 0, + } + + monkeypatch.setattr(ckpt_mgr, "prune_checkpoints", _fake_prune) + + +def test_console_checkpoints_prune_does_not_reprompt_for_orphans( + _isolate_hermes_home, monkeypatch +): + """`checkpoints prune` is console-mutating, so the nested prompt must be skipped. + + The console asks for confirmation itself before dispatching any command in the + `checkpoints` mutating set, and `_apply_confirmed_defaults` exists to keep the + CLI layer from asking a second time. `clear` and `clear-legacy` are force + defaulted; `prune` was not, so its orphan confirmation still called `input()`. + """ + prune_calls: list = [] + _patch_checkpoint_manager(monkeypatch, prune_calls) + + def _unexpected_input(_prompt): + raise AssertionError( + "input() must not be called: the console already confirmed `checkpoints prune`" + ) + + monkeypatch.setattr("builtins.input", _unexpected_input) + + result = HermesConsoleEngine().execute("checkpoints prune", confirmed=True) + + assert result.status == "ok" + assert len(prune_calls) == 1 + assert prune_calls[0]["delete_orphans"] is True + # No preview was shown, so there is nothing to bind the deletion to — the + # documented `--force` case for `orphan_allowlist`. + assert prune_calls[0]["orphan_allowlist"] is None + + +def test_console_checkpoints_prune_succeeds_without_a_tty( + _isolate_hermes_home, monkeypatch +): + """The dashboard console has no stdin, so an unskipped prompt aborts the command. + + `_capture_output` redirects stdout/stderr but never stdin, so `input()` raises + `EOFError`, `_confirm` returns False, and `cmd_prune` returns 1 — which the + console surfaces as a failed command for every user with an orphan project. + """ + prune_calls: list = [] + _patch_checkpoint_manager(monkeypatch, prune_calls) + + def _eof_input(_prompt): + raise EOFError + + monkeypatch.setattr("builtins.input", _eof_input) + + result = HermesConsoleEngine().execute("checkpoints prune", confirmed=True) + + assert result.status == "ok" + assert "Aborted." not in result.output + assert len(prune_calls) == 1 + assert prune_calls[0]["orphan_allowlist"] is None From 1025cb0e663f37bcaf6a71f147c79be7bf6aed3f Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 23:09:07 -0700 Subject: [PATCH 299/376] fix(desktop): stop deleted sessions resurrecting through racing list fetches MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A session deleted in the sidebar disappeared optimistically but flashed back when any list fetch raced the in-flight DELETE RPC — the backend page still carried the doomed row until the transaction committed (#50928, reproduced with 'Load more' and auto-refresh). The optimistic tombstone ($removedSessionIds) was only honored by the recents slice of refreshSessions; the messaging slice, the per-platform pager, and refreshMessagingSessions ingested backend pages unfiltered. Extract the tombstone filter into dropTombstoned() and apply it at every session-list ingestion point. Tombstones only exist while a delete or archive is in flight (they self-clear on confirmation and are removed immediately on failure), so non-destructive refresh paths are untouched. Fixes #50928 --- .../hooks/use-session-list-actions.test.tsx | 34 +++++++++++++++++++ .../session/hooks/use-session-list-actions.ts | 30 ++++++++++------ 2 files changed, 54 insertions(+), 10 deletions(-) diff --git a/apps/desktop/src/app/session/hooks/use-session-list-actions.test.tsx b/apps/desktop/src/app/session/hooks/use-session-list-actions.test.tsx index 8b32a8b85ee6d..f25bfc9d53dd8 100644 --- a/apps/desktop/src/app/session/hooks/use-session-list-actions.test.tsx +++ b/apps/desktop/src/app/session/hooks/use-session-list-actions.test.tsx @@ -171,6 +171,40 @@ describe('refreshSessions identity + loading hygiene', () => { expect($sessions.get().map(s => s.id)).toEqual(['a']) }) + it('drops tombstoned rows from the messaging slice and per-platform paging too (#50928)', async () => { + // The same delete race exists on every ingestion point: the batched + // refresh's messaging slice and the per-platform "load more" pager must + // both honor the tombstone, or a deleted platform thread resurrects. + removed.ids = new Set(['tg-2']) + listSidebarSessions.mockResolvedValue( + sidebar( + { sessions: [] }, + [], + [row('tg-1', { source: 'telegram' }), row('tg-2', { source: 'telegram' })] + ) + ) + + const { result } = renderHook(() => useSessionListActions({ profileScope: 'default' })) + + await act(async () => { + await result.current.refreshSessions() + }) + + expect($messagingSessions.get().map(s => s.id)).toEqual(['tg-1']) + + // Per-platform pager: backend page still lists the doomed row. + listAllProfileSessions.mockResolvedValue({ + sessions: [row('tg-1', { source: 'telegram' }), row('tg-2', { source: 'telegram' }), row('tg-3', { source: 'telegram' })], + total: 3 + }) + + await act(async () => { + await result.current.loadMoreMessagingForPlatform('telegram') + }) + + expect($messagingSessions.get().map(s => s.id)).toEqual(['tg-1', 'tg-3']) + }) + it('still shows loading for the initial (empty-list) fetch', async () => { listSidebarSessions.mockResolvedValue(sidebar({ sessions: [row('a')] })) const { result } = renderHook(() => useSessionListActions({ profileScope: 'default' })) diff --git a/apps/desktop/src/app/session/hooks/use-session-list-actions.ts b/apps/desktop/src/app/session/hooks/use-session-list-actions.ts index bad975a5b31eb..959edfc5a15bc 100644 --- a/apps/desktop/src/app/session/hooks/use-session-list-actions.ts +++ b/apps/desktop/src/app/session/hooks/use-session-list-actions.ts @@ -50,6 +50,22 @@ const SIDEBAR_EXCLUDED_SOURCES = ['cron', 'kanban', 'subagent', 'tool', ...MESSA // external-platform conversations remain, then split per platform in the UI. const MESSAGING_EXCLUDED_SOURCES = ['cron', ...LOCAL_SESSION_SOURCE_IDS] +// Drop rows the user just deleted/archived: ANY list fetch (full refresh, +// "Load more" paging, a per-platform messaging page, the cron slice) can race +// an in-flight delete RPC, and the backend page still carries the doomed row +// until the DELETE commits — so it flashed back into the sidebar (#50928). +// Honoring the optimistic tombstone at every ingestion point keeps the removal +// stable; the tombstone self-clears once projects.tree confirms the delete, +// and a failed delete untombstones immediately, so nothing is filtered on the +// non-destructive paths. +function dropTombstoned(sessions: SessionInfo[]): SessionInfo[] { + const tombstones = $removedSessionIds.get() + + return tombstones.size + ? sessions.filter(s => !tombstones.has(s.id) && !(s._lineage_root_id && tombstones.has(s._lineage_root_id))) + : sessions +} + // Rows a session refresh must preserve even if the aggregator omits them: // in-flight first turns (message_count 0), pinned rows aged off the page, the // actively-viewed chat (its "working" flag clears a beat before the aggregator @@ -98,7 +114,7 @@ export function useSessionListActions({ profileScope }: UseSessionListActionsArg // Drop any non-messaging source the broad exclude didn't catch (custom // sources) — those stay in local recents, not a platform section. - const rows = result.sessions.filter(s => isMessagingSource(s.source)) + const rows = dropTombstoned(result.sessions.filter(s => isMessagingSource(s.source))) setMessagingSessions(prev => (sameCronSignature(prev, rows) ? prev : rows)) // Hit the cap → at least one platform may have more on disk than loaded, @@ -120,7 +136,7 @@ export function useSessionListActions({ profileScope }: UseSessionListActionsArg source: platform }) - const incoming = result.sessions.filter(s => normalizeSessionSource(s.source) === platform) + const incoming = dropTombstoned(result.sessions.filter(s => normalizeSessionSource(s.source) === platform)) setMessagingSessions(prev => [ ...prev.filter(s => !inPlatform(s)), @@ -193,13 +209,7 @@ export function useSessionListActions({ profileScope }: UseSessionListActionsArg // in-flight mutation and the backend page still carries the doomed row. // Honoring the optimistic tombstone keeps the removal from flashing back // (the tombstone self-clears once projects.tree confirms the delete). - const tombstones = $removedSessionIds.get() - - const incoming = tombstones.size - ? recents.sessions.filter( - s => !tombstones.has(s.id) && !(s._lineage_root_id && tombstones.has(s._lineage_root_id)) - ) - : recents.sessions + const incoming = dropTombstoned(recents.sessions) // Signature-gate the swap (same pattern as cron/messaging): a refresh // that returns content-identical rows must keep the previous array @@ -244,7 +254,7 @@ export function useSessionListActions({ profileScope }: UseSessionListActionsArg // Messaging sections: drop any non-messaging source the broad exclude // didn't catch (custom sources stay in local recents), then split per // platform in the UI. - const messagingRows = result.messaging.sessions.filter(s => isMessagingSource(s.source)) + const messagingRows = dropTombstoned(result.messaging.sessions.filter(s => isMessagingSource(s.source))) setMessagingSessions(prev => (sameCronSignature(prev, messagingRows) ? prev : messagingRows)) // Hit the cap → at least one platform may have more on disk than loaded. From eaaecbcc3064770265f1955123ad8a525f6b364e Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 23:10:19 -0700 Subject: [PATCH 300/376] chore: update contributor attribution map --- contributors/emails/RichardGuan1@users.noreply.github.com | 1 + 1 file changed, 1 insertion(+) create mode 100644 contributors/emails/RichardGuan1@users.noreply.github.com diff --git a/contributors/emails/RichardGuan1@users.noreply.github.com b/contributors/emails/RichardGuan1@users.noreply.github.com new file mode 100644 index 0000000000000..1e1f075dd78ec --- /dev/null +++ b/contributors/emails/RichardGuan1@users.noreply.github.com @@ -0,0 +1 @@ +RichardGuan1 From 3c2daf5ae730d1c1df469b6d4c6005b215317bf2 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 23:55:03 -0700 Subject: [PATCH 301/376] i18n(web): add kanban bulk-confirm keys to all locales (English fallback) The salvaged kanban ConfirmDialog added confirmDoneMany/confirmArchiveMany/ confirmBlockedMany to types.ts and en.ts only; the strict locale type requires every locale to carry them. English fallback pending translation. --- web/src/i18n/af.ts | 6 ++++++ web/src/i18n/ar.ts | 6 ++++++ web/src/i18n/de.ts | 6 ++++++ web/src/i18n/es.ts | 6 ++++++ web/src/i18n/fr.ts | 6 ++++++ web/src/i18n/ga.ts | 6 ++++++ web/src/i18n/hu.ts | 6 ++++++ web/src/i18n/it.ts | 6 ++++++ web/src/i18n/ja.ts | 6 ++++++ web/src/i18n/ko.ts | 6 ++++++ web/src/i18n/pt.ts | 6 ++++++ web/src/i18n/ru.ts | 6 ++++++ web/src/i18n/tr.ts | 6 ++++++ web/src/i18n/uk.ts | 6 ++++++ web/src/i18n/zh-hant.ts | 6 ++++++ web/src/i18n/zh.ts | 6 ++++++ 16 files changed, 96 insertions(+) diff --git a/web/src/i18n/af.ts b/web/src/i18n/af.ts index f43e81aea697d..ebc0c9a566a46 100644 --- a/web/src/i18n/af.ts +++ b/web/src/i18n/af.ts @@ -618,6 +618,12 @@ export const af: Translations = { "Borde laat u toe om onverwante werkstrome te skei — een per projek, repositorium of domein. Werkers op een bord sien nooit 'n ander bord se take nie.", slug: "Slug", slugHint: "— kleinletters, koppeltekens, bv. atm10-server", + confirmDoneMany: + "Mark {n} tasks as done? The workers' claims are released and dependent children become ready.", + confirmArchiveMany: + "Archive {n} tasks? They disappear from the default board view.", + confirmBlockedMany: + "Mark {n} tasks as blocked? The workers' claims are released.", displayName: "Vertoonnaam", displayNameHint: "(opsioneel)", description: "Beskrywing", diff --git a/web/src/i18n/ar.ts b/web/src/i18n/ar.ts index be7a6f06036c5..996b95b7fc511 100644 --- a/web/src/i18n/ar.ts +++ b/web/src/i18n/ar.ts @@ -553,6 +553,12 @@ export const ar = defineLocale({ "تتيح اللوحات فصل تدفقات العمل غير المرتبطة — واحدة لكل مشروع أو مستودع أو مجال.", slug: "المعرِّف", slugHint: "— أحرف صغيرة، واصلات، مثال atm10-server", + confirmDoneMany: + "Mark {n} tasks as done? The workers' claims are released and dependent children become ready.", + confirmArchiveMany: + "Archive {n} tasks? They disappear from the default board view.", + confirmBlockedMany: + "Mark {n} tasks as blocked? The workers' claims are released.", displayName: "الاسم المعروض", displayNameHint: "(اختياري)", description: "الوصف", diff --git a/web/src/i18n/de.ts b/web/src/i18n/de.ts index 58c8998cf2f44..71bd62d1eaebb 100644 --- a/web/src/i18n/de.ts +++ b/web/src/i18n/de.ts @@ -617,6 +617,12 @@ export const de: Translations = { "Mit Boards kannst du voneinander unabhängige Arbeitsabläufe trennen — eines pro Projekt, Repository oder Domäne. Worker auf einem Board sehen niemals die Aufgaben eines anderen Boards.", slug: "Slug", slugHint: "— Kleinbuchstaben, Bindestriche, z. B. atm10-server", + confirmDoneMany: + "Mark {n} tasks as done? The workers' claims are released and dependent children become ready.", + confirmArchiveMany: + "Archive {n} tasks? They disappear from the default board view.", + confirmBlockedMany: + "Mark {n} tasks as blocked? The workers' claims are released.", displayName: "Anzeigename", displayNameHint: "(optional)", description: "Beschreibung", diff --git a/web/src/i18n/es.ts b/web/src/i18n/es.ts index 4832e44be9e42..1a67a55aea28a 100644 --- a/web/src/i18n/es.ts +++ b/web/src/i18n/es.ts @@ -618,6 +618,12 @@ export const es: Translations = { "Los tableros te permiten separar flujos de trabajo no relacionados — uno por proyecto, repositorio o dominio. Los workers de un tablero nunca ven las tareas de otro.", slug: "Slug", slugHint: "— minúsculas, guiones, p. ej. atm10-server", + confirmDoneMany: + "Mark {n} tasks as done? The workers' claims are released and dependent children become ready.", + confirmArchiveMany: + "Archive {n} tasks? They disappear from the default board view.", + confirmBlockedMany: + "Mark {n} tasks as blocked? The workers' claims are released.", displayName: "Nombre visible", displayNameHint: "(opcional)", description: "Descripción", diff --git a/web/src/i18n/fr.ts b/web/src/i18n/fr.ts index d882674eed779..cf9cc08bb4e73 100644 --- a/web/src/i18n/fr.ts +++ b/web/src/i18n/fr.ts @@ -618,6 +618,12 @@ export const fr: Translations = { "Les tableaux vous permettent de séparer des flux de travail indépendants — un par projet, dépôt ou domaine. Les workers d'un tableau ne voient jamais les tâches d'un autre.", slug: "Slug", slugHint: "— minuscules, tirets, par ex. atm10-server", + confirmDoneMany: + "Mark {n} tasks as done? The workers' claims are released and dependent children become ready.", + confirmArchiveMany: + "Archive {n} tasks? They disappear from the default board view.", + confirmBlockedMany: + "Mark {n} tasks as blocked? The workers' claims are released.", displayName: "Nom affiché", displayNameHint: "(facultatif)", description: "Description", diff --git a/web/src/i18n/ga.ts b/web/src/i18n/ga.ts index 91a1c5858ed63..4caf9a68c4320 100644 --- a/web/src/i18n/ga.ts +++ b/web/src/i18n/ga.ts @@ -626,6 +626,12 @@ export const ga: Translations = { "Ligeann boards duit sruthanna oibre neamhghaolmhara a scaradh — ceann amháin in aghaidh an tionscadail, an repo nó an fhearainn. Ní fheiceann workers ar bhord amháin tascanna board eile riamh.", slug: "Slug", slugHint: "— litreacha beaga, fleiscíní, m.sh. atm10-server", + confirmDoneMany: + "Mark {n} tasks as done? The workers' claims are released and dependent children become ready.", + confirmArchiveMany: + "Archive {n} tasks? They disappear from the default board view.", + confirmBlockedMany: + "Mark {n} tasks as blocked? The workers' claims are released.", displayName: "Ainm taispeána", displayNameHint: "(roghnach)", description: "Cur síos", diff --git a/web/src/i18n/hu.ts b/web/src/i18n/hu.ts index b3f7f1b7b7825..d05a8986d2d8e 100644 --- a/web/src/i18n/hu.ts +++ b/web/src/i18n/hu.ts @@ -618,6 +618,12 @@ export const hu: Translations = { "A táblákkal külön tudod választani az egymással nem összefüggő munkafolyamokat — egyet projektenként, repónként vagy területenként. Az egyik tábla workerei sosem látják a másik tábla feladatait.", slug: "Slug", slugHint: "— kisbetűk, kötőjelek, pl. atm10-server", + confirmDoneMany: + "Mark {n} tasks as done? The workers' claims are released and dependent children become ready.", + confirmArchiveMany: + "Archive {n} tasks? They disappear from the default board view.", + confirmBlockedMany: + "Mark {n} tasks as blocked? The workers' claims are released.", displayName: "Megjelenítendő név", displayNameHint: "(opcionális)", description: "Leírás", diff --git a/web/src/i18n/it.ts b/web/src/i18n/it.ts index a0b1410a94b3a..561185cbf0c3f 100644 --- a/web/src/i18n/it.ts +++ b/web/src/i18n/it.ts @@ -617,6 +617,12 @@ export const it: Translations = { "Le bacheche ti permettono di separare flussi di lavoro non correlati — una per progetto, repository o dominio. I worker su una bacheca non vedono mai le attività di un'altra.", slug: "Slug", slugHint: "— minuscolo, trattini, ad es. atm10-server", + confirmDoneMany: + "Mark {n} tasks as done? The workers' claims are released and dependent children become ready.", + confirmArchiveMany: + "Archive {n} tasks? They disappear from the default board view.", + confirmBlockedMany: + "Mark {n} tasks as blocked? The workers' claims are released.", displayName: "Nome visualizzato", displayNameHint: "(facoltativo)", description: "Descrizione", diff --git a/web/src/i18n/ja.ts b/web/src/i18n/ja.ts index 05688d12a3f76..ea17f83925d4f 100644 --- a/web/src/i18n/ja.ts +++ b/web/src/i18n/ja.ts @@ -617,6 +617,12 @@ export const ja: Translations = { "ボードを使うと、関連のない作業の流れを分けられます — プロジェクト、リポジトリ、ドメインごとに 1 つずつ。あるボードのワーカーは、別のボードのタスクを見ることはありません。", slug: "スラッグ", slugHint: "— 小文字とハイフン、例: atm10-server", + confirmDoneMany: + "Mark {n} tasks as done? The workers' claims are released and dependent children become ready.", + confirmArchiveMany: + "Archive {n} tasks? They disappear from the default board view.", + confirmBlockedMany: + "Mark {n} tasks as blocked? The workers' claims are released.", displayName: "表示名", displayNameHint: "(任意)", description: "説明", diff --git a/web/src/i18n/ko.ts b/web/src/i18n/ko.ts index a6c3ae11cdb26..fcf0f6f7a2de4 100644 --- a/web/src/i18n/ko.ts +++ b/web/src/i18n/ko.ts @@ -617,6 +617,12 @@ export const ko: Translations = { "보드를 사용하면 관련 없는 작업 흐름을 분리할 수 있습니다 — 프로젝트, 저장소, 도메인마다 하나씩. 한 보드의 워커는 다른 보드의 작업을 절대 보지 않습니다.", slug: "슬러그", slugHint: "— 소문자, 하이픈, 예: atm10-server", + confirmDoneMany: + "Mark {n} tasks as done? The workers' claims are released and dependent children become ready.", + confirmArchiveMany: + "Archive {n} tasks? They disappear from the default board view.", + confirmBlockedMany: + "Mark {n} tasks as blocked? The workers' claims are released.", displayName: "표시 이름", displayNameHint: "(선택)", description: "설명", diff --git a/web/src/i18n/pt.ts b/web/src/i18n/pt.ts index 7c013dd889485..cc05250315e17 100644 --- a/web/src/i18n/pt.ts +++ b/web/src/i18n/pt.ts @@ -619,6 +619,12 @@ export const pt: Translations = { "Os quadros permitem-lhe separar fluxos de trabalho não relacionados — um por projeto, repositório ou domínio. Os workers de um quadro nunca veem as tarefas de outro quadro.", slug: "Slug", slugHint: "— minúsculas, hífenes, p. ex. atm10-server", + confirmDoneMany: + "Mark {n} tasks as done? The workers' claims are released and dependent children become ready.", + confirmArchiveMany: + "Archive {n} tasks? They disappear from the default board view.", + confirmBlockedMany: + "Mark {n} tasks as blocked? The workers' claims are released.", displayName: "Nome a apresentar", displayNameHint: "(opcional)", description: "Descrição", diff --git a/web/src/i18n/ru.ts b/web/src/i18n/ru.ts index abc7c7bda1b9b..89a89d410b8d6 100644 --- a/web/src/i18n/ru.ts +++ b/web/src/i18n/ru.ts @@ -618,6 +618,12 @@ export const ru: Translations = { "Доски позволяют разделять не связанные между собой потоки работы — по одной на проект, репозиторий или область. Воркеры одной доски никогда не видят задачи другой.", slug: "Slug", slugHint: "— строчные буквы, дефисы, например atm10-server", + confirmDoneMany: + "Mark {n} tasks as done? The workers' claims are released and dependent children become ready.", + confirmArchiveMany: + "Archive {n} tasks? They disappear from the default board view.", + confirmBlockedMany: + "Mark {n} tasks as blocked? The workers' claims are released.", displayName: "Отображаемое имя", displayNameHint: "(необязательно)", description: "Описание", diff --git a/web/src/i18n/tr.ts b/web/src/i18n/tr.ts index 10c332d2a9ffd..181b411ec48b0 100644 --- a/web/src/i18n/tr.ts +++ b/web/src/i18n/tr.ts @@ -618,6 +618,12 @@ export const tr: Translations = { "Panolar, ilgisiz iş akışlarını ayırmanızı sağlar — proje, depo veya alan başına bir pano. Bir panodaki worker'lar başka bir panonun görevlerini asla görmez.", slug: "Slug", slugHint: "— küçük harf, tire, ör. atm10-server", + confirmDoneMany: + "Mark {n} tasks as done? The workers' claims are released and dependent children become ready.", + confirmArchiveMany: + "Archive {n} tasks? They disappear from the default board view.", + confirmBlockedMany: + "Mark {n} tasks as blocked? The workers' claims are released.", displayName: "Görünen ad", displayNameHint: "(isteğe bağlı)", description: "Açıklama", diff --git a/web/src/i18n/uk.ts b/web/src/i18n/uk.ts index 194d0e62928fd..4579e01a97811 100644 --- a/web/src/i18n/uk.ts +++ b/web/src/i18n/uk.ts @@ -619,6 +619,12 @@ export const uk: Translations = { "Дошки дозволяють розділяти непов'язані потоки роботи — по одній на проєкт, репозиторій або домен. Воркери на одній дошці ніколи не бачать задач іншої дошки.", slug: "Slug", slugHint: "— рядкові літери, дефіси, напр. atm10-server", + confirmDoneMany: + "Mark {n} tasks as done? The workers' claims are released and dependent children become ready.", + confirmArchiveMany: + "Archive {n} tasks? They disappear from the default board view.", + confirmBlockedMany: + "Mark {n} tasks as blocked? The workers' claims are released.", displayName: "Відображувана назва", displayNameHint: "(необов'язково)", description: "Опис", diff --git a/web/src/i18n/zh-hant.ts b/web/src/i18n/zh-hant.ts index 49bc2b992b67b..15f3790bb23fd 100644 --- a/web/src/i18n/zh-hant.ts +++ b/web/src/i18n/zh-hant.ts @@ -617,6 +617,12 @@ export const zhHant: Translations = { "看板可將不相關的工作流分開——每個專案、程式碼庫或網域一個看板。一個看板上的工作者不會看到另一個看板的任務。", slug: "識別碼", slugHint: "— 小寫字母、連字號,例如 atm10-server", + confirmDoneMany: + "Mark {n} tasks as done? The workers' claims are released and dependent children become ready.", + confirmArchiveMany: + "Archive {n} tasks? They disappear from the default board view.", + confirmBlockedMany: + "Mark {n} tasks as blocked? The workers' claims are released.", displayName: "顯示名稱", displayNameHint: "(選填)", description: "描述", diff --git a/web/src/i18n/zh.ts b/web/src/i18n/zh.ts index 38a5071a03182..32854698f9342 100644 --- a/web/src/i18n/zh.ts +++ b/web/src/i18n/zh.ts @@ -613,6 +613,12 @@ export const zh: Translations = { "看板可以将不相关的工作流分开——每个项目、代码库或域一个看板。一个看板上的工作者不会看到另一个看板的任务。", slug: "标识", slugHint: "— 小写字母、连字符,例如 atm10-server", + confirmDoneMany: + "Mark {n} tasks as done? The workers' claims are released and dependent children become ready.", + confirmArchiveMany: + "Archive {n} tasks? They disappear from the default board view.", + confirmBlockedMany: + "Mark {n} tasks as blocked? The workers' claims are released.", displayName: "显示名称", displayNameHint: "(可选)", description: "描述", From 6fb6123f3bc40a074153aa4e9053225d7eed25c3 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 15 Aug 2026 00:05:34 -0700 Subject: [PATCH 302/376] test: replace any-casts with typed globalThis narrowing in plugin SDK test --- web/src/plugins/registry.test.ts | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/web/src/plugins/registry.test.ts b/web/src/plugins/registry.test.ts index a45a318433c9a..13b5e4c57a2be 100644 --- a/web/src/plugins/registry.test.ts +++ b/web/src/plugins/registry.test.ts @@ -14,7 +14,7 @@ import { exposePluginSDK } from "./registry"; describe("plugin SDK dialog/toast surface", () => { beforeEach(() => { // Reset window between tests so exposePluginSDK() writes fresh. - (globalThis as any).window = { + (globalThis as unknown as { window: Record }).window = { __HERMES_PLUGINS__: undefined, __HERMES_PLUGIN_SDK__: undefined, }; @@ -22,7 +22,7 @@ describe("plugin SDK dialog/toast surface", () => { it("exposes Dialog + subcomponents on components", () => { exposePluginSDK(); - const sdk = (globalThis as any).window.__HERMES_PLUGIN_SDK__; + const sdk = (globalThis as unknown as { window: { __HERMES_PLUGIN_SDK__: { components: Record; hooks: Record } } }).window.__HERMES_PLUGIN_SDK__; expect(sdk.components.Dialog).toBeDefined(); expect(sdk.components.DialogContent).toBeDefined(); expect(sdk.components.DialogHeader).toBeDefined(); @@ -36,7 +36,7 @@ describe("plugin SDK dialog/toast surface", () => { it("exposes useToast and useConfirmDelete on hooks", () => { exposePluginSDK(); - const sdk = (globalThis as any).window.__HERMES_PLUGIN_SDK__; + const sdk = (globalThis as unknown as { window: { __HERMES_PLUGIN_SDK__: { components: Record; hooks: Record } } }).window.__HERMES_PLUGIN_SDK__; expect(typeof sdk.hooks.useToast).toBe("function"); expect(typeof sdk.hooks.useConfirmDelete).toBe("function"); // Original React hooks still present (no accidental removal). From 20beee774cc471aba8d6277df93cf202c27a8740 Mon Sep 17 00:00:00 2001 From: Takumi Sato Date: Wed, 3 Jun 2026 00:59:36 +0900 Subject: [PATCH 303/376] fix(desktop): prevent Enter from submitting during IME composition Check event.nativeEvent.isComposing in Enter-to-submit branches so CJK (Japanese, Chinese, Korean) and other compositional input methods commit the candidate instead of firing the send handler. Applied to all five sites in the desktop renderer that currently handle Enter with no composition guard: - chat composer main submit and trigger popover (apps/desktop/src/app/chat/composer/index.tsx) - message edit composer submit and trigger popover (apps/desktop/src/components/assistant-ui/thread.tsx) - onboarding API key and auth code inputs (apps/desktop/src/components/desktop-onboarding-overlay.tsx) - session rename input (apps/desktop/src/app/chat/sidebar/session-actions-menu.tsx) Fixes #37483 --- apps/desktop/src/app/chat/sidebar/session-actions-menu.tsx | 2 +- apps/desktop/src/components/onboarding/flow.tsx | 2 +- apps/desktop/src/components/onboarding/index.tsx | 4 ++-- 3 files changed, 4 insertions(+), 4 deletions(-) diff --git a/apps/desktop/src/app/chat/sidebar/session-actions-menu.tsx b/apps/desktop/src/app/chat/sidebar/session-actions-menu.tsx index 3a9e5b8c10c3d..5fc32e6c3b775 100644 --- a/apps/desktop/src/app/chat/sidebar/session-actions-menu.tsx +++ b/apps/desktop/src/app/chat/sidebar/session-actions-menu.tsx @@ -652,7 +652,7 @@ function RenameSessionDialog({ open, onOpenChange, sessionId, currentTitle, prof disabled={submitting} onChange={event => setValue(event.target.value)} onKeyDown={event => { - if (event.key === 'Enter') { + if (event.key === 'Enter' && !event.nativeEvent.isComposing) { event.preventDefault() void submit() } else if (event.key === 'Escape') { diff --git a/apps/desktop/src/components/onboarding/flow.tsx b/apps/desktop/src/components/onboarding/flow.tsx index c81419fc847d6..dc4b16a54a72f 100644 --- a/apps/desktop/src/components/onboarding/flow.tsx +++ b/apps/desktop/src/components/onboarding/flow.tsx @@ -82,7 +82,7 @@ export function FlowPanel({ setOnboardingCode(e.target.value)} - onKeyDown={e => e.key === 'Enter' && void submitOnboardingCode(ctx)} + onKeyDown={e => e.key === 'Enter' && !e.nativeEvent.isComposing && void submitOnboardingCode(ctx)} placeholder={t.onboarding.pasteAuthCode} value={flow.code} /> diff --git a/apps/desktop/src/components/onboarding/index.tsx b/apps/desktop/src/components/onboarding/index.tsx index 3b44c4cba057b..e75d12800097a 100644 --- a/apps/desktop/src/components/onboarding/index.tsx +++ b/apps/desktop/src/components/onboarding/index.tsx @@ -662,7 +662,7 @@ export function ApiKeyForm({ autoFocus className="font-mono" onChange={e => setValue(e.target.value)} - onKeyDown={e => e.key === 'Enter' && void submit()} + onKeyDown={e => e.key === 'Enter' && !e.nativeEvent.isComposing && void submit()} placeholder={ currentRedacted ?? (alreadySet ? t.onboarding.replaceCurrent : option.placeholder || t.onboarding.pasteApiKey) @@ -675,7 +675,7 @@ export function ApiKeyForm({ autoComplete="off" className="font-mono" onChange={e => setLocalKey(e.target.value)} - onKeyDown={e => e.key === 'Enter' && void submit()} + onKeyDown={e => e.key === 'Enter' && !e.nativeEvent.isComposing && void submit()} placeholder={t.onboarding.localApiKeyPlaceholder} type="password" value={localKey} From 3d51c4099725f0357807247799184b6ff4543353 Mon Sep 17 00:00:00 2001 From: izumi0uu Date: Sat, 6 Jun 2026 22:50:28 +0800 Subject: [PATCH 304/376] fix(desktop): prevent IME submit in inline edit composer --- .../thread/user-edit-composer.tsx | 41 +++++++++++++++---- .../thread/user-message-edit.test.tsx | 35 +++++++++++++++- 2 files changed, 67 insertions(+), 9 deletions(-) diff --git a/apps/desktop/src/components/assistant-ui/thread/user-edit-composer.tsx b/apps/desktop/src/components/assistant-ui/thread/user-edit-composer.tsx index c7af4d56e050a..a9b4fd051648d 100644 --- a/apps/desktop/src/components/assistant-ui/thread/user-edit-composer.tsx +++ b/apps/desktop/src/components/assistant-ui/thread/user-edit-composer.tsx @@ -95,6 +95,7 @@ export const UserEditComposer: FC = ({ cwd, gateway, sess // mount-time snapshot can incorrectly classify every later blur as dirty. const initialDraftRef = useRef(null) const draftRef = useRef(draft) + const composingRef = useRef(false) const dragDepthRef = useRef(0) const [dragActive, setDragActive] = useState(false) const [trigger, setTrigger] = useState(null) @@ -517,16 +518,27 @@ export const UserEditComposer: FC = ({ cwd, gateway, sess } } - const handleInput = (event: FormEvent) => { - const editor = event.currentTarget + const flushEditorToDraft = useCallback( + (editor: HTMLDivElement) => { + if (editor.childNodes.length === 1 && editor.firstChild?.nodeName === 'BR') { + editor.replaceChildren() + } - if (editor.childNodes.length === 1 && editor.firstChild?.nodeName === 'BR') { - editor.replaceChildren() + rememberInitialDraft() + const nextDraft = syncDraftFromEditor(editor) + window.setTimeout(refreshTrigger, 0) + + return nextDraft + }, + [refreshTrigger, rememberInitialDraft, syncDraftFromEditor] + ) + + const handleInput = (event: FormEvent) => { + if (composingRef.current) { + return } - rememberInitialDraft() - syncDraftFromEditor(editor) - window.setTimeout(refreshTrigger, 0) + flushEditorToDraft(event.currentTarget) } // Native typing/deleting still goes through Chromium's editing pipeline, whose @@ -567,6 +579,10 @@ export const UserEditComposer: FC = ({ cwd, gateway, sess } const submitEdit = (editor: HTMLDivElement) => { + if (composingRef.current) { + return + } + const nextDraft = syncDraftFromEditor(editor) if (submitting || staging || !nextDraft.trim()) { @@ -634,6 +650,10 @@ export const UserEditComposer: FC = ({ cwd, gateway, sess ) const handleKeyDown = (event: KeyboardEvent) => { + if (composingRef.current || event.nativeEvent.isComposing) { + return + } + if (trigger && triggerItems.length > 0) { if (event.key === 'ArrowDown') { event.preventDefault() @@ -781,6 +801,13 @@ export const UserEditComposer: FC = ({ cwd, gateway, sess data-slot={RICH_INPUT_SLOT} onBeforeInput={handleBeforeInput} onBlur={() => window.setTimeout(closeTrigger, 80)} + onCompositionEnd={event => { + composingRef.current = false + flushEditorToDraft(event.currentTarget) + }} + onCompositionStart={() => { + composingRef.current = true + }} onDragOver={handleDragOver} onDrop={handleDrop} onFocus={() => markActiveComposer('edit')} diff --git a/apps/desktop/src/components/assistant-ui/thread/user-message-edit.test.tsx b/apps/desktop/src/components/assistant-ui/thread/user-message-edit.test.tsx index e912856800a1a..7b89cacd643f7 100644 --- a/apps/desktop/src/components/assistant-ui/thread/user-message-edit.test.tsx +++ b/apps/desktop/src/components/assistant-ui/thread/user-message-edit.test.tsx @@ -1,4 +1,4 @@ -import { ExportedMessageRepository } from '@assistant-ui/react' +import { type AppendMessage, ExportedMessageRepository } from '@assistant-ui/react' // Clicking a user bubble must open the inline edit composer — through the // app's incremental external-store runtime (which reimplements capability // resolution, incl. `edit: onEdit !== undefined`) and the stock runtime. @@ -96,7 +96,7 @@ function assistantMessage(): ThreadMessage { } // Mirrors chat/index.tsx: incremental runtime + messageRepository + onEdit. -function IncrementalHarness({ onEdit }: { onEdit: () => Promise }) { +function IncrementalHarness({ onEdit }: { onEdit: (message: AppendMessage) => Promise }) { const repository = ExportedMessageRepository.fromArray([userMessage(), assistantMessage()]) const runtime = useIncrementalExternalStoreRuntime({ @@ -145,6 +145,37 @@ describe('click-to-edit user message', () => { }) }) + it('does not submit an inline edit while IME composition is active', async () => { + const onEdit = vi.fn(async (_message: AppendMessage) => {}) + + render() + fireEvent.click(await screen.findByRole('button', { name: 'Edit message' })) + + const editor = await screen.findByRole('textbox', { name: 'Edit message' }) + const editedText = 'edit me please\u4f60' + + await act(async () => { + fireEvent.compositionStart(editor) + editor.textContent = editedText + fireEvent.input(editor) + fireEvent.keyDown(editor, { isComposing: true, key: 'Enter' }) + }) + + expect(onEdit).not.toHaveBeenCalled() + + await act(async () => { + fireEvent.compositionEnd(editor) + fireEvent.keyDown(editor, { key: 'Enter' }) + }) + + await waitFor(() => expect(onEdit).toHaveBeenCalledTimes(1)) + expect(onEdit).toHaveBeenCalledWith( + expect.objectContaining({ + content: [{ text: editedText, type: 'text' }] + }) + ) + }) + it('keeps a dirty inline edit open when focus leaves the composer', async () => { const { container } = render( {}} />) From 202e5b813af4fac7661716826bb243773d4f6707 Mon Sep 17 00:00:00 2001 From: liyunlong Date: Sun, 7 Jun 2026 13:29:38 +0800 Subject: [PATCH 305/376] fix(desktop): block Enter keyCode 229 after IME compositionend MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit macOS Chinese IME (and some 3rd-party Windows IMEs) emit Enter with keyCode 229 (legacy VK_PROCESSKEY) after compositionend, while isComposing is already false. The existing guard only checked isComposing and the composingRef, so this Enter slipped through and submitted the message before the committed text was fully in the DOM. Add an explicit keyCode 229 check in handleEditorKeyDown. keyCode is deprecated, but it is the only reliable signal for this IME commit Enter on Chromium-based browsers. Includes a dom-repro test that simulates the macOS IME sequence. Fixes: "中文混英文按 Enter 直接上屏" --- .../ime-composition-dom-repro.test.tsx | 83 +++++++++++++++++++ apps/desktop/src/app/chat/composer/index.tsx | 9 ++ 2 files changed, 92 insertions(+) diff --git a/apps/desktop/src/app/chat/composer/ime-composition-dom-repro.test.tsx b/apps/desktop/src/app/chat/composer/ime-composition-dom-repro.test.tsx index 962183ec7d830..0653dfb463318 100644 --- a/apps/desktop/src/app/chat/composer/ime-composition-dom-repro.test.tsx +++ b/apps/desktop/src/app/chat/composer/ime-composition-dom-repro.test.tsx @@ -105,4 +105,87 @@ describe('composer IME composition — send button visibility (#39614)', () => { expect(hasPayload).toBe(false) } }) + + it('blocks Enter with keyCode 229 even after compositionend (macOS Chinese IME)', async () => { + let submitCount = 0 + let hasPayload = false + + function KeyDownHarness({ onPayload }: { onPayload: (hasPayload: boolean) => void }) { + const editorRef = useRef(null) + const composingRef = useRef(false) + const draftRef = useRef('') + const [draft, setDraft] = useState('') + + const flushEditorToDraft = (editor: HTMLDivElement) => { + const next = editor.textContent ?? '' + if (next !== draftRef.current) { + draftRef.current = next + setDraft(next) + } + } + + onPayload(draft.trim().length > 0) + + const handleKeyDown = (event: React.KeyboardEvent) => { + if (composingRef.current || event.nativeEvent.isComposing) { + return + } + if (event.key === 'Enter' && event.keyCode === 229) { + return + } + if (event.key === 'Enter' && !event.shiftKey) { + event.preventDefault() + submitCount++ + } + } + + return ( +
{ + composingRef.current = false + flushEditorToDraft(event.currentTarget) + }} + onCompositionStart={() => { + composingRef.current = true + }} + onInput={event => { + if (composingRef.current) return + flushEditorToDraft(event.currentTarget) + }} + onKeyDown={handleKeyDown} + ref={editorRef} + suppressContentEditableWarning + /> + ) + } + + const { getByTestId } = render( (hasPayload = p)} />) + const editor = getByTestId('editor') + + // Simulate macOS Chinese IME: compositionend fires, then Enter with keyCode 229. + await act(async () => { + fireEvent.compositionStart(editor) + editor.textContent = '测试' + fireEvent.input(editor) + fireEvent.compositionEnd(editor) + }) + + expect(hasPayload).toBe(true) + + // This Enter must NOT trigger submit. + await act(async () => { + fireEvent.keyDown(editor, { key: 'Enter', keyCode: 229 }) + }) + + expect(submitCount).toBe(0) + + // A normal Enter afterwards should still submit. + await act(async () => { + fireEvent.keyDown(editor, { key: 'Enter', keyCode: 13 }) + }) + + expect(submitCount).toBe(1) + }) }) diff --git a/apps/desktop/src/app/chat/composer/index.tsx b/apps/desktop/src/app/chat/composer/index.tsx index d583affcce1bb..ec041ddfd7829 100644 --- a/apps/desktop/src/app/chat/composer/index.tsx +++ b/apps/desktop/src/app/chat/composer/index.tsx @@ -560,6 +560,15 @@ export function ChatBar({ return } + // macOS Chinese IME (and some 3rd-party IMEs on Windows) emit Enter with + // keyCode 229 (legacy VK_PROCESSKEY) while isComposing is already false. + // The compositionend has fired but the keydown still carries 229, signalling + // "this Enter is an IME commit, not a user send". If we let it through, + // the message fires before the committed text is fully in the DOM. + if (event.key === 'Enter' && event.keyCode === 229) { + return + } + // Undo/redo before anything else — we own the stack (see useComposerUndo), // so these never reach Chromium's native history, which has no record of // the Range-based edits the rich editor makes. From 39e760779469e18e366ea346864559412f57d954 Mon Sep 17 00:00:00 2001 From: AIalliAI <285906080+AIalliAI@users.noreply.github.com> Date: Thu, 11 Jun 2026 08:50:21 +0000 Subject: [PATCH 306/376] fix(desktop): recover from a stale IME composition flag that wedged the composer MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A missed compositionend (focus jump, input-source switch, programmatic DOM swap mid-preedit) left composingRef stuck true, and the stuck flag silently swallowed every Enter in handleEditorKeyDown and every Send-button submit via the form onSubmit guard — no error, no RPC, until the composer remounted. For CJK IME users (where even ASCII typing runs through composition) this read as "Enter has no effect; messages cannot be sent", degrading composer instance by composer instance. Recover in two places, both grounded in invariants Chromium guarantees: - keydown: every keydown during a genuine composition carries isComposing=true, so when the native flag says we're not composing, clear the stale ref before the guard reads it. - blur: a composition never survives focus loss, so clear the flag unconditionally — this is what unblocks the Send button path, which has no native composition flag to consult. The genuine-IME protection (#37483 class) is untouched: Enter with isComposing=true is still swallowed. Fixes #44135 --- .../composer/enter-stale-ime-flag.test.tsx | 129 ++++++++++++++++++ apps/desktop/src/app/chat/composer/index.tsx | 22 ++- 2 files changed, 150 insertions(+), 1 deletion(-) create mode 100644 apps/desktop/src/app/chat/composer/enter-stale-ime-flag.test.tsx diff --git a/apps/desktop/src/app/chat/composer/enter-stale-ime-flag.test.tsx b/apps/desktop/src/app/chat/composer/enter-stale-ime-flag.test.tsx new file mode 100644 index 0000000000000..64d2ee43a1d4c --- /dev/null +++ b/apps/desktop/src/app/chat/composer/enter-stale-ime-flag.test.tsx @@ -0,0 +1,129 @@ +import { act, cleanup, fireEvent, render } from '@testing-library/react' +import { useRef } from 'react' +import { afterEach, describe, expect, it, vi } from 'vitest' + +afterEach(cleanup) + +// Faithful mirror of index.tsx's IME wiring: the composition guard at the top +// of handleEditorKeyDown (self-heal + swallow), the compositionstart/end +// handlers, and the blur reset. +// +// Regression repro for #44135: compositionend can be missed (focus jumps, +// input-source switches, programmatic DOM swaps mid-preedit), leaving +// composingRef wedged true. Before the fix, a wedged flag silently swallowed +// every Enter — and, via the form onSubmit guard, the Send button — until the +// composer remounted, which read as "Enter has no effect, no error, nothing +// reaches the gateway". The fix trusts Chromium's per-keydown isComposing flag +// to clear a stale ref, and clears it on blur (a composition never survives +// focus loss). +function Harness({ onSubmit, wedgeComposing }: { onSubmit: (text: string) => void; wedgeComposing?: boolean }) { + const editorRef = useRef(null) + const composingRef = useRef(Boolean(wedgeComposing)) + + const submitDraft = () => { + onSubmit(editorRef.current?.textContent ?? '') + } + + const handleKeyDown = (event: React.KeyboardEvent) => { + if (composingRef.current && !event.nativeEvent.isComposing) { + composingRef.current = false + } + + if (composingRef.current || event.nativeEvent.isComposing) { + return + } + + if (event.key === 'Enter' && !event.shiftKey) { + event.preventDefault() + submitDraft() + } + } + + return ( +
+
{ + composingRef.current = false + }} + onCompositionEnd={() => { + composingRef.current = false + }} + onCompositionStart={() => { + composingRef.current = true + }} + onKeyDown={handleKeyDown} + ref={editorRef} + suppressContentEditableWarning + /> +
+ ) +} + +describe('composer Enter — stale IME composition flag recovery (#44135)', () => { + it('sends on Enter despite a wedged composing flag when the native event says not composing', async () => { + const onSubmit = vi.fn() + const { getByTestId } = render() + const editor = getByTestId('editor') + + await act(async () => { + editor.textContent = 'hello after wedge' + fireEvent.keyDown(editor, { key: 'Enter', isComposing: false }) + }) + + expect(onSubmit).toHaveBeenCalledWith('hello after wedge') + }) + + it('still swallows Enter during a genuine composition (isComposing keydown)', async () => { + const onSubmit = vi.fn() + const { getByTestId } = render() + const editor = getByTestId('editor') + + await act(async () => { + fireEvent.compositionStart(editor) + editor.textContent = '你好' + // The Enter that confirms the preedit: Chromium stamps isComposing=true. + fireEvent.keyDown(editor, { key: 'Enter', isComposing: true }) + }) + + expect(onSubmit).not.toHaveBeenCalled() + + // After compositionend, the next Enter sends normally. + await act(async () => { + fireEvent.compositionEnd(editor) + fireEvent.keyDown(editor, { key: 'Enter', isComposing: false }) + }) + + expect(onSubmit).toHaveBeenCalledWith('你好') + }) + + it('unblocks the Send button after blur even when compositionend was missed', async () => { + const onSubmit = vi.fn() + const { getByTestId } = render() + const editor = getByTestId('editor') + + await act(async () => { + fireEvent.compositionStart(editor) + editor.textContent = '发送' + // compositionend never fires (the wedge) — the user mouses to Send, + // blurring the editor. + fireEvent.blur(editor) + fireEvent.click(getByTestId('send')) + }) + + expect(onSubmit).toHaveBeenCalledWith('发送') + }) +}) diff --git a/apps/desktop/src/app/chat/composer/index.tsx b/apps/desktop/src/app/chat/composer/index.tsx index ec041ddfd7829..c61f81f1fc2cb 100644 --- a/apps/desktop/src/app/chat/composer/index.tsx +++ b/apps/desktop/src/app/chat/composer/index.tsx @@ -551,6 +551,18 @@ export function ChatBar({ } const handleEditorKeyDown = (event: KeyboardEvent) => { + // Self-heal a stale composition flag before the guard below reads it. + // compositionend can be missed (focus jumps, input-source switches, or a + // programmatic DOM swap mid-preedit abort the composition without the + // event reaching us), and a wedged composingRef silently swallows every + // Enter — and, via the form onSubmit guard, the Send button — until the + // component remounts (#44135). Chromium stamps isComposing on every + // keydown of a genuine composition, so when the native flag says we're + // not composing, trust it and recover. + if (composingRef.current && !event.nativeEvent.isComposing) { + composingRef.current = false + } + // IME composition: Enter confirms composed text, not a message submission. // We check both composingRef (set by compositionstart/compositionend, robust // across browsers) and nativeEvent.isComposing (Chromium fallback). Without @@ -1012,7 +1024,15 @@ export function ChatBar({ data-placeholder={placeholder} data-slot={RICH_INPUT_SLOT} onBeforeInput={handleEditorBeforeInput} - onBlur={() => window.setTimeout(closeTrigger, 80)} + onBlur={() => { + // A composition never survives focus loss (Chromium commits the + // preedit and fires compositionend on blur) — but if that event is + // missed, the wedged flag would block the Send button's form-submit + // guard forever (#44135). Clear unconditionally: by the time blur + // runs there is nothing left composing in this editor. + composingRef.current = false + window.setTimeout(closeTrigger, 80) + }} onCompositionEnd={event => { composingRef.current = false From 9f2e6d05ab2a345cfc350998c9ea0a236fe6a6c8 Mon Sep 17 00:00:00 2001 From: BAS Lam Date: Thu, 13 Aug 2026 09:46:51 +0800 Subject: [PATCH 307/376] fix(desktop): ignore IME composition keydowns in keybind combo resolution MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Chinese/Japanese/Korean IMEs emit keydown events during composition that carry preedit keystrokes and the commit keypress (Enter/Space/Shift for candidate selection). Treating them as combos fires unrelated keybinds — e.g. typing 你 with a CJK IME could dispatch session.new and silently open a new session. Guard comboFromEvent(): - Bail out entirely while composing (event.isComposing or key === 'Process') - Ignore keydowns whose event.key is a bare modifier name but whose code is a regular key — legacy IMEs that synthesize keystrokes (Q9 2002 sends key="Control" with code="KeyW") would otherwise canonicalize into phantom combos like mod+w that close the active tab. Tested with Q9 (九方) legacy IME on Windows. --- apps/desktop/src/lib/keybinds/combo.ts | 21 +++++++++++++++++++++ 1 file changed, 21 insertions(+) diff --git a/apps/desktop/src/lib/keybinds/combo.ts b/apps/desktop/src/lib/keybinds/combo.ts index 7f6f757dcee40..97f1a51d22b6f 100644 --- a/apps/desktop/src/lib/keybinds/combo.ts +++ b/apps/desktop/src/lib/keybinds/combo.ts @@ -50,6 +50,9 @@ const MODIFIER_CODES = new Set([ 'ShiftRight' ]) +// Modifier names as reported by `event.key` on a bare modifier keydown. +const MODIFIER_KEYS = new Set(['Alt', 'Control', 'Meta', 'Shift']) + function baseKeyFromCode(code: string): string | null { if (code.startsWith('Key')) { return code.slice(3).toLowerCase() @@ -101,10 +104,28 @@ function baseKeyFromEventKey(key: string, shiftKey: boolean): string | null { // Returns the canonical combo for a keydown, or null while only modifiers are // held (so capture mode keeps waiting for a real key). export function comboFromEvent(event: KeyboardEvent): string | null { + // IME composition (Chinese/Japanese/Korean input): the keydown events + // during composition carry preedit keystrokes and the commit keypress + // (Enter/Space/Shift for candidate selection). Treating them as combos + // fires unrelated keybinds — e.g. typing 你 with a Chinese IME sent a + // keydown that dispatched `session.new` and silently opened a new session. + // Bail out entirely while composing. + if (event.isComposing || event.key === 'Process') { + return null + } + if (MODIFIER_CODES.has(event.code)) { return null } + // A keydown whose `key` is a modifier name but whose `code` is a regular + // key is not a real modifier chord — legacy IMEs that synthesize keystrokes + // (Q9 2002 sends key="Control" with code="KeyW") produce these, and they + // would canonicalize to phantom combos (Ctrl+W → close active tab). Ignore. + if (MODIFIER_KEYS.has(event.key)) { + return null + } + const base = baseKeyFromEventKey(event.key, event.shiftKey) ?? baseKeyFromCode(event.code) if (!base) { From 587788405a7e8e957a69dc565b277fc244e071b4 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 23:06:56 -0700 Subject: [PATCH 308/376] fix(desktop): extend stale-flag self-heal and keyCode 229 guard to the edit composer Widen the composer-side IME fixes to the inline edit composer: the same missed-compositionend wedge and post-compositionend keyCode 229 Enter apply to its handleKeyDown path. --- .../assistant-ui/thread/user-edit-composer.tsx | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/apps/desktop/src/components/assistant-ui/thread/user-edit-composer.tsx b/apps/desktop/src/components/assistant-ui/thread/user-edit-composer.tsx index a9b4fd051648d..c54e9ec6f6c9e 100644 --- a/apps/desktop/src/components/assistant-ui/thread/user-edit-composer.tsx +++ b/apps/desktop/src/components/assistant-ui/thread/user-edit-composer.tsx @@ -650,10 +650,23 @@ export const UserEditComposer: FC = ({ cwd, gateway, sess ) const handleKeyDown = (event: KeyboardEvent) => { + // Self-heal a stale composition flag (same recovery as the main composer's + // handleEditorKeyDown, #44135): compositionend can be missed, and a wedged + // composingRef would swallow every Enter until the edit composer remounts. + if (composingRef.current && !event.nativeEvent.isComposing) { + composingRef.current = false + } + if (composingRef.current || event.nativeEvent.isComposing) { return } + // IME commit Enter still carrying keyCode 229 (VK_PROCESSKEY) after + // compositionend — same guard as the main composer. + if (event.key === 'Enter' && event.keyCode === 229) { + return + } + if (trigger && triggerItems.length > 0) { if (event.key === 'ArrowDown') { event.preventDefault() From aa9593946921c1069c169ea21fe67f74a89e2c31 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 23:10:32 -0700 Subject: [PATCH 309/376] test(desktop): cover IME composition guards in keybind combo resolution --- apps/desktop/src/lib/keybinds/combo.test.ts | 34 +++++++++++++++++++++ 1 file changed, 34 insertions(+) diff --git a/apps/desktop/src/lib/keybinds/combo.test.ts b/apps/desktop/src/lib/keybinds/combo.test.ts index 7b1bd219cb41b..192107251a484 100644 --- a/apps/desktop/src/lib/keybinds/combo.test.ts +++ b/apps/desktop/src/lib/keybinds/combo.test.ts @@ -140,3 +140,37 @@ describe('actionAllowedInInput', () => { expect(actionAllowedInInput('view.findInPage', 'mod+end')).toBe(false) }) }) + +describe('comboFromEvent — IME composition keydowns never resolve to combos (#84957)', () => { + it('returns null while a composition is in progress (isComposing)', async () => { + const { comboFromEvent } = await loadCombo('MacIntel') + + // Typing 你 with a Chinese IME: the preedit keydowns carry isComposing. + // Before the guard, these canonicalized to combos and fired keybinds + // (e.g. dispatched `session.new` mid-composition). + expect(comboFromEvent(keydown({ code: 'KeyN', isComposing: true, key: 'n' }))).toBeNull() + expect(comboFromEvent(keydown({ code: 'Enter', isComposing: true, key: 'Enter' }))).toBeNull() + expect(comboFromEvent(keydown({ code: 'Space', isComposing: true, key: ' ' }))).toBeNull() + }) + + it('returns null for the legacy key="Process" (VK_PROCESSKEY) keydown', async () => { + const { comboFromEvent } = await loadCombo('Win32') + + expect(comboFromEvent(keydown({ code: 'KeyW', key: 'Process' }))).toBeNull() + }) + + it('ignores IME-synthesized modifier-name keys on non-modifier codes', async () => { + const { comboFromEvent } = await loadCombo('Win32') + + // Q9 2002-style legacy IMEs synthesize key="Control" with code="KeyW", + // which would otherwise canonicalize to a phantom ctrl+w (close tab). + expect(comboFromEvent(keydown({ code: 'KeyW', key: 'Control' }))).toBeNull() + expect(comboFromEvent(keydown({ code: 'KeyA', key: 'Shift' }))).toBeNull() + }) + + it('still resolves real combos after composition ends', async () => { + const { comboFromEvent } = await loadCombo('MacIntel') + + expect(comboFromEvent(keydown({ code: 'KeyN', isComposing: false, key: 'n', metaKey: true }))).toBe('mod+n') + }) +}) From 27a22b8de7077f85b7c3d19aa96f1a23669cf158 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 23:10:52 -0700 Subject: [PATCH 310/376] chore: map contributor emails for IME salvage --- contributors/emails/baslam@users.noreply.github.com | 1 + contributors/emails/liyunlong@nemo.video | 1 + contributors/emails/takumisatojpn@gmail.com | 1 + 3 files changed, 3 insertions(+) create mode 100644 contributors/emails/baslam@users.noreply.github.com create mode 100644 contributors/emails/liyunlong@nemo.video create mode 100644 contributors/emails/takumisatojpn@gmail.com diff --git a/contributors/emails/baslam@users.noreply.github.com b/contributors/emails/baslam@users.noreply.github.com new file mode 100644 index 0000000000000..2a9f8340d6006 --- /dev/null +++ b/contributors/emails/baslam@users.noreply.github.com @@ -0,0 +1 @@ +hkfiberlaser-svg \ No newline at end of file diff --git a/contributors/emails/liyunlong@nemo.video b/contributors/emails/liyunlong@nemo.video new file mode 100644 index 0000000000000..a9d67b9f329df --- /dev/null +++ b/contributors/emails/liyunlong@nemo.video @@ -0,0 +1 @@ +leeclouddragon diff --git a/contributors/emails/takumisatojpn@gmail.com b/contributors/emails/takumisatojpn@gmail.com new file mode 100644 index 0000000000000..01c4182b88cf4 --- /dev/null +++ b/contributors/emails/takumisatojpn@gmail.com @@ -0,0 +1 @@ +satotakumi From 15e7c563b33fdbf830991c49158b8f1743a6ab47 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 23:41:43 -0700 Subject: [PATCH 311/376] style: eslint --fix on salvaged IME test (curly + padding rules) --- .../src/app/chat/composer/ime-composition-dom-repro.test.tsx | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/apps/desktop/src/app/chat/composer/ime-composition-dom-repro.test.tsx b/apps/desktop/src/app/chat/composer/ime-composition-dom-repro.test.tsx index 0653dfb463318..36880a6cd3ed3 100644 --- a/apps/desktop/src/app/chat/composer/ime-composition-dom-repro.test.tsx +++ b/apps/desktop/src/app/chat/composer/ime-composition-dom-repro.test.tsx @@ -118,6 +118,7 @@ describe('composer IME composition — send button visibility (#39614)', () => { const flushEditorToDraft = (editor: HTMLDivElement) => { const next = editor.textContent ?? '' + if (next !== draftRef.current) { draftRef.current = next setDraft(next) @@ -130,9 +131,11 @@ describe('composer IME composition — send button visibility (#39614)', () => { if (composingRef.current || event.nativeEvent.isComposing) { return } + if (event.key === 'Enter' && event.keyCode === 229) { return } + if (event.key === 'Enter' && !event.shiftKey) { event.preventDefault() submitCount++ @@ -151,7 +154,7 @@ describe('composer IME composition — send button visibility (#39614)', () => { composingRef.current = true }} onInput={event => { - if (composingRef.current) return + if (composingRef.current) {return} flushEditorToDraft(event.currentTarget) }} onKeyDown={handleKeyDown} From 084fca9dbffd8bca5c6b3a359208fd2b1aeccf4a Mon Sep 17 00:00:00 2001 From: Cornna <96944678+ymylive@users.noreply.github.com> Date: Wed, 3 Jun 2026 20:04:31 +0800 Subject: [PATCH 312/376] fix(tui): clear input after Korean IME submit --- .../__tests__/textInputSubmitClear.test.tsx | 107 ++++++++++++++++++ ui-tui/src/components/textInput.tsx | 25 ++-- 2 files changed, 121 insertions(+), 11 deletions(-) create mode 100644 ui-tui/src/__tests__/textInputSubmitClear.test.tsx diff --git a/ui-tui/src/__tests__/textInputSubmitClear.test.tsx b/ui-tui/src/__tests__/textInputSubmitClear.test.tsx new file mode 100644 index 0000000000000..358e3a55e53ec --- /dev/null +++ b/ui-tui/src/__tests__/textInputSubmitClear.test.tsx @@ -0,0 +1,107 @@ +import { EventEmitter } from 'node:events' +import { PassThrough } from 'node:stream' + +import { renderSync } from '@hermes/ink' +import React, { useState } from 'react' +import { describe, expect, it, vi } from 'vitest' + +import { TextInput } from '../components/textInput.js' + +class FakeInput extends EventEmitter { + chunks: string[] = [] + isRaw = false + isTTY = true + readableLength = 0 + + read() { + const next = this.chunks.shift() ?? null + this.readableLength = this.chunks.length + + return next + } + + ref = vi.fn() + + send(...chunks: string[]) { + this.chunks.push(...chunks) + this.readableLength = this.chunks.length + this.emit('readable') + } + + setEncoding = vi.fn() + + setRawMode = vi.fn((enabled: boolean) => { + this.isRaw = enabled + }) + + unref = vi.fn() +} + +const settle = (ms = 0) => new Promise(resolve => setTimeout(resolve, ms)) + +function makeStreams() { + const stdin = new FakeInput() + const stdout = new PassThrough() + const stderr = new PassThrough() + + Object.assign(stdout, { columns: 80, isTTY: false, rows: 24 }) + Object.assign(stderr, { columns: 80, isTTY: false, rows: 24 }) + + return { stderr, stdin, stdout } +} + +describe('TextInput submit clearing', () => { + it('accepts the parent clear after a Korean IME commit immediately followed by Enter', async () => { + const streams = makeStreams() + const changes: string[] = [] + const submits: string[] = [] + + function Harness() { + const [value, setValue] = useState('') + + return ( + { + changes.push(next) + setValue(next) + }} + onSubmit={text => { + submits.push(text) + setValue('') + }} + value={value} + /> + ) + } + + const instance = renderSync(React.createElement(Harness), { + patchConsole: false, + stderr: streams.stderr as NodeJS.WriteStream, + stdin: streams.stdin as unknown as NodeJS.ReadStream, + stdout: streams.stdout as NodeJS.WriteStream + }) + + await settle() + + const prefix = '한글을 사용하면 마지막 문자가 남아있는 버그가 있어 리포트해' + const finalSyllable = '줘' + const full = prefix + finalSyllable + + streams.stdin.send(prefix) + await settle(25) + + streams.stdin.send(finalSyllable, '\r') + await settle(25) + + streams.stdin.send('x') + await settle(25) + + instance.unmount() + instance.cleanup() + + expect(submits).toEqual([full]) + expect(changes.at(-1)).toBe('x') + expect(changes).not.toContain(`${full}x`) + }) +}) diff --git a/ui-tui/src/components/textInput.tsx b/ui-tui/src/components/textInput.tsx index eb202ca1ba873..dc8cf3fac0a8b 100644 --- a/ui-tui/src/components/textInput.tsx +++ b/ui-tui/src/components/textInput.tsx @@ -736,18 +736,21 @@ export function TextInput({ }, [cur, display, focus, nativeCursor, placeholder, placeholderColor, selected]) useEffect(() => { - if (self.current) { - self.current = false - } else { - setCur(value.length) - setSel(null) - curRef.current = value.length - selRef.current = null - vRef.current = value - lineWidthRef.current = stringWidth(value.includes('\n') ? value.slice(value.lastIndexOf('\n') + 1) : value) - undo.current = [] - redo.current = [] + const ownEcho = self.current && value === vRef.current + self.current = false + + if (ownEcho) { + return } + + setCur(value.length) + setSel(null) + curRef.current = value.length + selRef.current = null + vRef.current = value + lineWidthRef.current = stringWidth(value.includes('\n') ? value.slice(value.lastIndexOf('\n') + 1) : value) + undo.current = [] + redo.current = [] }, [value]) useEffect(() => { From 834a9fc89d498f9c45d2ca9c48240ffcda75af76 Mon Sep 17 00:00:00 2001 From: InphinitiZ Date: Fri, 5 Jun 2026 17:13:15 +0800 Subject: [PATCH 313/376] fix(tui): preserve IME text before return submit Preserve printable IME commit text when xterm delivers it in the same input burst as Return, so Dashboard/TUI submits the visible draft instead of dropping the final segment. Also fixes the TUI type-check stdio tuple typing and adds focused regression coverage. --- .../__tests__/textInputReturnBurst.test.ts | 25 ++++++++++++++ ui-tui/src/components/textInput.tsx | 33 +++++++++++++++++-- 2 files changed, 56 insertions(+), 2 deletions(-) create mode 100644 ui-tui/src/__tests__/textInputReturnBurst.test.ts diff --git a/ui-tui/src/__tests__/textInputReturnBurst.test.ts b/ui-tui/src/__tests__/textInputReturnBurst.test.ts new file mode 100644 index 0000000000000..8470623b8215b --- /dev/null +++ b/ui-tui/src/__tests__/textInputReturnBurst.test.ts @@ -0,0 +1,25 @@ +import { describe, expect, it } from 'vitest' + +import { valueForReturnSubmit } from '../components/textInput.js' + +describe('valueForReturnSubmit', () => { + it('includes printable input that arrives in the same keypress as return', () => { + expect(valueForReturnSubmit('为什么打字上屏,', 8, '会丢失内容')).toEqual({ + cursor: 13, + value: '为什么打字上屏,会丢失内容' + }) + }) + + it('keeps IME commit text when it arrives in the same burst as return', () => { + expect(valueForReturnSubmit('为什么打字上屏,', 8, '会丢失内容\r')).toEqual({ + cursor: 13, + value: '为什么打字上屏,会丢失内容' + }) + }) + + it('leaves the draft unchanged when return carries no printable input', () => { + expect(valueForReturnSubmit('hello', 5, '')).toEqual({ cursor: 5, value: 'hello' }) + expect(valueForReturnSubmit('hello', 5, '\r')).toEqual({ cursor: 5, value: 'hello' }) + expect(valueForReturnSubmit('hello', 5, '\n')).toEqual({ cursor: 5, value: 'hello' }) + }) +}) diff --git a/ui-tui/src/components/textInput.tsx b/ui-tui/src/components/textInput.tsx index dc8cf3fac0a8b..423547ca8f9f5 100644 --- a/ui-tui/src/components/textInput.tsx +++ b/ui-tui/src/components/textInput.tsx @@ -157,6 +157,33 @@ export function applyPrintableInsert( export const shouldRouteMultiCharInputAsPaste = (text: string): boolean => text.includes('\n') +export function valueForReturnSubmit( + value: string, + cursor: number, + input: string, + range?: { end: number; start: number } | null +): TextInsertResult { + const pending = input.replace(BRACKET_PASTE, '').replace(/\r\n/g, '\n').replace(/\r/g, '\n') + + if (!pending) { + return { cursor, value } + } + + // Browser/xterm IME commits can arrive as one burst immediately followed by + // Return (for example "会丢失内容\r"). The Return keypath is already about to + // submit, but the committed text has not passed through the ordinary + // printable-input branch yet. Preserve the printable prefix before the first + // newline so the visible, just-committed IME text is part of the submitted + // prompt instead of being silently dropped. + const [beforeReturn] = pending.split('\n', 1) + + if (!beforeReturn) { + return { cursor, value } + } + + return applyPrintableInsert(value, cursor, beforeReturn, range) ?? { cursor, value } +} + export function shouldPreserveCtrlJNewline(env: MinimalEnv = process.env): boolean { if (env.WT_SESSION) { return true @@ -1163,13 +1190,15 @@ export function TextInput({ if (k.return) { flushKeyBurst() + const range = selRange() + const pending = valueForReturnSubmit(vRef.current, curRef.current, inp, range) const sequence = (event.keypress as { sequence?: string }).sequence const preserveBareLineFeed = shouldPreserveCtrlJNewline() && sequence === '\n' if (k.shift || k.ctrl || preserveBareLineFeed || (isMac ? isActionMod(k) : k.meta)) { - commit(ins(vRef.current, curRef.current, '\n'), curRef.current + 1) + commit(ins(pending.value, pending.cursor, '\n'), pending.cursor + 1) } else { - cbSubmit.current?.(vRef.current) + cbSubmit.current?.(pending.value) } return From 5dec501c6e2fdf6fd9cb2a7322b351854c4382f8 Mon Sep 17 00:00:00 2001 From: hanhvs Date: Tue, 30 Jun 2026 12:01:19 +0700 Subject: [PATCH 314/376] fix(tui): stop Vietnamese Telex IME from dropping characters MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Third-party Vietnamese IMEs (OpenKey/Unikey/EVKey in Telex mode) recompose a syllable by emitting an erase burst followed by the finished characters. Two layers of the TUI input pipeline mishandled this, dropping letters and leaving a stray space mid-syllable (e.g. "hạnh" rendered as "hạ ", and "vương sỹ hạnh" as "vương sỹ hạ "). Root causes, both confirmed from real captured byte streams: 1. parse-keypress: an IME often fuses a control byte (\x7f/\b, or even the U+202F marker OpenKey injects) with the recomposed text in a single stdin read. parseKeypress only recognizes a control key when the whole string is exactly that byte, so a mixed chunk fell through every branch, returned name:"" with a non-printable sequence, and the composer's printable gate discarded the entire chunk — taking the surrounding letters with it. Split text tokens on every control byte so the printable runs survive. CR/LF are deliberately not split, preserving paste/return semantics. 2. textInput: multi-character (IME/paste) inserts were committed through the 16ms deferred key-burst path, which raced an interleaved re-render and snapped the buffer back to a stale value, dropping the recomposed tail. Commit them synchronously. Additionally, the fast-echo "\b \b" backspace shortcut desynced the screen when it ran right after an Ink repaint (forced by the U+202F marker), stranding the marker glyph; suppress fast-echo for the recompose burst that follows an Ink repaint and resume it on the next real keystroke. Tested with real OpenKey and EVKey captures of "vương sỹ hạnh" across read timings, plus parser unit coverage and an EVKey no-regression guard. --- .../src/ink/parse-keypress-drop-probe.test.ts | 71 +++++++++ .../src/ink/parse-keypress-noregress.test.ts | 40 +++++ .../hermes-ink/src/ink/parse-keypress.test.ts | 65 ++++++++ .../hermes-ink/src/ink/parse-keypress.ts | 54 ++++++- .../src/__tests__/imeVietnameseTelex.test.tsx | 141 ++++++++++++++++++ ui-tui/src/components/textInput.tsx | 51 ++++++- 6 files changed, 419 insertions(+), 3 deletions(-) create mode 100644 ui-tui/packages/hermes-ink/src/ink/parse-keypress-drop-probe.test.ts create mode 100644 ui-tui/packages/hermes-ink/src/ink/parse-keypress-noregress.test.ts create mode 100644 ui-tui/src/__tests__/imeVietnameseTelex.test.tsx diff --git a/ui-tui/packages/hermes-ink/src/ink/parse-keypress-drop-probe.test.ts b/ui-tui/packages/hermes-ink/src/ink/parse-keypress-drop-probe.test.ts new file mode 100644 index 0000000000000..6cc013476dbba --- /dev/null +++ b/ui-tui/packages/hermes-ink/src/ink/parse-keypress-drop-probe.test.ts @@ -0,0 +1,71 @@ +import { describe, expect, it } from 'vitest' + +import { INITIAL_STATE, parseMultipleKeypresses } from './parse-keypress.js' + +// Probe: feed many exotic IME-ish byte patterns straight through the parser +// and assert NO printable character codepoint silently vanishes. This catches +// the "chunk falls through every branch -> name:'' with a non-printable +// sequence -> composer discards it" failure class for sequences we haven't +// hand-enumerated. + +function keysToText(keys: Array<{ name?: string; sequence?: string }>): string { + // Reconstruct what the composer would insert: backspaces delete, everything + // else with a printable sequence inserts its sequence. + let out = '' + for (const k of keys) { + if (k.name === 'backspace') { + out = out.slice(0, -1) + continue + } + const seq = k.sequence ?? '' + // Mirror the composer's PRINTABLE gate + if (/^[ -~\u00a0-\uffff]+$/.test(seq)) { + out += seq + } else if (seq) { + // Non-printable, non-backspace => composer drops it. Mark it so the + // assertion can show what was lost. + out += `«DROP:${[...seq].map(c => 'U+' + c.codePointAt(0)!.toString(16)).join(',')}»` + } + } + return out +} + +const cases: Array<[string, string, string]> = [ + // [label, input bytes, expected text after composer-emulation] + ['fused bs+char', '\x7fô', 'ô'], // starts empty, bs no-ops in our emul + ['fused bs+2char', '\x7fôi', 'ôi'], + ['embedded bs', 'ab\bç', 'aç'], + ['hard-erase \\b \\b + char', '\b \bô', 'ô'], + ['hard-erase x3 + ạnh (from anh)', 'anh\b \b\b \b\b \bạnh', 'ạnh'], + ['DEL-space-DEL + char', '\x7f \x7fô', 'ô'], + ['trailing text after multi DEL', '\x7f\x7f\x7fươn', 'ươn'], + ['char then DEL then char fused', 'o\x7fô', 'ô'], + ['multiple syllable fused', 'vuon\x7f\x7f\x7fương', 'vương'], + // CR/LF are intentionally NOT split (preserve paste/return semantics), so a + // text token with an embedded CR is left whole; assert it is NOT split into + // surviving letters here — that path is covered by the composer's return / + // paste handling, not parseTextKeypresses. +] + +describe('parser does not silently drop printable codepoints', () => { + for (const [label, input, expected] of cases) { + it(label, () => { + const [keys] = parseMultipleKeypresses(INITIAL_STATE, input) + const text = keysToText(keys as Array<{ name?: string; sequence?: string }>) + expect(text, `keys=${JSON.stringify(keys)}`).toBe(expected) + }) + } + + it('exhaustive: DEL between every pair of letters never drops a letter', () => { + const letters = [...'aăâeêioôơuưy'] + for (const a of letters) { + for (const b of letters) { + const input = `${a}\x7f${b}` + const [keys] = parseMultipleKeypresses(INITIAL_STATE, input) + const text = keysToText(keys as Array<{ name?: string; sequence?: string }>) + // a inserted, bs removes a, b inserted => "b" + expect(text, `input=${JSON.stringify(input)} keys=${JSON.stringify(keys)}`).toBe(b) + } + } + }) +}) diff --git a/ui-tui/packages/hermes-ink/src/ink/parse-keypress-noregress.test.ts b/ui-tui/packages/hermes-ink/src/ink/parse-keypress-noregress.test.ts new file mode 100644 index 0000000000000..ee6c384a322ee --- /dev/null +++ b/ui-tui/packages/hermes-ink/src/ink/parse-keypress-noregress.test.ts @@ -0,0 +1,40 @@ +import { describe, expect, it } from 'vitest' + +import { INITIAL_STATE, parseMultipleKeypresses } from './parse-keypress.js' + +// Confirm the control-byte split is a NO-OP for clean input (EVKey-style: +// backspace and recomposed text arrive in separate, non-fused reads). The +// fix must not change behavior for any text token that has no embedded +// control byte — otherwise it could regress IMEs that already work. +describe('control-byte split does not touch clean (EVKey-style) input', () => { + it('a plain printable text token yields exactly one keypress (no spurious split)', () => { + const [keys] = parseMultipleKeypresses(INITIAL_STATE, 'ạnh') + + expect(keys).toHaveLength(1) + expect(keys[0]).toMatchObject({ raw: 'ạnh' }) + }) + + it('a lone backspace read is unchanged', () => { + const [keys] = parseMultipleKeypresses(INITIAL_STATE, '\x7f') + + expect(keys).toHaveLength(1) + expect(keys[0]).toMatchObject({ name: 'backspace' }) + }) + + it('separate clean reads (bs read, then text read) each produce one key', () => { + const [k1] = parseMultipleKeypresses(INITIAL_STATE, '\x7f') + const [k2] = parseMultipleKeypresses(INITIAL_STATE, 'ô') + + expect(k1).toHaveLength(1) + expect(k1[0]).toMatchObject({ name: 'backspace' }) + expect(k2).toHaveLength(1) + expect(k2[0]).toMatchObject({ raw: 'ô' }) + }) + + it('a full clean Vietnamese word with no embedded control bytes is one text key', () => { + const [keys] = parseMultipleKeypresses(INITIAL_STATE, 'vương') + + expect(keys).toHaveLength(1) + expect(keys[0]).toMatchObject({ raw: 'vương' }) + }) +}) diff --git a/ui-tui/packages/hermes-ink/src/ink/parse-keypress.test.ts b/ui-tui/packages/hermes-ink/src/ink/parse-keypress.test.ts index fcd7090b8b0b8..aa07bf9a0975a 100644 --- a/ui-tui/packages/hermes-ink/src/ink/parse-keypress.test.ts +++ b/ui-tui/packages/hermes-ink/src/ink/parse-keypress.test.ts @@ -40,6 +40,71 @@ describe('parseMultipleKeypresses bracketed paste recovery', () => { }) }) +describe('parseMultipleKeypresses text control splitting', () => { + it('keeps an IME backspace plus composed character in the same read', () => { + const [keys, state] = parseMultipleKeypresses(INITIAL_STATE, '\x7fô') + + expect(keys).toEqual([ + expect.objectContaining({ name: 'backspace', raw: '\x7f' }), + expect.objectContaining({ name: '', raw: 'ô' }) + ]) + expect(state.mode).toBe('NORMAL') + }) + + it('keeps trailing IME text after a backspace in the same read', () => { + const [keys] = parseMultipleKeypresses(INITIAL_STATE, '\x7fôi') + + expect(keys).toEqual([ + expect.objectContaining({ name: 'backspace', raw: '\x7f' }), + expect.objectContaining({ name: '', raw: 'ôi' }) + ]) + }) + + it('splits embedded backspace control bytes without splitting surrounding text', () => { + const [keys] = parseMultipleKeypresses(INITIAL_STATE, 'ab\bç') + + expect(keys).toEqual([ + expect.objectContaining({ name: '', raw: 'ab' }), + expect.objectContaining({ name: 'backspace', raw: '\b' }), + expect.objectContaining({ name: '', raw: 'ç' }) + ]) + }) + + it('peels off a non-backspace control byte fused with text instead of dropping the whole chunk', () => { + // An IME can fuse a control byte other than \x7f/\b with the recomposed + // text (here U+0001). The original PR only split on \x7f/\b, so a chunk + // like "a\x01b" fell through every parseKeypress branch, returned + // name:"" with a non-printable sequence, and the composer discarded the + // entire chunk — eating the printable letters 'a' and 'b' too. Every + // control byte must be peeled off so the surrounding text survives. + const [keys] = parseMultipleKeypresses(INITIAL_STATE, 'a\x01b') + + // The leading and trailing printable letters must each survive as their + // own keypress (the control byte in between parses to ctrl+a). The bug was + // the WHOLE "a\x01b" chunk collapsing into one undeliverable key. + expect(keys).toHaveLength(3) + expect(keys[0]).toMatchObject({ name: 'a', raw: 'a' }) + expect(keys[1]).toMatchObject({ raw: '\x01' }) + expect(keys[2]).toMatchObject({ name: 'b', raw: 'b' }) + }) + + it('keeps printable letters around a fused ESC control byte', () => { + const [keys] = parseMultipleKeypresses(INITIAL_STATE, 'vương\x1b') + + // The trailing printable run must still be delivered as its own key. + expect(keys.some(k => 'raw' in k && k.raw === 'vương')).toBe(true) + }) + + it('does NOT split embedded CR/LF (preserves paste/return handling)', () => { + // CR/LF inside a text token come from non-bracketed paste; splitting them + // into `return` keys would prematurely submit the composer. They must stay + // inside the single text token. + const [keys] = parseMultipleKeypresses(INITIAL_STATE, 'a\rb') + + expect(keys).toEqual([expect.objectContaining({ raw: 'a\rb' })]) + }) +}) + describe('mouse wheel modifier decoding', () => { // SGR mouse format: ESC [ < button ; col ; row M // Wheel up = 64 (0x40), wheel down = 65 (0x41). diff --git a/ui-tui/packages/hermes-ink/src/ink/parse-keypress.ts b/ui-tui/packages/hermes-ink/src/ink/parse-keypress.ts index 59981f543fbdb..07e31c6f53959 100644 --- a/ui-tui/packages/hermes-ink/src/ink/parse-keypress.ts +++ b/ui-tui/packages/hermes-ink/src/ink/parse-keypress.ts @@ -200,6 +200,58 @@ function splitNumericParams(params: string): number[] { return params.split(';').map(p => parseInt(p, 10)) } +// A text token can carry stray control bytes fused with printable input — +// most commonly when a third-party IME (Vietnamese Telex via OpenKey/Unikey/ +// EVKey, etc.) recomposes a syllable by emitting an erase control byte +// immediately followed by the finished character(s) in a single stdin read +// (e.g. "\x7fô", "ab\bç"). parseKeypress only recognizes a control key when +// the WHOLE string is exactly that control byte, so a mixed chunk falls +// through every branch and returns name:"" with a non-printable sequence, +// which the composer's PRINTABLE gate then discards — taking the surrounding +// letters down with it. Split the token so every control byte becomes its own +// keypress and the printable runs between them survive. +// +// CR (\r) and LF (\n) are deliberately NOT treated as split points: a lone +// Enter already arrives as its own read, while a newline embedded in a text +// token only happens for non-bracketed paste, where peeling it into a +// `return` keypress would prematurely submit the composer. Leaving them in +// the token preserves the existing paste/return handling byte-for-byte. +function isControlChar(ch: string): boolean { + const code = ch.charCodeAt(0) + + if (code === 0x0a || code === 0x0d) { + return false + } + + return code < 0x20 || code === 0x7f +} + +function parseTextKeypresses(text: string): ParsedKey[] { + const keys: ParsedKey[] = [] + let textStart = 0 + + for (let i = 0; i < text.length; i++) { + const ch = text[i]! + + if (!isControlChar(ch)) { + continue + } + + if (i > textStart) { + keys.push(parseKeypress(text.slice(textStart, i))) + } + + keys.push(parseKeypress(ch)) + textStart = i + 1 + } + + if (textStart < text.length) { + keys.push(parseKeypress(text.slice(textStart))) + } + + return keys +} + export type KeyParseState = { mode: 'NORMAL' | 'IN_PASTE' incomplete: string @@ -294,7 +346,7 @@ export function parseMultipleKeypresses( const resynthesized = '\x1b' + token.value keys.push(parseKeypress(resynthesized)) } else { - keys.push(parseKeypress(token.value)) + keys.push(...parseTextKeypresses(token.value)) } } } diff --git a/ui-tui/src/__tests__/imeVietnameseTelex.test.tsx b/ui-tui/src/__tests__/imeVietnameseTelex.test.tsx new file mode 100644 index 0000000000000..74499f639f692 --- /dev/null +++ b/ui-tui/src/__tests__/imeVietnameseTelex.test.tsx @@ -0,0 +1,141 @@ +import { EventEmitter } from 'events' + +import { renderSync } from '@hermes/ink' +import React, { useState } from 'react' +import { describe, expect, it } from 'vitest' + +import { TextInput } from '../components/textInput.js' + +// End-to-end regression coverage for Vietnamese Telex IME recomposition +// (OpenKey / Unikey / EVKey). These IMEs commit a finished syllable by +// emitting a burst of backspaces (and, for OpenKey, a U+202F NARROW NO-BREAK +// SPACE marker) followed by the recomposed characters. The byte streams below +// are real captures taken from OpenKey and EVKey on macOS while typing the +// phrase "vương sỹ hạnh" (Telex: "vuonwg syx hanhj"). +// +// The bug these guard against: characters were dropped and a stray space was +// left mid-syllable (e.g. "hạnh" rendered as "hạ "). Root causes fixed: +// 1. parse-keypress split fused control-byte+text chunks so the recomposed +// text survives instead of being discarded with the control byte. +// 2. textInput commits multi-character (IME/paste) inserts synchronously +// instead of through the 16ms key-burst path that raced re-renders. + +class FakeTty extends EventEmitter { + chunks: string[] = [] + columns = 80 + rows = 24 + isTTY = true + isRaw = false + private pendingReads: string[] = [] + ref(): void {} + unref(): void {} + read(): string | null { + return this.pendingReads.shift() ?? null + } + send(chunk: string): void { + this.pendingReads.push(chunk) + this.emit('readable') + } + setEncoding(): this { + return this + } + setRawMode(mode: boolean): this { + this.isRaw = mode + + return this + } + write(chunk: string | Uint8Array, cb?: (err?: Error | null) => void): boolean { + this.chunks.push(typeof chunk === 'string' ? chunk : Buffer.from(chunk).toString('utf8')) + cb?.() + + return true + } +} + +const tick = () => new Promise(resolve => setImmediate(resolve)) +const wait = (ms: number) => new Promise(resolve => setTimeout(resolve, ms)) + +function Harness({ initial = '', onValue }: { initial?: string; onValue: (value: string) => void }) { + const [value, setValue] = useState(initial) + + return React.createElement(TextInput, { + onChange: (next: string) => { + setValue(next) + onValue(next) + }, + value + }) +} + +async function drive(reads: string[], { initial = '', gapMs = 0 }: { initial?: string; gapMs?: number } = {}): Promise { + const stdout = new FakeTty() + const stdin = new FakeTty() + const stderr = new FakeTty() + const values: string[] = [] + + const instance = renderSync(React.createElement(Harness, { initial, onValue: v => values.push(v) }), { + patchConsole: false, + stderr: stderr as unknown as NodeJS.WriteStream, + stdin: stdin as unknown as NodeJS.ReadStream, + stdout: stdout as unknown as NodeJS.WriteStream + }) + + try { + await tick() + + for (const r of reads) { + stdin.send(r) + await tick() + + if (gapMs) { + await wait(gapMs) + } + } + + await wait(60) + + return values.at(-1) ?? '' + } finally { + instance.unmount() + instance.cleanup() + } +} + +const NNBSP = '\u202f' + +describe('Vietnamese Telex IME recomposition', () => { + it('applies a parser-split backspace plus composed character through useInput', async () => { + // OpenKey fuses the erase + recomposed glyph into a single stdin read. + expect(await drive(['\x7fô'], { initial: 'o' })).toBe('ô') + }) + + it('commits a multi-character recompose synchronously (no dropped tail)', async () => { + // "hanhj" -> a U+202F marker, four backspaces, then the recomposed "ạnh". + // Only a single microtask after the last read — the sync commit must have + // already delivered the final value (the deferred path dropped "nh" here). + const reads = ['h', 'a', 'n', 'h', NNBSP, '\x7f\x7f', '\x7f\x7f', '\u1EA1nh'] + + expect(await drive(reads)).toBe('h\u1EA1nh') + }) + + it('reproduces the full phrase "vương sỹ hạnh" from a real OpenKey capture', async () => { + // Captured byte stream for Telex "vuonwg syx hanhj": each syllable injects a + // U+202F marker, erases, and re-emits. Verified across read timings. + const reads = [ + 'v', 'u', 'o', NNBSP, '\x7f\x7f', '\x7f\u01B0\u01A1', 'n', 'g', + ' ', 's', 'y', NNBSP, '\x7f', '\x7f\u1EF9', + ' ', 'h', 'a', 'n', 'h', NNBSP, '\x7f\x7f\x7f\x7f\u1EA1nh' + ] + + for (const gapMs of [0, 17, 25]) { + expect(await drive(reads, { gapMs })).toBe('vương sỹ hạnh') + } + }) + + it('handles the EVKey capture (clean backspaces, no marker) for "hạnh"', async () => { + // EVKey emits three clean backspaces and no U+202F; must also yield "hạnh". + const reads = ['h', 'a', 'n', 'h', '\x7f', '\x7f', '\x7f', '\u1EA1nh'] + + expect(await drive(reads)).toBe('h\u1EA1nh') + }) +}) diff --git a/ui-tui/src/components/textInput.tsx b/ui-tui/src/components/textInput.tsx index 423547ca8f9f5..1bb0c88282cd4 100644 --- a/ui-tui/src/components/textInput.tsx +++ b/ui-tui/src/components/textInput.tsx @@ -666,6 +666,15 @@ export function TextInput({ const parentChangeTimer = useRef | null>(null) const pendingParentValue = useRef(null) const localRenderTimer = useRef | null>(null) + // True for one keystroke after a commit took the full Ink render path + // (syncParent). Ink repaints the whole input line, so the terminal cursor + // baseline that the fast-echo "\b \b" shortcut assumes is no longer valid; + // a fast-echo backspace fired right after an Ink repaint desyncs the screen + // and strands glyphs (the OpenKey Vietnamese "hạ␣␣" bug: an injected U+202F + // marker forces an Ink repaint, then the recompose backspaces fast-echo + // against a stale baseline). Suppress fast-echo for that one next edit. + const inkRepaintedRef = useRef(false) + const inkRepaintResetTimer = useRef | null>(null) const lineWidthRef = useRef(stringWidth(value.includes('\n') ? value.slice(value.lastIndexOf('\n') + 1) : value)) const mouseAnchorRef = useRef(null) const lastClickRef = useRef<{ at: number; offset: number }>({ at: 0, offset: -1 }) @@ -822,6 +831,10 @@ export function TextInput({ if (localRenderTimer.current) { clearTimeout(localRenderTimer.current) } + + if (inkRepaintResetTimer.current) { + clearTimeout(inkRepaintResetTimer.current) + } }, [] ) @@ -876,7 +889,7 @@ export function TextInput({ canFastEchoBase() && canFastAppendShape(current, cursor, text, columns, lineWidthRef.current) const canFastBackspace = (current: string, cursor: number) => - canFastEchoBase() && canFastBackspaceShape(current, cursor, columns) + !inkRepaintedRef.current && canFastEchoBase() && canFastBackspaceShape(current, cursor, columns) const commit = ( next: string, @@ -922,6 +935,22 @@ export function TextInput({ flushParentChange() self.current = true cbChange.current(next) + // A full Ink repaint just happened. Mark it so any fast-echo backspace + // later in this IME recompose burst is suppressed (it would write + // "\b \b" against a baseline Ink just invalidated, stranding the U+202F + // marker glyph — the "hạ␣␣" bug). IME reads arrive as SEPARATE stdin + // events with small macrotask gaps, so a setTimeout(0) reset would + // clear the flag between reads and miss the very backspaces it must + // guard. Use a short real-time window that spans a recompose burst; + // normal typing re-enables fast-echo via the append path below. + inkRepaintedRef.current = true + if (inkRepaintResetTimer.current) { + clearTimeout(inkRepaintResetTimer.current) + } + inkRepaintResetTimer.current = setTimeout(() => { + inkRepaintResetTimer.current = null + inkRepaintedRef.current = false + }, 60) } else { self.current = true scheduleParentChange(next) @@ -1375,7 +1404,17 @@ export function TextInput({ v = inserted.value c = inserted.cursor - scheduleKeyBurstCommit(v, c) + // Multi-character inserts are IME recompositions or pastes, NOT rapid + // single-key typing. Committing them through the 16ms deferred + // key-burst path opens a race: when an IME recompose arrives as a + // burst of backspaces followed by this text in one stdin read (e.g. + // OpenKey Vietnamese Telex, which injects a U+202F marker then erases + // and re-emits the syllable), the single `self.current` guard can be + // consumed by an interleaved re-render before the deferred commit + // flushes, snapping the buffer back to a stale parent value and + // dropping the recomposed tail (the "hanhj -> hạ␣␣" bug). Commit + // synchronously so the recomposed value reaches the parent atomically. + commit(v, c) return } @@ -1403,6 +1442,14 @@ export function TextInput({ // Same explicit fg as the Ink render (see the ) — // the bypass cell must not flash the terminal-default color. stdout!.write(colorizeEcho(effect.write, color)) + // A real character was just fast-echoed to the screen, so the + // terminal baseline is synced again — clear any pending Ink-repaint + // fast-echo suppression so normal backspace fast-echo resumes. + inkRepaintedRef.current = false + if (inkRepaintResetTimer.current) { + clearTimeout(inkRepaintResetTimer.current) + inkRepaintResetTimer.current = null + } // ASCII-printable text advances the physical cursor by exactly // text.length cells (canFastAppendShape rejects non-ASCII, // wide chars, newlines). Notify Ink so the cached displayCursor From e2f7850ec389518c3762932ec74743cc4c3ffde8 Mon Sep 17 00:00:00 2001 From: hanhvs Date: Wed, 15 Jul 2026 20:12:14 +0700 Subject: [PATCH 315/376] fix(tui): tighten IME test per teknium1 review (#55415) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Remove unconditional 60ms wait that let deferred path pass sync-commit test - Use fake timers (setTimeout/setInterval/Date only, NOT setImmediate) - Assert immediately after final read — no trailing wait/advance - Add deterministic coverage for 60ms fast-echo suppression reset: * suppresses backspace after Ink repaint (IME recompose) * does NOT suppress on normal ASCII typing - Verified: revert sync commit -> deferred path makes 4/6 tests fail --- .../src/__tests__/imeVietnameseTelex.test.tsx | 112 ++++++++++++++++-- 1 file changed, 105 insertions(+), 7 deletions(-) diff --git a/ui-tui/src/__tests__/imeVietnameseTelex.test.tsx b/ui-tui/src/__tests__/imeVietnameseTelex.test.tsx index 74499f639f692..59d2be7782501 100644 --- a/ui-tui/src/__tests__/imeVietnameseTelex.test.tsx +++ b/ui-tui/src/__tests__/imeVietnameseTelex.test.tsx @@ -2,7 +2,7 @@ import { EventEmitter } from 'events' import { renderSync } from '@hermes/ink' import React, { useState } from 'react' -import { describe, expect, it } from 'vitest' +import { describe, expect, it, vi, beforeEach, afterEach } from 'vitest' import { TextInput } from '../components/textInput.js' @@ -41,19 +41,16 @@ class FakeTty extends EventEmitter { } setRawMode(mode: boolean): this { this.isRaw = mode - return this } write(chunk: string | Uint8Array, cb?: (err?: Error | null) => void): boolean { this.chunks.push(typeof chunk === 'string' ? chunk : Buffer.from(chunk).toString('utf8')) cb?.() - return true } } const tick = () => new Promise(resolve => setImmediate(resolve)) -const wait = (ms: number) => new Promise(resolve => setTimeout(resolve, ms)) function Harness({ initial = '', onValue }: { initial?: string; onValue: (value: string) => void }) { const [value, setValue] = useState(initial) @@ -67,7 +64,14 @@ function Harness({ initial = '', onValue }: { initial?: string; onValue: (value: }) } -async function drive(reads: string[], { initial = '', gapMs = 0 }: { initial?: string; gapMs?: number } = {}): Promise { +// Core driver: feeds reads, optionally advancing fake timers between reads to +// simulate the small macrotask gaps real IME reads arrive with. Returns the +// final value seen by the parent immediately after the last read (no trailing +// wait) so a passing assertion proves the commit was synchronous, not deferred. +async function drive( + reads: string[], + { initial = '', gapMs = 0 }: { initial?: string; gapMs?: number } = {} +): Promise { const stdout = new FakeTty() const stdin = new FakeTty() const stderr = new FakeTty() @@ -88,11 +92,18 @@ async function drive(reads: string[], { initial = '', gapMs = 0 }: { initial?: s await tick() if (gapMs) { - await wait(gapMs) + // Advance the fake clock to flush any pending FRAME_BATCH_MS timers + // between reads (mirrors the real macrotask gap), then let microtasks run. + vi.advanceTimersByTime(gapMs) + await tick() } } - await wait(60) + // Assert IMMEDIATELY after the final read — no trailing 60ms wait and + // WITHOUT advancing the fake clock past the deferred key-burst window. + // If the value is already correct here, the multi-char insert committed + // synchronously; the old deferred path (scheduleKeyBurstCommit, 16ms) + // has NOT flushed yet, so a stale/dropped tail would still be visible. return values.at(-1) ?? '' } finally { @@ -104,6 +115,15 @@ async function drive(reads: string[], { initial = '', gapMs = 0 }: { initial?: s const NNBSP = '\u202f' describe('Vietnamese Telex IME recomposition', () => { + beforeEach(() => { + // Only fake setTimeout/setInterval/Date — NOT setImmediate (used by tick()). + vi.useFakeTimers({ toFake: ['setTimeout', 'setInterval', 'Date'] }) + }) + + afterEach(() => { + vi.useRealTimers() + }) + it('applies a parser-split backspace plus composed character through useInput', async () => { // OpenKey fuses the erase + recomposed glyph into a single stdin read. expect(await drive(['\x7fô'], { initial: 'o' })).toBe('ô') @@ -115,6 +135,7 @@ describe('Vietnamese Telex IME recomposition', () => { // already delivered the final value (the deferred path dropped "nh" here). const reads = ['h', 'a', 'n', 'h', NNBSP, '\x7f\x7f', '\x7f\x7f', '\u1EA1nh'] + // No gapMs, no advanceTimersMs — we assert BEFORE the 16ms FRAME_BATCH_MS could fire. expect(await drive(reads)).toBe('h\u1EA1nh') }) @@ -139,3 +160,80 @@ describe('Vietnamese Telex IME recomposition', () => { expect(await drive(reads)).toBe('h\u1EA1nh') }) }) + +describe('Fast-echo suppression reset (60ms window)', () => { + beforeEach(() => { + // Only fake setTimeout/setInterval/Date — NOT setImmediate (used by tick()). + vi.useFakeTimers({ toFake: ['setTimeout', 'setInterval', 'Date'] }) + }) + + afterEach(() => { + vi.useRealTimers() + }) + + it('suppresses fast-echo backspace for one keystroke after an Ink repaint (IME recompose)', async () => { + // Simulate: user types "ha" -> Ink commits normally -> then IME recompose arrives + // as NNBSP + backspaces + recomposed text. The first backspace after the Ink + // repaint must NOT fast-echo (would strand the NNBSP marker as a stray space). + + // Type "ha" normally (each char goes through fast-echo append path) + let reads = ['h', 'a'] + const stdout1 = new FakeTty() + const stdin1 = new FakeTty() + const stderr1 = new FakeTty() + const values1: string[] = [] + + const instance1 = renderSync(React.createElement(Harness, { initial: '', onValue: v => values1.push(v) }), { + patchConsole: false, + stderr: stderr1 as unknown as NodeJS.WriteStream, + stdin: stdin1 as unknown as NodeJS.ReadStream, + stdout: stdout1 as unknown as NodeJS.WriteStream + }) + + try { + await tick() + for (const r of reads) { + stdin1.send(r) + await tick() + } + // After "ha", fast-echo is enabled (inkRepaintedRef.current = false) + expect(values1.at(-1)).toBe('ha') + + // Now simulate an IME recompose burst that forces an Ink repaint: + // NNBSP marker forces a full Ink render (syncParent=true in commit). + // The next backspace should be SUPPRESSED (fast-echo backspace disabled). + stdin1.send(NNBSP + '\x7f\x7f\u1EA1nh') // fused chunk: marker + 2x backspace + "ạnh" + await tick() + + // The recomposed value must be committed synchronously (no dropped tail). + // The first backspace after the Ink repaint must NOT have written "\b \b" to stdout. + // We can't directly inspect stdout here, but we verify the FINAL value is correct. + expect(values1.at(-1)).toBe('h\u1EA1nh') + + // Advance fake timers past the 60ms suppression window so the + // inkRepaintResetTimer fires and re-enables fast-echo backspace. + vi.advanceTimersByTime(60) + await tick() + + // Now fast-echo backspace is RE-ENABLED. One backspace deletes exactly + // one grapheme ("h") off the end of "hạnh" -> "hạn". + stdin1.send('\x7f') + await tick() + + expect(values1.at(-1)).toBe('h\u1EA1n') + } finally { + instance1.unmount() + instance1.cleanup() + } + }) + + it('does NOT suppress fast-echo backspace when no Ink repaint occurred (normal typing)', async () => { + // Normal ASCII typing never triggers the Ink-repaint suppression. + const reads = ['h', 'e', 'l', 'l', 'o'] + expect(await drive(reads)).toBe('hello') + + // Two backspaces off "hello" -> "hel" via the fast-echo path. + const reads2 = [...reads, '\x7f', '\x7f'] + expect(await drive(reads2)).toBe('hel') + }) +}) \ No newline at end of file From 2af4da45ec24d849c1c707eacbc3bd61891c469c Mon Sep 17 00:00:00 2001 From: liuhao1024 Date: Thu, 25 Jun 2026 03:47:28 +0800 Subject: [PATCH 316/376] fix(dashboard): prevent React from dropping first keystroke during IME composition MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit React 18's root-level event delegation intercepts keydown events with keyCode 229 (the "composition in progress" signal sent by the browser during non-Latin IME input) and synthesises an onCompositionStart event. That synthetic path sets internal composing state that interferes with xterm.js's own IME handling on its hidden textarea, causing the first keystroke of each composition chunk to be silently dropped. The fix adds a capture-phase keydown listener on the terminal host div that stops propagation of keyCode-229 events before they reach React's delegation layer. xterm.js relies on native compositionstart/ compositionend on its internal textarea — not on keydown — so blocking the propagation is safe. Fixes #52111 --- web/src/pages/ChatPage.tsx | 25 +++++++++++++++++++++++++ 1 file changed, 25 insertions(+) diff --git a/web/src/pages/ChatPage.tsx b/web/src/pages/ChatPage.tsx index ea6d9f36e5e16..c97d247cc23dd 100644 --- a/web/src/pages/ChatPage.tsx +++ b/web/src/pages/ChatPage.tsx @@ -748,6 +748,30 @@ export default function ChatPage({ isActive = true }: { isActive?: boolean }) { let mobileInputCleanup: (() => void) | null = null; term.open(host); + // IME composition guard (fixes #52111). + // + // React 18's root-level event delegation intercepts keydown events with + // keyCode 229 (the "composition in progress" signal sent by the browser + // during non-Latin IME input) and synthesises an onCompositionStart + // event. That synthetic path sets internal composing state that + // interferes with xterm.js's own IME handling on its hidden textarea, + // causing the first keystroke of each composition chunk to be silently + // dropped — most visible with Cyrillic (Ukrainian/Russian) on + // Firefox-based browsers, but affects any locale that uses composition + // events (CJK, Arabic, Hebrew). + // + // xterm.js relies on native compositionstart/compositionend on its + // internal textarea, not on keydown, so blocking the keyCode-229 + // keydown from reaching React's delegation layer is safe. The listener + // sits in the *capture* phase on the terminal host so it fires before + // the event bubbles up to the React root. + const _imeCompositionGuard = (e: KeyboardEvent) => { + if (e.keyCode === 229 || e.key === "Process") { + e.stopPropagation(); + } + }; + host.addEventListener("keydown", _imeCompositionGuard, true); + const textarea = term.textarea; if (textarea) { textarea.setAttribute("autocomplete", "off"); @@ -1331,6 +1355,7 @@ export default function ChatPage({ isActive = true }: { isActive?: boolean }) { // the ticket fetch resolves and ``wsRef.current`` was never assigned. wsRef.current?.close(); wsRef.current = null; + host.removeEventListener("keydown", _imeCompositionGuard, true); term.dispose(); termRef.current = null; fitRef.current = null; From 278c6ebcb7288b3c65dbaf320f77470532fe3e9f Mon Sep 17 00:00:00 2001 From: Denis H <79399355+dplush@users.noreply.github.com> Date: Sat, 1 Aug 2026 17:24:24 +0200 Subject: [PATCH 317/376] test(dashboard): reproduce dropped dead-key composition input --- web/src/lib/pty-composition.test.ts | 25 +++++++++++++++++++++++++ 1 file changed, 25 insertions(+) create mode 100644 web/src/lib/pty-composition.test.ts diff --git a/web/src/lib/pty-composition.test.ts b/web/src/lib/pty-composition.test.ts new file mode 100644 index 0000000000000..f9c68698a5ba8 --- /dev/null +++ b/web/src/lib/pty-composition.test.ts @@ -0,0 +1,25 @@ +import { describe, expect, it, vi } from "vitest"; + +import { createPtyCompositionForwarder } from "./pty-composition"; + +describe("createPtyCompositionForwarder", () => { + it("forwards committed dead-key text and suppresses xterm's duplicate onData", () => { + const send = vi.fn(); + const forwarder = createPtyCompositionForwarder(send); + + forwarder.onCompositionEnd("ä"); + + expect(send).toHaveBeenCalledExactlyOnceWith("ä"); + expect(forwarder.shouldForwardTerminalData("ä")).toBe(false); + expect(forwarder.shouldForwardTerminalData("x")).toBe(true); + }); + + it("does not send an empty cancelled composition", () => { + const send = vi.fn(); + const forwarder = createPtyCompositionForwarder(send); + + forwarder.onCompositionEnd(""); + + expect(send).not.toHaveBeenCalled(); + }); +}); From 6ece2da7389db8735a24035b09ba87b498530051 Mon Sep 17 00:00:00 2001 From: Denis H <79399355+dplush@users.noreply.github.com> Date: Sat, 1 Aug 2026 17:26:38 +0200 Subject: [PATCH 318/376] fix(dashboard): forward committed dead-key input to PTY --- web/src/lib/pty-composition.ts | 25 +++++++++++++++++++++++++ web/src/pages/ChatPage.tsx | 18 ++++++++++++++++-- 2 files changed, 41 insertions(+), 2 deletions(-) create mode 100644 web/src/lib/pty-composition.ts diff --git a/web/src/lib/pty-composition.ts b/web/src/lib/pty-composition.ts new file mode 100644 index 0000000000000..b44758a7970fc --- /dev/null +++ b/web/src/lib/pty-composition.ts @@ -0,0 +1,25 @@ +/** + * Sends committed IME/dead-key text when xterm does not emit onData. + * + * Some browser/layout combinations leave xterm's CompositionHelper without an + * onData callback. The DOM compositionend event is the authoritative commit. + * If xterm does emit the same bytes afterwards, consume that one duplicate. + */ +export function createPtyCompositionForwarder(send: (data: string) => void) { + let pendingDuplicate: string | null = null; + + return { + onCompositionEnd(data: string) { + if (!data) return; + pendingDuplicate = data; + send(data); + }, + shouldForwardTerminalData(data: string) { + if (data === pendingDuplicate) { + pendingDuplicate = null; + return false; + } + return true; + }, + }; +} diff --git a/web/src/pages/ChatPage.tsx b/web/src/pages/ChatPage.tsx index c97d247cc23dd..1078b5fecad9a 100644 --- a/web/src/pages/ChatPage.tsx +++ b/web/src/pages/ChatPage.tsx @@ -36,6 +36,7 @@ import { usePageHeader } from "@/contexts/usePageHeader"; import { useI18n } from "@/i18n"; import { api } from "@/lib/api"; import { latchChatActivation } from "@/lib/chat-activation"; +import { createPtyCompositionForwarder } from "@/lib/pty-composition"; import { normalizeSessionTitle } from "@/lib/chat-title"; import { PtyResumeSanitizer } from "@/lib/pty-resume-sanitizer"; import { @@ -746,6 +747,12 @@ export default function ChatPage({ isActive = true }: { isActive?: boolean }) { term.loadAddon(new WebLinksAddon()); let mobileInputCleanup: (() => void) | null = null; + // xterm occasionally drops committed dead-key/IME text instead of emitting + // onData. The compositionend event supplies the authoritative text. + let sendComposedText: (data: string) => void = () => undefined; + const compositionForwarder = createPtyCompositionForwarder((data) => { + sendComposedText(data); + }); term.open(host); // IME composition guard (fixes #52111). @@ -794,8 +801,9 @@ export default function ChatPage({ isActive = true }: { isActive?: boolean }) { mobileReplacementInputUntilRef.current = Date.now() + MOBILE_REPLACEMENT_WINDOW_MS; } }; - const markCompositionEnd = () => { + const markCompositionEnd = (ev: Event) => { mobileReplacementInputUntilRef.current = Date.now() + MOBILE_REPLACEMENT_WINDOW_MS; + compositionForwarder.onCompositionEnd((ev as CompositionEvent).data); }; textarea.addEventListener("beforeinput", markReplacementInput, true); @@ -1270,7 +1278,7 @@ export default function ChatPage({ isActive = true }: { isActive?: boolean }) { // behave normally. // eslint-disable-next-line no-control-regex -- intentional ESC byte in xterm SGR mouse report parser const SGR_MOUSE_RE = /^\x1b\[<(\d+);(\d+);(\d+)([Mm])$/; - onDataDisposable = term.onData((data) => { + const forwardPtyData = (data: string) => { // Mouse reports (scroll wheel etc.) are not typed input — swallow // them before the blocked-input check so scrolling a disconnected // terminal doesn't trip the "reconnecting" notice. @@ -1301,6 +1309,12 @@ export default function ChatPage({ isActive = true }: { isActive?: boolean }) { mobileReplacementInputUntilRef.current = 0; } ws.send(normalized.data); + }; + sendComposedText = forwardPtyData; + onDataDisposable = term.onData((data) => { + if (compositionForwarder.shouldForwardTerminalData(data)) { + forwardPtyData(data); + } }); onResizeDisposable = term.onResize(({ cols, rows }) => { From 0924e33369074fb2eab2fb28d0bb3db28d7f2800 Mon Sep 17 00:00:00 2001 From: Denis H <79399355+dplush@users.noreply.github.com> Date: Sat, 1 Aug 2026 17:29:20 +0200 Subject: [PATCH 319/376] fix(dashboard): defer IME fallback until xterm input settles --- web/src/lib/pty-composition.test.ts | 24 +++++++++++++++--- web/src/lib/pty-composition.ts | 38 +++++++++++++++++++---------- web/src/pages/ChatPage.tsx | 21 +++++++++------- 3 files changed, 57 insertions(+), 26 deletions(-) diff --git a/web/src/lib/pty-composition.test.ts b/web/src/lib/pty-composition.test.ts index f9c68698a5ba8..773c8be9d80dd 100644 --- a/web/src/lib/pty-composition.test.ts +++ b/web/src/lib/pty-composition.test.ts @@ -1,24 +1,40 @@ -import { describe, expect, it, vi } from "vitest"; +import { afterEach, describe, expect, it, vi } from "vitest"; import { createPtyCompositionForwarder } from "./pty-composition"; describe("createPtyCompositionForwarder", () => { - it("forwards committed dead-key text and suppresses xterm's duplicate onData", () => { + afterEach(() => vi.useRealTimers()); + + it("forwards committed dead-key text when xterm emits no onData", () => { + vi.useFakeTimers(); const send = vi.fn(); const forwarder = createPtyCompositionForwarder(send); forwarder.onCompositionEnd("ä"); + vi.runAllTimers(); expect(send).toHaveBeenCalledExactlyOnceWith("ä"); - expect(forwarder.shouldForwardTerminalData("ä")).toBe(false); - expect(forwarder.shouldForwardTerminalData("x")).toBe(true); + }); + + it("leaves xterm's committed input alone when it arrives before the fallback", () => { + vi.useFakeTimers(); + const send = vi.fn(); + const forwarder = createPtyCompositionForwarder(send); + + forwarder.onCompositionEnd("ä"); + forwarder.noteTerminalData("äx"); + vi.runAllTimers(); + + expect(send).not.toHaveBeenCalled(); }); it("does not send an empty cancelled composition", () => { + vi.useFakeTimers(); const send = vi.fn(); const forwarder = createPtyCompositionForwarder(send); forwarder.onCompositionEnd(""); + vi.runAllTimers(); expect(send).not.toHaveBeenCalled(); }); diff --git a/web/src/lib/pty-composition.ts b/web/src/lib/pty-composition.ts index b44758a7970fc..329cb4c173ef2 100644 --- a/web/src/lib/pty-composition.ts +++ b/web/src/lib/pty-composition.ts @@ -1,25 +1,37 @@ /** - * Sends committed IME/dead-key text when xterm does not emit onData. + * Delays an IME/dead-key commit just long enough for xterm to emit onData. * - * Some browser/layout combinations leave xterm's CompositionHelper without an - * onData callback. The DOM compositionend event is the authoritative commit. - * If xterm does emit the same bytes afterwards, consume that one duplicate. + * xterm is authoritative when it emits the commit. Browsers/layouts where it + * does not emit onData still forward the compositionend text on the next turn. */ export function createPtyCompositionForwarder(send: (data: string) => void) { - let pendingDuplicate: string | null = null; + let pending: string | null = null; + let timer: ReturnType | null = null; + + const clearPending = () => { + pending = null; + if (timer) { + clearTimeout(timer); + timer = null; + } + }; return { - onCompositionEnd(data: string) { + onCompositionEnd(data: string | null) { if (!data) return; - pendingDuplicate = data; - send(data); + clearPending(); + pending = data; + timer = setTimeout(() => { + const committed = pending; + clearPending(); + if (committed) send(committed); + }, 0); }, - shouldForwardTerminalData(data: string) { - if (data === pendingDuplicate) { - pendingDuplicate = null; - return false; + noteTerminalData(data: string) { + if (pending && data.startsWith(pending)) { + clearPending(); } - return true; }, + dispose: clearPending, }; } diff --git a/web/src/pages/ChatPage.tsx b/web/src/pages/ChatPage.tsx index 1078b5fecad9a..b53f4086d620a 100644 --- a/web/src/pages/ChatPage.tsx +++ b/web/src/pages/ChatPage.tsx @@ -36,8 +36,8 @@ import { usePageHeader } from "@/contexts/usePageHeader"; import { useI18n } from "@/i18n"; import { api } from "@/lib/api"; import { latchChatActivation } from "@/lib/chat-activation"; -import { createPtyCompositionForwarder } from "@/lib/pty-composition"; import { normalizeSessionTitle } from "@/lib/chat-title"; +import { createPtyCompositionForwarder } from "@/lib/pty-composition"; import { PtyResumeSanitizer } from "@/lib/pty-resume-sanitizer"; import { PTY_CONNECTING_TIMEOUT_MS, @@ -801,9 +801,9 @@ export default function ChatPage({ isActive = true }: { isActive?: boolean }) { mobileReplacementInputUntilRef.current = Date.now() + MOBILE_REPLACEMENT_WINDOW_MS; } }; - const markCompositionEnd = (ev: Event) => { + const markCompositionEnd = (ev: CompositionEvent) => { mobileReplacementInputUntilRef.current = Date.now() + MOBILE_REPLACEMENT_WINDOW_MS; - compositionForwarder.onCompositionEnd((ev as CompositionEvent).data); + compositionForwarder.onCompositionEnd(ev.data); }; textarea.addEventListener("beforeinput", markReplacementInput, true); @@ -1278,7 +1278,7 @@ export default function ChatPage({ isActive = true }: { isActive?: boolean }) { // behave normally. // eslint-disable-next-line no-control-regex -- intentional ESC byte in xterm SGR mouse report parser const SGR_MOUSE_RE = /^\x1b\[<(\d+);(\d+);(\d+)([Mm])$/; - const forwardPtyData = (data: string) => { + const forwardPtyData = (data: string, useMobileReplacement = true) => { // Mouse reports (scroll wheel etc.) are not typed input — swallow // them before the blocked-input check so scrolling a disconnected // terminal doesn't trip the "reconnecting" notice. @@ -1302,7 +1302,7 @@ export default function ChatPage({ isActive = true }: { isActive?: boolean }) { const normalized = normalizePtyMobileInput( data, ptyInputLineRef.current, - Date.now() <= mobileReplacementInputUntilRef.current, + useMobileReplacement && Date.now() <= mobileReplacementInputUntilRef.current, ); ptyInputLineRef.current = normalized.nextLine; if (normalized.normalized) { @@ -1310,11 +1310,13 @@ export default function ChatPage({ isActive = true }: { isActive?: boolean }) { } ws.send(normalized.data); }; - sendComposedText = forwardPtyData; + // The deferred composition fallback is already committed text, so it + // must not consume the mobile replacement window intended for xterm's + // normal onData path. + sendComposedText = (data) => forwardPtyData(data, false); onDataDisposable = term.onData((data) => { - if (compositionForwarder.shouldForwardTerminalData(data)) { - forwardPtyData(data); - } + compositionForwarder.noteTerminalData(data); + forwardPtyData(data); }); onResizeDisposable = term.onResize(({ cols, rows }) => { @@ -1344,6 +1346,7 @@ export default function ChatPage({ isActive = true }: { isActive?: boolean }) { onResizeDisposable?.dispose(); onScrollDisposable?.dispose(); mobileInputCleanup?.(); + compositionForwarder.dispose(); host.removeEventListener("paste", handleBrowserPaste, true); host.removeEventListener("dragover", handleBrowserDragOver, true); host.removeEventListener("drop", handleBrowserDrop, true); From c066b5a1d377d874630979265d2fb319e81280ad Mon Sep 17 00:00:00 2001 From: Denis H <79399355+dplush@users.noreply.github.com> Date: Sat, 1 Aug 2026 17:31:37 +0200 Subject: [PATCH 320/376] test(dashboard): cover IME fallback lifecycle --- web/src/lib/pty-composition.test.ts | 25 +++++++++++++++++++++++++ 1 file changed, 25 insertions(+) diff --git a/web/src/lib/pty-composition.test.ts b/web/src/lib/pty-composition.test.ts index 773c8be9d80dd..45fb275b906dd 100644 --- a/web/src/lib/pty-composition.test.ts +++ b/web/src/lib/pty-composition.test.ts @@ -28,6 +28,31 @@ describe("createPtyCompositionForwarder", () => { expect(send).not.toHaveBeenCalled(); }); + it("keeps the fallback pending when xterm emits unrelated input", () => { + vi.useFakeTimers(); + const send = vi.fn(); + const forwarder = createPtyCompositionForwarder(send); + + forwarder.onCompositionEnd("ä"); + forwarder.noteTerminalData("x"); + vi.runAllTimers(); + + expect(send).toHaveBeenCalledExactlyOnceWith("ä"); + }); + + it("supersedes an earlier composition and cancels it on disposal", () => { + vi.useFakeTimers(); + const send = vi.fn(); + const forwarder = createPtyCompositionForwarder(send); + + forwarder.onCompositionEnd("a"); + forwarder.onCompositionEnd("ä"); + forwarder.dispose(); + vi.runAllTimers(); + + expect(send).not.toHaveBeenCalled(); + }); + it("does not send an empty cancelled composition", () => { vi.useFakeTimers(); const send = vi.fn(); From 3d9fdb2e2b07d5c637cb85a8d32cfe85a59ad5b7 Mon Sep 17 00:00:00 2001 From: Denis H <79399355+dplush@users.noreply.github.com> Date: Sat, 1 Aug 2026 17:33:21 +0200 Subject: [PATCH 321/376] fix(dashboard): avoid duplicate IME fallback input --- web/src/lib/pty-composition.test.ts | 18 ++++++++++++++++-- web/src/lib/pty-composition.ts | 11 ++++++----- web/src/pages/ChatPage.tsx | 5 ++++- 3 files changed, 26 insertions(+), 8 deletions(-) diff --git a/web/src/lib/pty-composition.test.ts b/web/src/lib/pty-composition.test.ts index 45fb275b906dd..d60761857e0e3 100644 --- a/web/src/lib/pty-composition.test.ts +++ b/web/src/lib/pty-composition.test.ts @@ -28,7 +28,7 @@ describe("createPtyCompositionForwarder", () => { expect(send).not.toHaveBeenCalled(); }); - it("keeps the fallback pending when xterm emits unrelated input", () => { + it("prefers xterm input in the grace window over the fallback", () => { vi.useFakeTimers(); const send = vi.fn(); const forwarder = createPtyCompositionForwarder(send); @@ -37,7 +37,21 @@ describe("createPtyCompositionForwarder", () => { forwarder.noteTerminalData("x"); vi.runAllTimers(); - expect(send).toHaveBeenCalledExactlyOnceWith("ä"); + expect(send).not.toHaveBeenCalled(); + }); + + it("forwards a second composition after the first fallback completes", () => { + vi.useFakeTimers(); + const send = vi.fn(); + const forwarder = createPtyCompositionForwarder(send); + + forwarder.onCompositionEnd("ä"); + vi.runAllTimers(); + forwarder.onCompositionEnd("ö"); + vi.runAllTimers(); + + expect(send).toHaveBeenNthCalledWith(1, "ä"); + expect(send).toHaveBeenNthCalledWith(2, "ö"); }); it("supersedes an earlier composition and cancels it on disposal", () => { diff --git a/web/src/lib/pty-composition.ts b/web/src/lib/pty-composition.ts index 329cb4c173ef2..7944b8cb556e8 100644 --- a/web/src/lib/pty-composition.ts +++ b/web/src/lib/pty-composition.ts @@ -25,12 +25,13 @@ export function createPtyCompositionForwarder(send: (data: string) => void) { const committed = pending; clearPending(); if (committed) send(committed); - }, 0); + }, 16); }, - noteTerminalData(data: string) { - if (pending && data.startsWith(pending)) { - clearPending(); - } + noteTerminalData(_data: string) { + // Any xterm input in the short grace window is its own composition + // delivery (possibly chunked or normalization-different), so prefer it + // over the fallback to avoid duplicate text. + if (pending) clearPending(); }, dispose: clearPending, }; diff --git a/web/src/pages/ChatPage.tsx b/web/src/pages/ChatPage.tsx index b53f4086d620a..a67525234a87d 100644 --- a/web/src/pages/ChatPage.tsx +++ b/web/src/pages/ChatPage.tsx @@ -1313,7 +1313,10 @@ export default function ChatPage({ isActive = true }: { isActive?: boolean }) { // The deferred composition fallback is already committed text, so it // must not consume the mobile replacement window intended for xterm's // normal onData path. - sendComposedText = (data) => forwardPtyData(data, false); + sendComposedText = (data) => { + forwardPtyData(data, false); + mobileReplacementInputUntilRef.current = 0; + }; onDataDisposable = term.onData((data) => { compositionForwarder.noteTerminalData(data); forwardPtyData(data); From 874ae74a10fa03f3ca2d489adba0e5108d5135e0 Mon Sep 17 00:00:00 2001 From: Denis H <79399355+dplush@users.noreply.github.com> Date: Sat, 1 Aug 2026 17:35:40 +0200 Subject: [PATCH 322/376] fix(dashboard): keep mouse input out of IME fallback --- web/src/lib/pty-composition.ts | 2 +- web/src/pages/ChatPage.tsx | 9 ++++----- 2 files changed, 5 insertions(+), 6 deletions(-) diff --git a/web/src/lib/pty-composition.ts b/web/src/lib/pty-composition.ts index 7944b8cb556e8..6eec3d5a2213d 100644 --- a/web/src/lib/pty-composition.ts +++ b/web/src/lib/pty-composition.ts @@ -27,7 +27,7 @@ export function createPtyCompositionForwarder(send: (data: string) => void) { if (committed) send(committed); }, 16); }, - noteTerminalData(_data: string) { + noteTerminalData() { // Any xterm input in the short grace window is its own composition // delivery (possibly chunked or normalization-different), so prefer it // over the fallback to avoid duplicate text. diff --git a/web/src/pages/ChatPage.tsx b/web/src/pages/ChatPage.tsx index a67525234a87d..00538e5915c14 100644 --- a/web/src/pages/ChatPage.tsx +++ b/web/src/pages/ChatPage.tsx @@ -1313,12 +1313,11 @@ export default function ChatPage({ isActive = true }: { isActive?: boolean }) { // The deferred composition fallback is already committed text, so it // must not consume the mobile replacement window intended for xterm's // normal onData path. - sendComposedText = (data) => { - forwardPtyData(data, false); - mobileReplacementInputUntilRef.current = 0; - }; + sendComposedText = (data) => forwardPtyData(data, false); onDataDisposable = term.onData((data) => { - compositionForwarder.noteTerminalData(data); + if (!SGR_MOUSE_RE.test(data)) { + compositionForwarder.noteTerminalData(); + } forwardPtyData(data); }); From a428531ef653b439ad40d5e33949c1586fa519c7 Mon Sep 17 00:00:00 2001 From: Denis H <79399355+dplush@users.noreply.github.com> Date: Sat, 1 Aug 2026 17:37:46 +0200 Subject: [PATCH 323/376] fix(dashboard): preserve consecutive IME composition input --- web/src/lib/pty-composition.test.ts | 20 ++++++++++---------- web/src/lib/pty-composition.ts | 12 +++++++----- web/src/pages/ChatPage.tsx | 2 +- 3 files changed, 18 insertions(+), 16 deletions(-) diff --git a/web/src/lib/pty-composition.test.ts b/web/src/lib/pty-composition.test.ts index d60761857e0e3..0a0189320c68b 100644 --- a/web/src/lib/pty-composition.test.ts +++ b/web/src/lib/pty-composition.test.ts @@ -28,38 +28,38 @@ describe("createPtyCompositionForwarder", () => { expect(send).not.toHaveBeenCalled(); }); - it("prefers xterm input in the grace window over the fallback", () => { + it("forwards a second composition after the first fallback completes", () => { vi.useFakeTimers(); const send = vi.fn(); const forwarder = createPtyCompositionForwarder(send); forwarder.onCompositionEnd("ä"); - forwarder.noteTerminalData("x"); + vi.runAllTimers(); + forwarder.onCompositionEnd("ö"); vi.runAllTimers(); - expect(send).not.toHaveBeenCalled(); + expect(send).toHaveBeenNthCalledWith(1, "ä"); + expect(send).toHaveBeenNthCalledWith(2, "ö"); }); - it("forwards a second composition after the first fallback completes", () => { + it("preserves an earlier rapid composition before scheduling the next", () => { vi.useFakeTimers(); const send = vi.fn(); const forwarder = createPtyCompositionForwarder(send); + forwarder.onCompositionEnd("a"); forwarder.onCompositionEnd("ä"); vi.runAllTimers(); - forwarder.onCompositionEnd("ö"); - vi.runAllTimers(); - expect(send).toHaveBeenNthCalledWith(1, "ä"); - expect(send).toHaveBeenNthCalledWith(2, "ö"); + expect(send).toHaveBeenNthCalledWith(1, "a"); + expect(send).toHaveBeenNthCalledWith(2, "ä"); }); - it("supersedes an earlier composition and cancels it on disposal", () => { + it("cancels a pending composition on disposal", () => { vi.useFakeTimers(); const send = vi.fn(); const forwarder = createPtyCompositionForwarder(send); - forwarder.onCompositionEnd("a"); forwarder.onCompositionEnd("ä"); forwarder.dispose(); vi.runAllTimers(); diff --git a/web/src/lib/pty-composition.ts b/web/src/lib/pty-composition.ts index 6eec3d5a2213d..0fd7367c98f36 100644 --- a/web/src/lib/pty-composition.ts +++ b/web/src/lib/pty-composition.ts @@ -19,7 +19,10 @@ export function createPtyCompositionForwarder(send: (data: string) => void) { return { onCompositionEnd(data: string | null) { if (!data) return; + // Preserve rapid consecutive commits instead of discarding the first. + const previous = pending; clearPending(); + if (previous) send(previous); pending = data; timer = setTimeout(() => { const committed = pending; @@ -27,11 +30,10 @@ export function createPtyCompositionForwarder(send: (data: string) => void) { if (committed) send(committed); }, 16); }, - noteTerminalData() { - // Any xterm input in the short grace window is its own composition - // delivery (possibly chunked or normalization-different), so prefer it - // over the fallback to avoid duplicate text. - if (pending) clearPending(); + noteTerminalData(data: string) { + // A non-protocol xterm input in the short grace window is its own + // composition delivery (possibly chunked or normalization-different). + if (pending && !data.startsWith("\x1b")) clearPending(); }, dispose: clearPending, }; diff --git a/web/src/pages/ChatPage.tsx b/web/src/pages/ChatPage.tsx index 00538e5915c14..cfda855f39073 100644 --- a/web/src/pages/ChatPage.tsx +++ b/web/src/pages/ChatPage.tsx @@ -1316,7 +1316,7 @@ export default function ChatPage({ isActive = true }: { isActive?: boolean }) { sendComposedText = (data) => forwardPtyData(data, false); onDataDisposable = term.onData((data) => { if (!SGR_MOUSE_RE.test(data)) { - compositionForwarder.noteTerminalData(); + compositionForwarder.noteTerminalData(data); } forwardPtyData(data); }); From 434f1e954c98671b20dffb83b00fac069bd9fb71 Mon Sep 17 00:00:00 2001 From: Denis H <79399355+dplush@users.noreply.github.com> Date: Sat, 1 Aug 2026 18:01:26 +0200 Subject: [PATCH 324/376] fix(dashboard): retain composition after unrelated input --- web/src/lib/pty-composition.test.ts | 12 ++++++++++++ web/src/lib/pty-composition.ts | 5 ++--- 2 files changed, 14 insertions(+), 3 deletions(-) diff --git a/web/src/lib/pty-composition.test.ts b/web/src/lib/pty-composition.test.ts index 0a0189320c68b..5b533112c6aae 100644 --- a/web/src/lib/pty-composition.test.ts +++ b/web/src/lib/pty-composition.test.ts @@ -28,6 +28,18 @@ describe("createPtyCompositionForwarder", () => { expect(send).not.toHaveBeenCalled(); }); + it("forwards a pending composition after unrelated terminal data", () => { + vi.useFakeTimers(); + const send = vi.fn(); + const forwarder = createPtyCompositionForwarder(send); + + forwarder.onCompositionEnd("ä"); + forwarder.noteTerminalData("x"); + vi.runAllTimers(); + + expect(send).toHaveBeenCalledExactlyOnceWith("ä"); + }); + it("forwards a second composition after the first fallback completes", () => { vi.useFakeTimers(); const send = vi.fn(); diff --git a/web/src/lib/pty-composition.ts b/web/src/lib/pty-composition.ts index 0fd7367c98f36..f5074826ed5e3 100644 --- a/web/src/lib/pty-composition.ts +++ b/web/src/lib/pty-composition.ts @@ -31,9 +31,8 @@ export function createPtyCompositionForwarder(send: (data: string) => void) { }, 16); }, noteTerminalData(data: string) { - // A non-protocol xterm input in the short grace window is its own - // composition delivery (possibly chunked or normalization-different). - if (pending && !data.startsWith("\x1b")) clearPending(); + // xterm delivers the committed text before any following terminal input. + if (pending && !data.startsWith("\x1b") && data.startsWith(pending)) clearPending(); }, dispose: clearPending, }; From 0d40955fd1f818daa716a2459ed1ff0af83ee2ae Mon Sep 17 00:00:00 2001 From: Denis H <79399355+dplush@users.noreply.github.com> Date: Sat, 8 Aug 2026 03:56:23 +0200 Subject: [PATCH 325/376] fix(dashboard): handle chunked IME composition input --- web/src/lib/pty-composition.test.ts | 41 +++++++++++++++++++++++++++++ web/src/lib/pty-composition.ts | 19 +++++++++++-- 2 files changed, 58 insertions(+), 2 deletions(-) diff --git a/web/src/lib/pty-composition.test.ts b/web/src/lib/pty-composition.test.ts index 5b533112c6aae..82b3bd636d865 100644 --- a/web/src/lib/pty-composition.test.ts +++ b/web/src/lib/pty-composition.test.ts @@ -40,6 +40,47 @@ describe("createPtyCompositionForwarder", () => { expect(send).toHaveBeenCalledExactlyOnceWith("ä"); }); + it("forwards a pending composition when unrelated data precedes matching chunks", () => { + vi.useFakeTimers(); + const send = vi.fn(); + const forwarder = createPtyCompositionForwarder(send); + + forwarder.onCompositionEnd("ab"); + forwarder.noteTerminalData("x"); + forwarder.noteTerminalData("a"); + forwarder.noteTerminalData("b"); + vi.runAllTimers(); + + expect(send).toHaveBeenCalledExactlyOnceWith("ab"); + }); + + it("cancels a pending composition when matching text arrives in clean chunks", () => { + vi.useFakeTimers(); + const send = vi.fn(); + const forwarder = createPtyCompositionForwarder(send); + + forwarder.onCompositionEnd("ab"); + forwarder.noteTerminalData("a"); + forwarder.noteTerminalData("b"); + vi.runAllTimers(); + + expect(send).not.toHaveBeenCalled(); + }); + + it("ignores ESC/SGR data while matching composition chunks", () => { + vi.useFakeTimers(); + const send = vi.fn(); + const forwarder = createPtyCompositionForwarder(send); + + forwarder.onCompositionEnd("ab"); + forwarder.noteTerminalData("a"); + forwarder.noteTerminalData("\x1b[<0;10;10M"); + forwarder.noteTerminalData("b"); + vi.runAllTimers(); + + expect(send).not.toHaveBeenCalled(); + }); + it("forwards a second composition after the first fallback completes", () => { vi.useFakeTimers(); const send = vi.fn(); diff --git a/web/src/lib/pty-composition.ts b/web/src/lib/pty-composition.ts index f5074826ed5e3..82876c58804b1 100644 --- a/web/src/lib/pty-composition.ts +++ b/web/src/lib/pty-composition.ts @@ -7,9 +7,13 @@ export function createPtyCompositionForwarder(send: (data: string) => void) { let pending: string | null = null; let timer: ReturnType | null = null; + let matchedTerminalPrefix = ""; + let sawUnrelatedTerminalData = false; const clearPending = () => { pending = null; + matchedTerminalPrefix = ""; + sawUnrelatedTerminalData = false; if (timer) { clearTimeout(timer); timer = null; @@ -31,8 +35,19 @@ export function createPtyCompositionForwarder(send: (data: string) => void) { }, 16); }, noteTerminalData(data: string) { - // xterm delivers the committed text before any following terminal input. - if (pending && !data.startsWith("\x1b") && data.startsWith(pending)) clearPending(); + if (!pending || data.startsWith("\x1b") || sawUnrelatedTerminalData) return; + + // xterm may split committed text across callbacks, but only a clean, + // leading match is authoritative. Once unrelated data arrives, retain + // the fallback even if later callbacks happen to spell the composition. + const observed = matchedTerminalPrefix + data; + if (observed.startsWith(pending)) { + clearPending(); + } else if (pending.startsWith(observed)) { + matchedTerminalPrefix = observed; + } else { + sawUnrelatedTerminalData = true; + } }, dispose: clearPending, }; From 00516e6e8edab6aa2bb32d8cac68edfbbd4181e9 Mon Sep 17 00:00:00 2001 From: Denis H <79399355+dplush@users.noreply.github.com> Date: Wed, 12 Aug 2026 07:55:13 +0200 Subject: [PATCH 326/376] test(dashboard): cover delayed IME fallback after unrelated input --- web/src/lib/pty-composition.test.ts | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/web/src/lib/pty-composition.test.ts b/web/src/lib/pty-composition.test.ts index 82b3bd636d865..95429082a646c 100644 --- a/web/src/lib/pty-composition.test.ts +++ b/web/src/lib/pty-composition.test.ts @@ -35,7 +35,9 @@ describe("createPtyCompositionForwarder", () => { forwarder.onCompositionEnd("ä"); forwarder.noteTerminalData("x"); - vi.runAllTimers(); + vi.advanceTimersByTime(15); + expect(send).not.toHaveBeenCalled(); + vi.advanceTimersByTime(1); expect(send).toHaveBeenCalledExactlyOnceWith("ä"); }); From daa3f66d082f0dc05f32580df138ef19672a2945 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 23:11:13 -0700 Subject: [PATCH 327/376] chore: map contributor emails for IME TUI salvage --- contributors/emails/jinshi.zjs@antgroup.com | 1 + contributors/emails/nsovipgl@gmail.com | 1 + 2 files changed, 2 insertions(+) create mode 100644 contributors/emails/jinshi.zjs@antgroup.com create mode 100644 contributors/emails/nsovipgl@gmail.com diff --git a/contributors/emails/jinshi.zjs@antgroup.com b/contributors/emails/jinshi.zjs@antgroup.com new file mode 100644 index 0000000000000..0926adbb4b2cd --- /dev/null +++ b/contributors/emails/jinshi.zjs@antgroup.com @@ -0,0 +1 @@ +InphinitiZ diff --git a/contributors/emails/nsovipgl@gmail.com b/contributors/emails/nsovipgl@gmail.com new file mode 100644 index 0000000000000..b0ee3b389de3a --- /dev/null +++ b/contributors/emails/nsovipgl@gmail.com @@ -0,0 +1 @@ +hanhvs From 29ce4820478509c9e04f94ce3de5eae02be185a0 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 23:42:22 -0700 Subject: [PATCH 328/376] style: eslint --fix import ordering in salvaged IME tests --- .../hermes-ink/src/ink/parse-keypress-drop-probe.test.ts | 6 ++++++ ui-tui/src/__tests__/imeVietnameseTelex.test.tsx | 6 +++++- 2 files changed, 11 insertions(+), 1 deletion(-) diff --git a/ui-tui/packages/hermes-ink/src/ink/parse-keypress-drop-probe.test.ts b/ui-tui/packages/hermes-ink/src/ink/parse-keypress-drop-probe.test.ts index 6cc013476dbba..cfd1940369866 100644 --- a/ui-tui/packages/hermes-ink/src/ink/parse-keypress-drop-probe.test.ts +++ b/ui-tui/packages/hermes-ink/src/ink/parse-keypress-drop-probe.test.ts @@ -12,12 +12,16 @@ function keysToText(keys: Array<{ name?: string; sequence?: string }>): string { // Reconstruct what the composer would insert: backspaces delete, everything // else with a printable sequence inserts its sequence. let out = '' + for (const k of keys) { if (k.name === 'backspace') { out = out.slice(0, -1) + continue } + const seq = k.sequence ?? '' + // Mirror the composer's PRINTABLE gate if (/^[ -~\u00a0-\uffff]+$/.test(seq)) { out += seq @@ -27,6 +31,7 @@ function keysToText(keys: Array<{ name?: string; sequence?: string }>): string { out += `«DROP:${[...seq].map(c => 'U+' + c.codePointAt(0)!.toString(16)).join(',')}»` } } + return out } @@ -58,6 +63,7 @@ describe('parser does not silently drop printable codepoints', () => { it('exhaustive: DEL between every pair of letters never drops a letter', () => { const letters = [...'aăâeêioôơuưy'] + for (const a of letters) { for (const b of letters) { const input = `${a}\x7f${b}` diff --git a/ui-tui/src/__tests__/imeVietnameseTelex.test.tsx b/ui-tui/src/__tests__/imeVietnameseTelex.test.tsx index 59d2be7782501..f4dc2e2438b28 100644 --- a/ui-tui/src/__tests__/imeVietnameseTelex.test.tsx +++ b/ui-tui/src/__tests__/imeVietnameseTelex.test.tsx @@ -2,7 +2,7 @@ import { EventEmitter } from 'events' import { renderSync } from '@hermes/ink' import React, { useState } from 'react' -import { describe, expect, it, vi, beforeEach, afterEach } from 'vitest' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { TextInput } from '../components/textInput.js' @@ -41,11 +41,13 @@ class FakeTty extends EventEmitter { } setRawMode(mode: boolean): this { this.isRaw = mode + return this } write(chunk: string | Uint8Array, cb?: (err?: Error | null) => void): boolean { this.chunks.push(typeof chunk === 'string' ? chunk : Buffer.from(chunk).toString('utf8')) cb?.() + return true } } @@ -192,10 +194,12 @@ describe('Fast-echo suppression reset (60ms window)', () => { try { await tick() + for (const r of reads) { stdin1.send(r) await tick() } + // After "ha", fast-echo is enabled (inkRepaintedRef.current = false) expect(values1.at(-1)).toBe('ha') From ee9ec6164ced44838282017a3b3d3b8fe841b148 Mon Sep 17 00:00:00 2001 From: Adolanium <94890352+Adolanium@users.noreply.github.com> Date: Sun, 2 Aug 2026 21:55:04 +0300 Subject: [PATCH 329/376] perf(state): stop selecting full message content in session search Every search route in _search_messages_impl (FTS, CJK bigram, trigram, LIKE fallback, rebuild-gap supplement) selected m.content, then the result tail popped it unread. On DBs with multi-MB tool rows, each search read and materialized up to `limit` full rows only to discard them. Snippets come from snippet()/substr() in SQL and the context window is re-fetched by id, so no code path ever read the column. Drop the column from all six SELECT lists. Returned dicts are unchanged: content was never part of the public result (the pop ran before return), and tests/test_hermes_state.py already documents that contract. (cherry picked from commit d0c3af167e7dd4eb18e1bea29ba107a94911ea24) --- hermes_state_search.py | 17 +++++++++-------- tests/test_hermes_state.py | 3 +++ 2 files changed, 12 insertions(+), 8 deletions(-) diff --git a/hermes_state_search.py b/hermes_state_search.py index b1cf669210fe6..e8d29f413ee85 100644 --- a/hermes_state_search.py +++ b/hermes_state_search.py @@ -1390,7 +1390,6 @@ def _run_trigram_search( m.session_id, m.role, snippet({table}, -1, '>>>', '<<<', '...', 40) AS snippet, - m.content, m.timestamp, m.tool_name, s.source, @@ -1583,7 +1582,7 @@ def _search_messages_like_fallback( sql = f""" SELECT m.id, m.session_id, m.role, substr(m.content, max(1, instr(m.content, ?) - 40), 120) AS snippet, - m.content, m.timestamp, m.tool_name, + m.timestamp, m.tool_name, s.source, s.model, s.started_at AS session_started FROM messages m JOIN sessions s ON s.id = m.session_id @@ -1689,7 +1688,12 @@ def _finalize_search_matches( except Exception: match["context"] = [] - # Remove full content from result (snippet is enough, saves tokens) + # Full message content is never selected by any search route: every + # SELECT returns snippet + metadata only (saves I/O on multi-MB tool + # rows and the tokens a content column would cost downstream). The + # context query above re-fetches its 3-message window by id, so + # nothing reads content from the match rows themselves. The pop stays + # as a belt-and-braces guard for any future route that selects it. for match in matches: match.pop("content", None) @@ -1823,7 +1827,6 @@ def _search_messages_impl( m.session_id, m.role, snippet(messages_fts, -1, '>>>', '<<<', '...', 40) AS snippet, - m.content, m.timestamp, m.tool_name, s.source, @@ -1913,7 +1916,6 @@ def _search_messages_impl( m.session_id, m.role, snippet(messages_fts_cjk, -1, '>>>', '<<<', '...', 40) AS snippet, - m.content, m.timestamp, m.tool_name, s.source, @@ -2002,7 +2004,6 @@ def _search_messages_impl( m.session_id, m.role, snippet(messages_fts_trigram, -1, '>>>', '<<<', '...', 40) AS snippet, - m.content, m.timestamp, m.tool_name, s.source, @@ -2095,7 +2096,7 @@ def _search_messages_impl( substr(m.content, max(1, instr(m.content, ?) - 40), 120) AS snippet, - m.content, m.timestamp, m.tool_name, + m.timestamp, m.tool_name, s.source, s.model, s.started_at AS session_started FROM messages m JOIN sessions s ON s.id = m.session_id @@ -2272,7 +2273,7 @@ def _search_unindexed_gap( substr(m.content, max(1, instr(m.content, ?) - 40), 120) AS snippet, - m.content, m.timestamp, m.tool_name, + m.timestamp, m.tool_name, s.source, s.model, s.started_at AS session_started FROM messages m JOIN sessions s ON s.id = m.session_id diff --git a/tests/test_hermes_state.py b/tests/test_hermes_state.py index 1b2f688ae76da..2d720669784ea 100644 --- a/tests/test_hermes_state.py +++ b/tests/test_hermes_state.py @@ -473,6 +473,7 @@ def connect_without_trigram(*args, **kwargs): results = db.search_messages("大别山") assert len(results) == 1 # Note: search_messages strips 'content' from results; use 'snippet'. + assert "content" not in results[0] assert "大别山" in results[0]["snippet"] finally: db.close() @@ -739,6 +740,8 @@ def test_search_finds_content(self, db): # At least one result should mention docker snippets = [r.get("snippet", "") for r in results] assert any("docker" in s.lower() or "Docker" in s for s in snippets) + # Results never carry full content; snippet + metadata only. + assert all("content" not in r for r in results) From 6e1bdc0a186495635ddaab1c421cea033c9e4fcf Mon Sep 17 00:00:00 2001 From: blunkjamie-dev <235017879+blunkjamie-dev@users.noreply.github.com> Date: Sun, 9 Aug 2026 11:00:29 -0500 Subject: [PATCH 330/376] perf(session-search): adapt discovery result hydration (cherry picked from commit 60a3530444f65c2cdcd4e5b983e4aa380ed651c7) --- tests/tools/test_session_search.py | 89 ++++++++++++++++++++++-- tools/session_search_tool.py | 106 +++++++++++++++++++++-------- 2 files changed, 161 insertions(+), 34 deletions(-) diff --git a/tests/tools/test_session_search.py b/tests/tools/test_session_search.py index af36d90dc2491..c8d92d5e9e72f 100644 --- a/tests/tools/test_session_search.py +++ b/tests/tools/test_session_search.py @@ -1,12 +1,14 @@ """Tests for the single-shape session_search tool. -Three calling shapes: - 1. DISCOVERY — pass query → FTS5 + anchored window + bookends per hit +Four calling shapes: + 1. DISCOVERY — pass query → FTS5 + adaptive/full hydration 2. SCROLL — pass session_id + around_message_id → just the window - 3. BROWSE — no args → recent sessions chronologically + 3. READ — pass session_id → whole or head/tail-truncated session + 4. BROWSE — no args → recent sessions chronologically All run zero LLM calls. """ +import inspect import json import time @@ -72,6 +74,8 @@ def test_schema_params_cover_every_shape(self): assert "query" in params assert "limit" in params assert params["sort"]["enum"] == ["newest", "oldest"] + assert params["detail"]["enum"] == ["adaptive", "full"] + assert params["detail"]["default"] == "adaptive" # Scroll shape assert "session_id" in params assert "around_message_id" in params @@ -81,6 +85,10 @@ def test_schema_params_cover_every_shape(self): # Mode is inferred from which args are set — no explicit mode param assert "mode" not in params + def test_detail_parameter_is_appended_for_positional_compatibility(self): + parameters = list(inspect.signature(session_search).parameters) + assert parameters[-1] == "detail" + class TestFormatTimestamp: def test_formats_unix_and_passes_through_the_rest(self): @@ -176,17 +184,22 @@ def search_spy(*args, **kwargs): assert "context" not in requested_fields assert len(result["results"]) == 1 hit = result["results"][0] + assert hit["detail"] == "full" assert "bookend_start" in hit assert hit["messages"] assert "bookend_end" in hit - def test_discovery_result_has_bookends_and_window(self, db): + def test_full_detail_returns_bookends_and_window_for_every_hit(self, db): _seed_modpack_sessions(db) - result = json.loads(session_search(query="modpack", limit=3, db=db)) + result = json.loads(session_search( + query="modpack", limit=3, detail="full", db=db + )) assert result["success"] is True assert result["mode"] == "discover" + assert result["detail"] == "full" assert result["count"] >= 1 for hit in result["results"]: + assert hit["detail"] == "full" assert "bookend_start" in hit assert "messages" in hit assert "bookend_end" in hit @@ -195,6 +208,72 @@ def test_discovery_result_has_bookends_and_window(self, db): assert "messages_before" in hit assert "messages_after" in hit + def test_default_discovery_keeps_top_full_and_compacts_lower_hits(self, db): + _seed_modpack_sessions(db) + + result = json.loads(session_search(query="modpack", limit=3, db=db)) + + assert result["success"] is True + assert result["detail"] == "adaptive" + assert len(result["results"]) == 3 + + top, *lower = result["results"] + assert top["detail"] == "full" + assert "bookend_start" in top + assert len(top["messages"]) > 1 + assert "bookend_end" in top + + for hit in lower: + assert hit["detail"] == "compact" + assert hit["bookend_start"] == [] + assert len(hit["messages"]) == 1 + assert hit["messages"][0]["id"] == hit["match_message_id"] + assert hit["messages"][0]["anchor"] is True + assert hit["bookend_end"] == [] + + def test_adaptive_detail_preserves_ranking_and_reduces_payload(self, db): + now = int(time.time()) + for session_index in range(3): + session_id = f"payload_{session_index}" + db.create_session(session_id, source="cli") + db._conn.execute( + "UPDATE sessions SET started_at = ? WHERE id = ?", + (now - session_index, session_id), + ) + for message_index in range(8): + db.append_message( + session_id, + role="user" if message_index % 2 == 0 else "assistant", + content=f"opening {session_index}-{message_index} " + "o" * 2500, + ) + db.append_message( + session_id, + role="user", + content=f"payloadneedle anchor {session_index} " + "a" * 3500, + ) + for message_index in range(8): + db.append_message( + session_id, + role="assistant" if message_index % 2 == 0 else "user", + content=f"closing {session_index}-{message_index} " + "c" * 2500, + ) + db._conn.commit() + + adaptive_json = session_search(query="payloadneedle", limit=3, db=db) + full_json = session_search( + query="payloadneedle", limit=3, detail="full", db=db + ) + adaptive = json.loads(adaptive_json) + full = json.loads(full_json) + + assert [r["session_id"] for r in adaptive["results"]] == [ + r["session_id"] for r in full["results"] + ] + assert [r["match_message_id"] for r in adaptive["results"]] == [ + r["match_message_id"] for r in full["results"] + ] + assert len(adaptive_json.encode("utf-8")) < len(full_json.encode("utf-8")) * 0.6 + def test_current_session_filtered_out(self, db): _seed_modpack_sessions(db) diff --git a/tools/session_search_tool.py b/tools/session_search_tool.py index 759ed1c44f6d6..9bab7addfc576 100644 --- a/tools/session_search_tool.py +++ b/tools/session_search_tool.py @@ -2,23 +2,27 @@ """ Session Search Tool - Long-Term Conversation Recall -Single-shape tool with three calling modes (inferred from args, no explicit +Single-shape tool with four calling modes (inferred from args, no explicit mode parameter): - 1. DISCOVERY — pass ``query``. Runs FTS5, dedupes hits by session lineage, - returns top N sessions each with: snippet, ±5 message window around the - match, plus bookend_start (first 3 user+assistant msgs of session) and - bookend_end (last 3). Zero LLM cost. + 1. DISCOVERY — pass ``query``. Runs FTS5 and dedupes hits by session lineage. + Adaptive detail (the default) fully hydrates the top result with a ±5 + message window and bookends, while lower-ranked results keep the exact + anchor message plus metadata. Pass ``detail="full"`` to fully hydrate + every result. Zero LLM cost. 2. SCROLL — pass ``session_id`` + ``around_message_id``. Returns a window of ±window messages centered on the anchor, no FTS5, no bookends. To scroll forward / backward, re-anchor on the last / first message id of the returned window. - 3. BROWSE — no args. Returns recent sessions chronologically (titles, + 3. READ — pass ``session_id`` without an anchor. Returns the whole session, + or a bounded head/tail view for large sessions. + + 4. BROWSE — no args. Returns recent sessions chronologically (titles, previews, timestamps). -All three modes operate on the SQLite session DB via the FTS5 index and +All four modes operate on the SQLite session DB via the FTS5 index and the get_anchored_view / get_messages_around primitives in hermes_state. No LLM calls anywhere — every shape returns actual messages from the DB. @@ -740,6 +744,7 @@ def _title_match_result( "bookend_end": [_shape_message(m) for m in (view.get("bookend_end") or messages[-3:])], "messages_before": view.get("messages_before", 0), "messages_after": view.get("messages_after", max(len(messages) - 5, 0)), + "detail": "full", "_lineage_root": lineage_root, } if lineage_root and lineage_root != session_id: @@ -753,10 +758,11 @@ def _discover( role_filter: Optional[List[str]], limit: int, sort: Optional[str], + detail: str, current_session_id: str = None, link_profile: str = None, ) -> str: - """Discovery shape: FTS5 + anchored window + bookends per hit. Single call.""" + """Discovery shape: FTS5 plus adaptive or full result hydration.""" role_list = role_filter if role_filter else ["user", "assistant"] current_lineage_root = _resolve_lineage(db, current_session_id) if current_session_id else None title_result = _title_match_result(db, query, current_lineage_root) @@ -788,6 +794,7 @@ def _discover( "success": True, "mode": "discover", "query": query, + "detail": detail, "results": [], "count": 0, "message": "No matching sessions found.", @@ -864,6 +871,11 @@ def _discover( except Exception: session_meta = {} + result_detail = "full" if detail == "full" or not results else "compact" + window_messages = view.get("window") or [] + if result_detail == "compact": + window_messages = [m for m in window_messages if m.get("id") == msg_id] + entry = { "session_id": hit_sid, "when": _format_timestamp( @@ -875,19 +887,31 @@ def _discover( "matched_role": match_info.get("role"), "match_message_id": msg_id, "snippet": match_info.get("snippet") or "", - "bookend_start": [ - _shape_message(m, max_content_len=1200) - for m in (view.get("bookend_start") or []) - if not _is_compaction_summary(m.get("content", "")) - ], - "messages": [_shape_message(m, anchor_id=msg_id, max_content_len=4000) for m in (view.get("window") or [])], - "bookend_end": [ - _shape_message(m, max_content_len=1200) - for m in (view.get("bookend_end") or []) - if not _is_compaction_summary(m.get("content", "")) + "bookend_start": ( + [ + _shape_message(m, max_content_len=1200) + for m in (view.get("bookend_start") or []) + if not _is_compaction_summary(m.get("content", "")) + ] + if result_detail == "full" + else [] + ), + "messages": [ + _shape_message(m, anchor_id=msg_id, max_content_len=4000) + for m in window_messages ], + "bookend_end": ( + [ + _shape_message(m, max_content_len=1200) + for m in (view.get("bookend_end") or []) + if not _is_compaction_summary(m.get("content", "")) + ] + if result_detail == "full" + else [] + ), "messages_before": view.get("messages_before", 0), "messages_after": view.get("messages_after", 0), + "detail": result_detail, } if lineage_root and lineage_root != hit_sid: entry["parent_session_id"] = lineage_root @@ -900,6 +924,7 @@ def _discover( "success": True, "mode": "discover", "query": query, + "detail": detail, "results": results, "count": len(results), "sessions_searched": len(seen_sessions), @@ -922,12 +947,14 @@ def _session_search_impl( sort: str = None, # Cross-profile (any shape) profile: str = None, + # Discovery result shaping (appended to preserve positional compatibility) + detail: str = "adaptive", *, _owned_dbs: Optional[List[Any]] = None, ) -> str: """Single-shape tool. Mode inferred from which args are set. - Discovery: pass ``query``. + Discovery: pass ``query``; ``detail="full"`` hydrates every result. Scroll: pass ``session_id`` + ``around_message_id``. Read: pass ``session_id`` (no anchor) — dumps the whole session. Browse: pass nothing. @@ -1017,12 +1044,19 @@ def _session_search_impl( if candidate in ("newest", "oldest"): sort_norm = candidate + detail_norm = ( + "full" + if isinstance(detail, str) and detail.strip().lower() == "full" + else "adaptive" + ) + return _discover( db=db, query=query.strip(), role_filter=role_list, limit=limit, sort=sort_norm, + detail=detail_norm, current_session_id=current_session_id, link_profile=profile, ) @@ -1108,19 +1142,21 @@ def check_session_search_requirements() -> bool: "FOUR CALLING SHAPES\n\n" " 1) DISCOVERY — pass `query`:\n" " session_search(query=\"auth refactor\", limit=3)\n" - " Runs FTS5, dedupes hits by session lineage, returns the top N sessions. " - "Each result carries:\n" + " Runs FTS5, dedupes hits by session lineage, and returns the top N " + "sessions. Adaptive detail is the default: the top-ranked result carries " + "full context, while lower-ranked results stay compact. Pass `detail=\"full\"` " + "to fully hydrate every result. Every result carries:\n" " - session_id, title, when, source\n" " - snippet: FTS5-highlighted match excerpt\n" - " - bookend_start: first 3 user+assistant messages of the session " - "(the goal / kickoff)\n" - " - messages: ±5 messages around the FTS5 match, with the anchor message " - "flagged (the hit in context)\n" - " - bookend_end: last 3 user+assistant messages of the session " - "(the resolution / decisions)\n" + " - detail: `full` or `compact`\n" + " - bookend_start/bookend_end: the first/last 3 user+assistant messages " + "for full results; empty lists for compact results\n" + " - messages: ±5 messages around the FTS5 match for full results; only " + "the flagged anchor message for compact results\n" " - match_message_id, messages_before, messages_after\n" - " Bookends + window together let you reconstruct goal → match → resolution " - "without paying for the whole transcript.\n\n" + " The top result's bookends + window let you reconstruct goal → match → " + "resolution immediately. Scroll a compact result when another session looks " + "more promising.\n\n" " 2) SCROLL — pass `session_id` + `around_message_id`:\n" " session_search(session_id=\"...\", around_message_id=12345, window=10)\n" " Returns a window of ±`window` messages centered on the anchor. No FTS5, " @@ -1197,6 +1233,17 @@ def check_session_search_requirements() -> bool: "and browse shapes." ), }, + "detail": { + "type": "string", + "enum": ["adaptive", "full"], + "description": ( + "Discovery shape only. 'adaptive' (default) fully hydrates the " + "top-ranked result and returns only the exact anchor message for " + "lower-ranked results. 'full' returns bookends and the complete " + "anchored window for every result." + ), + "default": "adaptive", + }, "session_id": { "type": "string", "description": ( @@ -1261,6 +1308,7 @@ def check_session_search_requirements() -> bool: around_message_id=args.get("around_message_id"), window=args.get("window", 5), sort=args.get("sort"), + detail=args.get("detail", "adaptive"), profile=args.get("profile"), db=kw.get("db"), current_session_id=kw.get("current_session_id"), From 163d7af310be38a6b2d8549d54f889bcdcf9ed44 Mon Sep 17 00:00:00 2001 From: blunkjamie-dev <235017879+blunkjamie-dev@users.noreply.github.com> Date: Sun, 9 Aug 2026 11:10:15 -0500 Subject: [PATCH 331/376] fix(session-search): forward detail through agent paths (cherry picked from commit 5f6de984f170ef470c7fbbd7662484bfaffc821d) --- agent/agent_runtime_helpers.py | 1 + agent/tool_executor.py | 1 + .../test_token_persistence_non_cli.py | 42 ++++++++++++++++++- website/docs/user-guide/sessions.md | 27 ++++++++---- 4 files changed, 62 insertions(+), 9 deletions(-) diff --git a/agent/agent_runtime_helpers.py b/agent/agent_runtime_helpers.py index 3f3404b40b02b..e0c51fb35388e 100644 --- a/agent/agent_runtime_helpers.py +++ b/agent/agent_runtime_helpers.py @@ -3076,6 +3076,7 @@ def _execute(next_args: dict) -> Any: around_message_id=next_args.get("around_message_id"), window=next_args.get("window", 5), sort=next_args.get("sort"), + detail=next_args.get("detail", "adaptive"), db=session_db, current_session_id=agent.session_id, ), diff --git a/agent/tool_executor.py b/agent/tool_executor.py index d0ce5621aab79..bdf4efc23585d 100644 --- a/agent/tool_executor.py +++ b/agent/tool_executor.py @@ -1975,6 +1975,7 @@ def _execute(next_args: dict) -> Any: around_message_id=next_args.get("around_message_id"), window=next_args.get("window", 5), sort=next_args.get("sort"), + detail=next_args.get("detail", "adaptive"), db=session_db, current_session_id=agent.session_id, ) diff --git a/tests/run_agent/test_token_persistence_non_cli.py b/tests/run_agent/test_token_persistence_non_cli.py index dd82395d237bf..7479c2af85cae 100644 --- a/tests/run_agent/test_token_persistence_non_cli.py +++ b/tests/run_agent/test_token_persistence_non_cli.py @@ -80,9 +80,49 @@ def fake_session_search(**kwargs): monkeypatch.setitem(sys.modules, "tools.session_search_tool", session_search_mod) agent = _make_agent(None, platform="acp") - result = json.loads(agent._invoke_tool("session_search", {"query": "Hermes"}, "task-id")) + result = json.loads(agent._invoke_tool( + "session_search", + {"query": "Hermes", "detail": "full"}, + "task-id", + )) assert result["success"] is True assert captured["db"] is sentinel_db assert captured["query"] == "Hermes" + assert captured["detail"] == "full" assert agent._session_db is sentinel_db + + +def test_sequential_session_search_forwards_detail(monkeypatch): + session_db = MagicMock() + captured = {} + + session_search_mod = ModuleType("tools.session_search_tool") + + def fake_session_search(**kwargs): + captured.update(kwargs) + return json.dumps({"success": True, "results": []}) + + session_search_mod.session_search = fake_session_search + monkeypatch.setitem(sys.modules, "tools.session_search_tool", session_search_mod) + + agent = _make_agent(session_db, platform="acp") + tool_call = SimpleNamespace( + id="search-1", + function=SimpleNamespace( + name="session_search", + arguments=json.dumps({"query": "Hermes", "detail": "full"}), + ), + ) + assistant_message = SimpleNamespace(tool_calls=[tool_call]) + messages = [] + + agent._execute_tool_calls_sequential( + assistant_message, + messages, + "task-id", + ) + + assert captured["db"] is session_db + assert captured["query"] == "Hermes" + assert captured["detail"] == "full" diff --git a/website/docs/user-guide/sessions.md b/website/docs/user-guide/sessions.md index 3f02dd2be20b9..d08d8a4b85018 100644 --- a/website/docs/user-guide/sessions.md +++ b/website/docs/user-guide/sessions.md @@ -612,9 +612,9 @@ routing is the only thing the repair changes. Back up first ## Session Search Tool -The agent has a built-in `session_search` tool that performs full-text search across all past conversations using SQLite's FTS5 engine — and lets the agent scroll through any session it finds. No LLM calls, no summarization, no truncation. Every shape returns actual messages from the DB. +The agent has a built-in `session_search` tool that performs full-text search across all past conversations using SQLite's FTS5 engine — and lets the agent scroll through any session it finds. It makes no LLM calls and returns bounded views of actual messages from the DB rather than generated summaries. -### Three calling shapes +### Four calling shapes The tool infers what you want from which arguments you set. There's no `mode` parameter. @@ -624,16 +624,18 @@ The tool infers what you want from which arguments you set. There's no `mode` pa session_search(query="auth refactor", limit=3) ``` -Runs FTS5, dedupes hits by session lineage, returns the top N sessions. Each result carries: +Runs FTS5, dedupes hits by session lineage, and returns the top N sessions. Discovery uses adaptive detail by default: the highest-ranked result includes its full context window and bookends, while lower-ranked results stay compact. Pass `detail="full"` to fully hydrate every result. + +Each result carries: - `session_id`, `title`, `when`, `source` - `snippet` — FTS5-highlighted match excerpt -- `bookend_start` — first 3 user+assistant messages of the session (the goal/kickoff) -- `messages` — ±5 messages around the FTS5 match, with the anchor message flagged (the hit in context) -- `bookend_end` — last 3 user+assistant messages of the session (the resolution/decisions) +- `detail` — `full` or `compact` +- `bookend_start` / `bookend_end` — first/last 3 user+assistant messages for full results; empty lists for compact results +- `messages` — ±5 messages around the FTS5 match for full results; only the flagged anchor message for compact results - `match_message_id`, `messages_before`, `messages_after` -Bookends + window together reconstruct goal → match → resolution without paying for the whole transcript. Typical wall time: 15–50ms on a real session DB. +The top result reconstructs goal → match → resolution immediately. If another compact result looks more promising, use its session and message IDs with the scroll shape. Typical wall time is tens of milliseconds on a real session DB. **2. Scroll — pass `session_id` + `around_message_id`:** @@ -650,7 +652,15 @@ Returns a window of ±`window` messages centered on the anchor. No FTS5, no book Typical wall time: 1–2ms per scroll call. -**3. Browse — no args:** +**3. Read — pass `session_id` without an anchor:** + +```python +session_search(session_id="20260510_174648_805cc2") +``` + +Returns the whole session, or a bounded head/tail view for large sessions. This shape is also used to resolve an `@session:/` link. + +**4. Browse — no args:** ```python session_search() @@ -670,6 +680,7 @@ The keyword mode supports standard FTS5 query syntax: ### Optional parameters - `sort` — `newest` or `oldest`, on top of FTS5 ranking. Omit for relevance-only ordering (the default; suitable for exploratory recall). Use `newest` for "where did we leave X" questions, `oldest` for "how did X start" questions. +- `detail` — `adaptive` (default) fully hydrates only the top discovery result; `full` hydrates every discovery result. - `role_filter` — comma-separated roles to include. Discovery defaults to `user,assistant` (tool output is usually noise). Pass `user,assistant,tool` to include tool output (debugging tool behaviour) or `tool` to search tool output only. ### When It's Used From 888807d55c0ab15ada640490ce94531186fe6cc4 Mon Sep 17 00:00:00 2001 From: blunkjamie-dev <235017879+blunkjamie-dev@users.noreply.github.com> Date: Sun, 9 Aug 2026 11:18:51 -0500 Subject: [PATCH 332/376] docs(sessions): clarify actual-message retrieval (cherry picked from commit c9b1286be40651a3d5d2a0877a06be9bf34d7294) --- website/docs/user-guide/sessions.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/website/docs/user-guide/sessions.md b/website/docs/user-guide/sessions.md index d08d8a4b85018..7d36dfa256e15 100644 --- a/website/docs/user-guide/sessions.md +++ b/website/docs/user-guide/sessions.md @@ -612,7 +612,7 @@ routing is the only thing the repair changes. Back up first ## Session Search Tool -The agent has a built-in `session_search` tool that performs full-text search across all past conversations using SQLite's FTS5 engine — and lets the agent scroll through any session it finds. It makes no LLM calls and returns bounded views of actual messages from the DB rather than generated summaries. +The agent has a built-in `session_search` tool that performs full-text search across all past conversations using SQLite's FTS5 engine — and lets the agent scroll through any session it finds. It makes no LLM calls and returns views of actual messages from the DB rather than generating summaries. ### Four calling shapes From 4415f917b47471929b5675ef32b8c387d7001140 Mon Sep 17 00:00:00 2001 From: blunkjamie-dev <235017879+blunkjamie-dev@users.noreply.github.com> Date: Sun, 9 Aug 2026 11:29:52 -0500 Subject: [PATCH 333/376] test(session-search): lock positional parameter prefix (cherry picked from commit 73592200c69a4f0b6d7c290ce45832847df608e2) --- tests/tools/test_session_search.py | 14 +++++++++++++- 1 file changed, 13 insertions(+), 1 deletion(-) diff --git a/tests/tools/test_session_search.py b/tests/tools/test_session_search.py index c8d92d5e9e72f..fb61db973f72e 100644 --- a/tests/tools/test_session_search.py +++ b/tests/tools/test_session_search.py @@ -87,7 +87,19 @@ def test_schema_params_cover_every_shape(self): def test_detail_parameter_is_appended_for_positional_compatibility(self): parameters = list(inspect.signature(session_search).parameters) - assert parameters[-1] == "detail" + historical_prefix = [ + "query", + "role_filter", + "limit", + "db", + "current_session_id", + "session_id", + "around_message_id", + "window", + "sort", + "profile", + ] + assert parameters == [*historical_prefix, "detail"] class TestFormatTimestamp: From 2162d583b118b06d12046ff83ada9c34a4771c34 Mon Sep 17 00:00:00 2001 From: joaomarcos Date: Sun, 2 Aug 2026 20:10:24 -0300 Subject: [PATCH 334/376] perf(run-agent): reuse the Anthropic request-local client instead of rebuilding it per call _create_request_anthropic_client() built a fresh anthropic.Anthropic client (and httpx pool) on every single LLM call, and _close_request_anthropic_client() always fully closed it right after - unlike the OpenAI-wire path, which caches and reuses one warm client across sequential calls via a single-slot cache keyed on the effective client kwargs. Add the same single-slot cache to the Anthropic-wire path: keyed on credentials, base URL/Bedrock region, per-model timeout, and the 1M-beta flag; in_use guards concurrent calls from sharing one pool's close/abort lifecycle; poisoned marks a cross-thread-aborted slot so the owner-thread close discards it; reuse only on request_complete / stream_request_complete (the same _REQUEST_CLIENT_REUSE_REASONS the OpenAI path already uses). Wires a teardown hook into release_clients()/close() mirroring _close_cached_request_openai_client. Fixes #HPA-02 (cherry picked from commit 37f90df15593e6ded0390f827f8e5604bf0acc86) --- run_agent.py | 142 ++++++++++++++-- .../test_anthropic_request_client_reuse.py | 157 ++++++++++++++++++ 2 files changed, 287 insertions(+), 12 deletions(-) create mode 100644 tests/agent/test_anthropic_request_client_reuse.py diff --git a/run_agent.py b/run_agent.py index 86aee2e53ae49..b6f655c355aaa 100644 --- a/run_agent.py +++ b/run_agent.py @@ -4385,6 +4385,10 @@ def release_clients(self) -> None: self._close_cached_request_openai_client(reason="cache_evict") except Exception: pass + try: + self._close_cached_request_anthropic_client(reason="cache_evict") + except Exception: + pass def close(self) -> None: """Release all resources held by this agent instance. @@ -4472,6 +4476,10 @@ def close(self) -> None: self._close_cached_request_openai_client(reason="agent_close") except Exception: pass + try: + self._close_cached_request_anthropic_client(reason="agent_close") + except Exception: + pass # 6c. Close the Codex app-server session. The runtime already drops # it on turn crash / retirement (agent/codex_runtime.py), but hard @@ -5386,8 +5394,31 @@ def _abort_request_openai_client(self, client: Any, *, reason: str) -> None: exc, ) + def _request_anthropic_client_cache_ref(self) -> dict: + # Lazy init — tests build agents via AIAgent.__new__ without __init__. + cache = getattr(self, "_request_anthropic_client_cache", None) + if cache is None: + cache = {"client": None, "key": None, "poisoned": False, "in_use": False} + self._request_anthropic_client_cache = cache + return cache + + def _request_anthropic_client_key(self) -> tuple: + """Cache key covering everything that forces a fresh client: credential + rotation, base URL / region changes, timeout changes (model switch), + and the 1M-context beta flag.""" + if getattr(self, "provider", None) == "bedrock": + region = getattr(self, "_bedrock_region", "us-east-1") or "us-east-1" + return ("bedrock", region) + return ( + "direct", + self._anthropic_api_key, + getattr(self, "_anthropic_base_url", None), + get_provider_request_timeout(self.provider, self.model), + bool(getattr(self, "_oauth_1m_beta_disabled", False)), + ) + def _create_request_anthropic_client(self, *, reason: str) -> Any: - """Build a request-local Anthropic client for one in-flight call. + """Build (or reuse) a request-local Anthropic client for one in-flight call. The shared ``_anthropic_client`` stays the long-lived primary, but the stale/interrupt watchdog runs on the poll thread and must never call @@ -5399,24 +5430,56 @@ def _create_request_anthropic_client(self, *, reason: str) -> Any: worker performs the SDK-level close from its own context — the same ownership contract the OpenAI-wire path already uses. + Also mirrors the OpenAI-wire path's single-slot cache + (``_create_request_openai_client``): building ``anthropic.Anthropic`` + means a fresh httpx pool and TCP+TLS handshake per call, so the client + is kept warm across sequential calls whose cache key (credentials, + base URL/region, timeout, 1M-beta flag) hasn't changed. ``in_use`` + keeps a second concurrent call from sharing one pool's close/abort + lifecycle — it gets a fresh untracked client instead. + Mirrors ``_rebuild_anthropic_client`` construction (direct + Bedrock, - 1M-beta drop) but returns a fresh client instead of swapping the shared - one. + 1M-beta drop) but returns a fresh/cached client instead of swapping + the shared one. """ if self.api_mode == "anthropic_messages": self._try_refresh_anthropic_client_credentials() - _drop_1m = bool(getattr(self, "_oauth_1m_beta_disabled", False)) - if getattr(self, "provider", None) == "bedrock": + key = self._request_anthropic_client_key() + + stale = None + with self._openai_client_lock(): + cache = self._request_anthropic_client_cache_ref() + cached = cache["client"] + if cached is not None and not cache["in_use"]: + if ( + not cache["poisoned"] + and cache["key"] == key + and not self._is_openai_client_closed(cached) + ): + cache["in_use"] = True + return cached + # Key changed (credential rotation, base URL/region, timeout, + # 1M-beta flip), poisoned by a cross-thread abort, or + # externally closed — never reuse; discard and rebuild below. + stale = cached + cache["client"] = None + cache["key"] = None + cache["poisoned"] = False + if stale is not None: + # Safe to close from this thread: in_use was False, so no worker + # thread owns the pool's FDs (same #29507 reasoning as OpenAI). + self._close_request_anthropic_client(stale, reason=f"reuse_evict:{reason}") + + if key[0] == "bedrock": from agent.anthropic_adapter import build_anthropic_bedrock_client - region = getattr(self, "_bedrock_region", "us-east-1") or "us-east-1" - client = build_anthropic_bedrock_client(region) + client = build_anthropic_bedrock_client(key[1]) else: from agent.anthropic_adapter import build_anthropic_client client = build_anthropic_client( self._anthropic_api_key, getattr(self, "_anthropic_base_url", None), timeout=get_provider_request_timeout(self.provider, self.model), - drop_context_1m_beta=_drop_1m, + drop_context_1m_beta=key[4], ) logger.debug( "Anthropic request client created (%s, shared=False) provider=%s model=%s", @@ -5424,17 +5487,41 @@ def _create_request_anthropic_client(self, *, reason: str) -> Any: getattr(self, "provider", None), getattr(self, "model", None), ) + with self._openai_client_lock(): + cache = self._request_anthropic_client_cache_ref() + if cache["client"] is None: + cache["client"] = client + cache["key"] = key + cache["poisoned"] = False + cache["in_use"] = True + # else: a concurrent call holds the slot — hand this client out + # untracked; _close_request_anthropic_client fully closes + # untracked clients, preserving the per-request lifecycle. return client def _close_request_anthropic_client(self, client: Any, *, reason: str) -> None: - """Owner-thread full close of a request-local Anthropic client. - - Force-closes the pool's TCP sockets first (CLOSE-WAIT hygiene, parity - with ``_close_openai_client``), then does the graceful SDK close. Safe + """Owner-thread close of a request-local Anthropic client. + + On a clean finish (``reason`` in ``_REQUEST_CLIENT_REUSE_REASONS``) + the pool is kept warm in the cache slot for the next sequential call, + mirroring ``_close_request_openai_client``. Any other outcome + (error / kill / abort / stale-slot eviction) force-closes the pool's + TCP sockets first (CLOSE-WAIT hygiene, parity with + ``_close_openai_client``), then does the graceful SDK close. Safe because the caller owns the connection. """ if client is None: return + with self._openai_client_lock(): + cache = self._request_anthropic_client_cache_ref() + if cache["client"] is client: + if reason in self._REQUEST_CLIENT_REUSE_REASONS and not cache["poisoned"]: + cache["in_use"] = False + return + cache["client"] = None + cache["key"] = None + cache["poisoned"] = False + cache["in_use"] = False try: self._force_close_tcp_sockets(client) client.close() @@ -5453,6 +5540,30 @@ def _close_request_anthropic_client(self, client: Any, *, reason: str) -> None: exc, ) + def _close_cached_request_anthropic_client(self, *, reason: str) -> None: + """Teardown hook: really close the cached per-request Anthropic client.""" + with self._openai_client_lock(): + cache = getattr(self, "_request_anthropic_client_cache", None) + client = cache["client"] if cache else None + in_use = bool(cache["in_use"]) if cache else False + if cache is not None: + cache["client"] = None + cache["key"] = None + cache["poisoned"] = False + cache["in_use"] = False + if client is None: + return + if in_use: + # A worker thread has this client checked out for an in-flight + # request — same #29507 reasoning as the OpenAI teardown hook. + self._abort_request_anthropic_client(client, reason=f"{reason}_in_flight") + return + try: + self._force_close_tcp_sockets(client) + client.close() + except Exception: + pass + def _abort_request_anthropic_client(self, client: Any, *, reason: str) -> None: """Cross-thread abort for request-local Anthropic clients. @@ -5464,6 +5575,13 @@ def _abort_request_anthropic_client(self, client: Any, *, reason: str) -> None: """ if client is None: return + # A pool whose sockets were shut down from a stranger thread must + # never be reused: poison the cache slot so the owner-thread close + # discards it and the next create builds a fresh client. + with self._openai_client_lock(): + cache = self._request_anthropic_client_cache_ref() + if cache["client"] is client: + cache["poisoned"] = True try: shutdown_count = self._force_close_tcp_sockets(client) # Same visibility contract as the OpenAI abort path (#72975): diff --git a/tests/agent/test_anthropic_request_client_reuse.py b/tests/agent/test_anthropic_request_client_reuse.py new file mode 100644 index 0000000000000..d5983d62ec30e --- /dev/null +++ b/tests/agent/test_anthropic_request_client_reuse.py @@ -0,0 +1,157 @@ +"""Per-request Anthropic wire client reuse across sequential LLM calls. + +Mirrors ``tests/agent/test_request_client_reuse.py`` (the OpenAI-wire cache) +for the Anthropic request-local client. Before this cache existed, +``_create_request_anthropic_client`` built a fresh ``anthropic.Anthropic`` +(and its httpx pool) on every single LLM call and ``_close_request_anthropic_client`` +always fully closed it — no reuse across a turn's sequential tool-loop calls, +unlike the OpenAI-wire path. + +- identical cache key (credentials, base URL, timeout, 1M-beta flag) → same + client object handed back (the reuse win); +- key changes (credential rotation, base URL change) → evict + rebuild; +- cross-thread abort poisons the slot → the owner-thread close does a real + close and the next create rebuilds; +- non-reuse close reasons (error cleanups, stale/interrupt kills) discard — + only request_complete / stream_request_complete reuse; +- teardown (release_clients / close) really closes the cached client, or + detaches it to the in-flight worker's own close when checked out. +""" + +from unittest.mock import MagicMock, patch + +from run_agent import AIAgent + + +class _StubClient: + """Minimal non-Mock client: _is_openai_client_closed reads ``is_closed``.""" + + def __init__(self): + self.is_closed = False + + def close(self): + self.is_closed = True + + +def _make_agent(provider="anthropic", base_url="https://api.anthropic.com", model="claude-sonnet-5"): + agent = AIAgent.__new__(AIAgent) + agent.provider = provider + agent.model = model + agent.api_mode = "anthropic_messages" + agent._anthropic_api_key = "sk-ant-test" + agent._anthropic_base_url = base_url + agent._oauth_1m_beta_disabled = False + # Real credential-refresh reaches auth/network state we don't need here; + # the cache logic under test is agnostic to it. + agent._try_refresh_anthropic_client_credentials = MagicMock(return_value=False) + return agent + + +class _Harness: + """Patch the Anthropic client build/socket seams and record calls.""" + + def __init__(self, agent): + self.agent = agent + self.built = [] # reason + self._patchers = [] + + def __enter__(self): + def _fake_build(*a, **k): + self.built.append(k.get("drop_context_1m_beta")) + return _StubClient() + + self._patchers = [ + patch("agent.anthropic_adapter.build_anthropic_client", side_effect=_fake_build), + patch.object(self.agent, "_force_close_tcp_sockets", return_value=0), + ] + for p in self._patchers: + p.start() + return self + + def __exit__(self, *exc): + for p in self._patchers: + p.stop() + + +def test_reuse_on_identical_key_same_object(): + agent = _make_agent() + with _Harness(agent) as h: + a = agent._create_request_anthropic_client(reason="chat_completion_request") + agent._close_request_anthropic_client(a, reason="request_complete") + assert not a.is_closed # kept for reuse, not really closed + + b = agent._create_request_anthropic_client(reason="chat_completion_request") + assert b is a + assert len(h.built) == 1 + + +def test_rebuild_on_credential_rotation(): + agent = _make_agent() + with _Harness(agent): + a = agent._create_request_anthropic_client(reason="r") + agent._close_request_anthropic_client(a, reason="request_complete") + + agent._anthropic_api_key = "sk-ant-rotated" + b = agent._create_request_anthropic_client(reason="r") + assert b is not a + assert a.is_closed # stale slot really closed on eviction + + agent._close_request_anthropic_client(b, reason="request_complete") + c = agent._create_request_anthropic_client(reason="r") + assert c is b + + +def test_non_reuse_reason_discards_client(): + agent = _make_agent() + with _Harness(agent): + a = agent._create_request_anthropic_client(reason="r") + agent._close_request_anthropic_client(a, reason="request_error_cleanup") + assert a.is_closed + + b = agent._create_request_anthropic_client(reason="r") + assert b is not a + + +def test_cross_thread_abort_poisons_slot(): + agent = _make_agent() + with _Harness(agent): + a = agent._create_request_anthropic_client(reason="r") + agent._abort_request_anthropic_client(a, reason="interrupt") + # Owner thread's close now sees the poisoned slot and really closes. + agent._close_request_anthropic_client(a, reason="request_complete") + assert a.is_closed + + b = agent._create_request_anthropic_client(reason="r") + assert b is not a + + +def test_concurrent_call_gets_untracked_client(): + agent = _make_agent() + with _Harness(agent): + a = agent._create_request_anthropic_client(reason="r") + # Slot still checked out (in_use=True) — a second concurrent call + # must not share it. + b = agent._create_request_anthropic_client(reason="r") + assert b is not a + + # Finishing the untracked one does a real close, not a slot release. + agent._close_request_anthropic_client(b, reason="request_complete") + assert b.is_closed + # The tracked slot is unaffected and still reusable. + agent._close_request_anthropic_client(a, reason="request_complete") + c = agent._create_request_anthropic_client(reason="r") + assert c is a + + +def test_agent_close_closes_cached_request_client(): + agent = _make_agent() + with _Harness(agent): + a = agent._create_request_anthropic_client(reason="r") + agent._close_request_anthropic_client(a, reason="request_complete") + assert not a.is_closed + + agent._close_cached_request_anthropic_client(reason="agent_close") + assert a.is_closed + + # Idempotent: a second teardown must not error or double-act. + agent._close_cached_request_anthropic_client(reason="agent_close") From 7e439dbb1bc7ad6b56dea0492ba4437d7c9d2eb6 Mon Sep 17 00:00:00 2001 From: lepetitprince716-prog Date: Thu, 6 Aug 2026 11:04:34 -0400 Subject: [PATCH 335/376] perf: parallelize provider model-list fetches in model picker When the 1h provider_models_cache.json TTL lapses, the model picker serially fetches /v1/models for each authenticated provider. With 10+ providers this stacks to 15-30s of blocking before the picker renders. Add a parallel prefetch step before the serial picker build loops: - _collect_authed_provider_slugs(): lightweight credential pre-scan that mirrors sections 1/2/2b without fetching model lists - _prefetch_provider_models_parallel(): ThreadPoolExecutor-based concurrent fetch of stale/missing cache entries (max 8 workers) - update_provider_cache_entry(): thread-safe single-entry cache writer with threading.Lock to prevent concurrent write races Guardrails: - Skipped when <=3 authed providers (overhead not worth it) - Skipped when refresh=True (serial path force-refreshes) - Exception-isolated (falls back to serial path on any failure) - No behavioral change (same model lists, same picker output) Closes #80413 (cherry picked from commit 89dddd6cb5d53d73278e0518c375fb5b878e5c6b) --- hermes_cli/model_switch.py | 282 +++++++++++++++++- hermes_cli/models.py | 28 ++ .../test_model_cache_parallel_prefetch.py | 250 ++++++++++++++++ 3 files changed, 559 insertions(+), 1 deletion(-) create mode 100644 tests/hermes_cli/test_model_cache_parallel_prefetch.py diff --git a/hermes_cli/model_switch.py b/hermes_cli/model_switch.py index 3299843d2ab69..9b88f8465fb34 100644 --- a/hermes_cli/model_switch.py +++ b/hermes_cli/model_switch.py @@ -21,7 +21,9 @@ from __future__ import annotations import logging +import os import re +import time from dataclasses import dataclass from typing import Any, List, NamedTuple, Optional @@ -2054,6 +2056,262 @@ def _scoped_key_env(name: str) -> str: return "" +# --- Parallel prefetch for provider model catalogs ----------------------- +# +# When the 1h disk cache lapses (or on first cold open), list_authenticated_providers() +# calls cached_provider_model_ids() serially for each authed provider. Each call +# that misses the cache blocks on a live /v1/models HTTP round-trip (1-8s per +# provider depending on endpoint latency). With 10+ authed providers the +# cumulative serial blocking time is 15-30+ seconds. +# +# This prefetch function runs those same cached_provider_model_ids() calls in +# parallel via ThreadPoolExecutor before the main picker build loop starts. +# The main loop then hits warm cache entries instead of blocking on live +# fetches. Providers whose cache was already fresh (SWR or within TTL) are +# skipped entirely — no wasted network calls. +# +# Net effect on a 13-provider setup with an expired cache: +# Before: ~20s serial blocking (sum of all provider latencies) +# After: ~8s parallel (max single provider latency), rest served from cache + +_PARALLEL_PREFETCH_WORKERS = 8 + + +def _prefetch_provider_models_parallel(provider_slugs: list[str]) -> None: + """Fetch model catalogs for multiple providers in parallel. + + Only providers whose cache entry is stale or missing are fetched; fresh + entries are skipped to avoid unnecessary network calls. Each worker uses + :func:`update_provider_cache_entry` (thread-safe) to persist its result, + so concurrent writes to ``provider_models_cache.json`` don't clobber each + other. + + :param provider_slugs: Hermes provider IDs to prefetch (e.g. ``["openrouter", + "anthropic", "deepseek"]``). Unknown providers are silently skipped. + """ + from hermes_cli.models import cached_provider_model_ids + + # Quick-stale-check: skip providers whose cache is already fresh so we + # don't waste network calls on a warm cache. We check staleness the same + # way cached_provider_model_ids does internally: load the cache, compare + # age to TTL. This is a read-only check — if the cache file changes + # between this check and the actual fetch, cached_provider_model_ids will + # still do the right thing (it re-reads the cache internally). + from hermes_cli.models import ( + _load_provider_models_cache, + _credential_fingerprint, + _PROVIDER_MODELS_CACHE_TTL, + normalize_provider, + ) + + now = time.time() + stale_slugs: list[str] = [] + cache = _load_provider_models_cache() + for slug in provider_slugs: + normalized = normalize_provider(slug) or (slug or "") + if not normalized: + continue + entry = cache.get(normalized) + fp = _credential_fingerprint(normalized) + if ( + isinstance(entry, dict) + and entry.get("fp") == fp + and isinstance(entry.get("models"), list) + and entry["models"] + ): + age = now - float(entry.get("at", 0)) + if age < _PROVIDER_MODELS_CACHE_TTL: + continue # fresh, skip + stale_slugs.append(normalized) + + if not stale_slugs: + return + + import concurrent.futures + + def _fetch_one(slug: str) -> None: + try: + models = cached_provider_model_ids(slug, force_refresh=True) + # cached_provider_model_ids already persists the result, but in a + # non-locked read-modify-write. Re-persist via the thread-safe + # path to guarantee no lost writes under concurrency. + if models: + from hermes_cli.models import update_provider_cache_entry + update_provider_cache_entry(slug, models) + except Exception: + pass # best-effort; picker falls back to curated list + + with concurrent.futures.ThreadPoolExecutor( + max_workers=min(_PARALLEL_PREFETCH_WORKERS, len(stale_slugs)), + thread_name_prefix="model-cache-prefetch", + ) as executor: + list(executor.map(_fetch_one, stale_slugs)) + + +def _collect_authed_provider_slugs( + models_dev_data: dict, + curated: dict[str, list[str]], + excluded: list[str], +) -> list[str]: + """Quick-scan which providers have credentials, without fetching model lists. + + Mirrors the credential-check logic from sections 1, 2, and 2b of + :func:`list_authenticated_providers` but **only** collects the provider + slugs — it never calls ``cached_provider_model_ids``. The returned list + is consumed by :func:`_prefetch_provider_models_parallel` to warm the disk + cache in parallel before the serial picker build loop starts. + + :param models_dev_data: The models.dev registry dict (from ``fetch_models_dev()``). + :param curated: The curated model-lists dict (``_PROVIDER_MODELS`` + extras). + :param excluded: Provider slugs to exclude (from ``model_catalog.excluded_providers``). + :returns: List of normalized provider slugs that have credentials. + """ + import os + from agent.models_dev import PROVIDER_TO_MODELS_DEV + from hermes_cli.auth import PROVIDER_REGISTRY, _load_auth_store + from hermes_cli.providers import HERMES_OVERLAYS, ALIASES as _PROVIDER_ALIAS_TABLE + from hermes_cli.models import _AGGREGATOR_PROVIDERS as _AGG_PROVIDERS, CANONICAL_PROVIDERS + + _excluded_set = {str(p).strip().lower() for p in excluded if p} + slugs: list[str] = [] + seen: set[str] = set() + + # --- Section 1: Hermes-mapped providers (PROVIDER_TO_MODELS_DEV) --- + for hermes_id, mdev_id in PROVIDER_TO_MODELS_DEV.items(): + _alias_target = _PROVIDER_ALIAS_TABLE.get(hermes_id) + if ( + _alias_target + and _alias_target != hermes_id + and _alias_target in _AGG_PROVIDERS + ): + continue + _canonical = hermes_id + try: + from providers import get_provider_profile as _gpp + _prof = _gpp(hermes_id) + if _prof is not None: + _canonical = _prof.name + except Exception: + pass + if _canonical != hermes_id: + continue + if hermes_id.lower() in seen: + continue + if hermes_id.lower() in _excluded_set or mdev_id.lower() in _excluded_set: + continue + pdata = models_dev_data.get(mdev_id) + if not isinstance(pdata, dict): + continue + pconfig = PROVIDER_REGISTRY.get(hermes_id) + if pconfig and pconfig.auth_type != "api_key": + continue + from hermes_cli.auth import is_runtime_provider_routable + if not is_runtime_provider_routable(hermes_id): + continue + if pconfig and pconfig.api_key_env_vars: + env_vars = list(pconfig.api_key_env_vars) + else: + env_vars = pdata.get("env", []) + if not isinstance(env_vars, list): + continue + has_creds = any(_scoped_key_env(ev) for ev in env_vars) + if not has_creds: + try: + store = _load_auth_store() + raw_pool_present = bool( + store and store.get("credential_pool", {}).get(hermes_id) + ) + if raw_pool_present: + has_creds = _credential_pool_is_usable( + hermes_id, raw_pool_present=True + ) + except Exception: + pass + if has_creds: + slugs.append(hermes_id) + seen.add(hermes_id.lower()) + + # --- Section 2: Hermes-only providers (HERMES_OVERLAYS) --- + _mdev_to_hermes = {v: k for k, v in PROVIDER_TO_MODELS_DEV.items()} + for pid, overlay in HERMES_OVERLAYS.items(): + if pid.lower() in seen: + continue + hermes_slug = _mdev_to_hermes.get(pid, pid) + if hermes_slug.lower() in seen: + continue + if pid.lower() in _excluded_set or hermes_slug.lower() in _excluded_set: + continue + has_creds = False + if overlay.auth_type == "aws_sdk": + # Skip AWS SDK providers in prefetch — credential detection is heavier + continue + elif overlay.auth_type == "vertex": + try: + from agent.vertex_adapter import has_vertex_credentials + has_creds = has_vertex_credentials() + except Exception: + pass + elif overlay.extra_env_vars: + has_creds = any(_scoped_key_env(ev) for ev in overlay.extra_env_vars) + if not has_creds and overlay.auth_type == "api_key": + for _key in (pid, hermes_slug): + pcfg = PROVIDER_REGISTRY.get(_key) + if pcfg and pcfg.api_key_env_vars: + if any(_scoped_key_env(ev) for ev in pcfg.api_key_env_vars): + has_creds = True + break + if not has_creds: + try: + store = _load_auth_store() + providers_store = store.get("providers", {}) if store else {} + if pid in providers_store or hermes_slug in providers_store: + has_creds = True + except Exception: + pass + if not has_creds: + try: + if _credential_pool_is_usable(hermes_slug): + has_creds = True + except Exception: + pass + if has_creds: + slugs.append(hermes_slug) + seen.add(pid.lower()) + seen.add(hermes_slug.lower()) + + # --- Section 2b: Canonical providers cross-check --- + for _cp in CANONICAL_PROVIDERS: + if _cp.slug.lower() in seen: + continue + if _cp.slug.lower() in _excluded_set: + continue + _cp_config = PROVIDER_REGISTRY.get(_cp.slug) + _cp_has_creds = False + if _cp_config and _cp_config.api_key_env_vars: + _cp_has_creds = any(_scoped_key_env(ev) for ev in _cp_config.api_key_env_vars) + if not _cp_has_creds: + try: + _cp_store = _load_auth_store() + _cp_providers_store = _cp_store.get("providers", {}) if _cp_store else {} + if _cp.slug in _cp_providers_store: + _cp_has_creds = True + except Exception: + pass + if not _cp_has_creds: + try: + if _credential_pool_is_usable(_cp.slug): + _cp_has_creds = True + except Exception: + pass + if not _cp_has_creds and _cp_config and getattr(_cp_config, "auth_type", "") == "aws_sdk": + continue # skip AWS SDK in prefetch + if _cp_has_creds: + slugs.append(_cp.slug) + seen.add(_cp.slug.lower()) + + return slugs + + def list_authenticated_providers( current_provider: str = "", current_base_url: str = "", @@ -2129,7 +2387,6 @@ def list_authenticated_providers( except Exception: pass - results: List[dict] = [] seen_slugs: set = set() # lowercase-normalized to catch case variants (#9545) _current_provider_norm = str(current_provider or "").strip().lower() @@ -2258,6 +2515,29 @@ def _has_aws_sdk_creds_for_listing(slug: str) -> bool: live = [current_model] curated["lmstudio"] = live + # --- Parallel cache prefetch --------------------------------------------- + # The serial loops below (sections 1, 2, 2b) each call + # cached_provider_model_ids(slug) which blocks on a live /v1/models HTTP + # round-trip when the disk cache is stale or missing. With many authed + # providers those serial round-trips stack to 15-30s on a cold/expired + # cache. Pre-scanning which providers have credentials (without fetching + # their model lists) and warming their cache entries in parallel makes + # the subsequent serial calls hit fresh cache entries instead. + # + # Skipped entirely when refresh=True (the serial path already force-refreshes) + # and when there are 3 or fewer authed providers (serial is fast enough; + # avoids thread-pool overhead for the common 1-2 provider case). + _prefetch_slugs: list[str] = [] + if not refresh: + _prefetch_slugs = _collect_authed_provider_slugs( + data, curated, excluded_providers or [] + ) + if len(_prefetch_slugs) > 3: + try: + _prefetch_provider_models_parallel(_prefetch_slugs) + except Exception: + pass # best-effort; serial path still works as fallback + # --- 1. Check Hermes-mapped providers --- from hermes_cli.models import _AGGREGATOR_PROVIDERS as _AGG_PROVIDERS from hermes_cli.providers import ALIASES as _PROVIDER_ALIAS_TABLE diff --git a/hermes_cli/models.py b/hermes_cli/models.py index d10e02b412789..2050fb53934cc 100644 --- a/hermes_cli/models.py +++ b/hermes_cli/models.py @@ -3563,6 +3563,9 @@ def _load_provider_models_cache() -> dict: return {} +_cache_write_lock = threading.Lock() + + def _save_provider_models_cache(data: dict) -> None: """Persist the cache dict. Best-effort — silent on any error.""" try: @@ -3574,6 +3577,31 @@ def _save_provider_models_cache(data: dict) -> None: pass +def update_provider_cache_entry(provider: str, models: list[str]) -> None: + """Thread-safe single-entry update of the provider-models disk cache. + + Used by parallel prefetch workers so concurrent fetches don't clobber + each other's writes via read-modify-write races on the shared JSON file. + Each worker loads the latest cache state under the lock, writes its own + entry, and saves — best-effort, silent on any error. + """ + try: + normalized = normalize_provider(provider) or (provider or "") + if not normalized or not models: + return + fp = _credential_fingerprint(normalized) + with _cache_write_lock: + cache = _load_provider_models_cache() + cache[normalized] = { + "fp": fp, + "at": time.time(), + "models": list(models), + } + _save_provider_models_cache(cache) + except Exception: + pass + + def cached_provider_model_ids( provider: Optional[str], *, diff --git a/tests/hermes_cli/test_model_cache_parallel_prefetch.py b/tests/hermes_cli/test_model_cache_parallel_prefetch.py new file mode 100644 index 0000000000000..17a2ec202eff6 --- /dev/null +++ b/tests/hermes_cli/test_model_cache_parallel_prefetch.py @@ -0,0 +1,250 @@ +"""Tests for parallel model-catalog prefetch and thread-safe cache writes. + +Regression tests for the serial /v1/models bottleneck: when the 1h disk cache +lapses, ``list_authenticated_providers()`` previously fetched each authed +provider's model list serially. With 10+ providers this stacked to 15-30s of +blocking HTTP round-trips. The parallel prefetch warms stale cache entries +concurrently via ThreadPoolExecutor before the serial picker loop starts. +""" + +from __future__ import annotations + +import time +from unittest.mock import patch, MagicMock + +import pytest + + +# --------------------------------------------------------------------------- +# Thread-safe cache entry update (hermes_cli/models.py) +# --------------------------------------------------------------------------- + +class TestUpdateProviderCacheEntry: + """Verify ``update_provider_cache_entry`` writes safely under concurrency.""" + + def test_writes_new_entry(self, tmp_path, monkeypatch): + """A new entry is persisted to the cache file.""" + import hermes_cli.models as mod + + cache_path = tmp_path / "provider_models_cache.json" + monkeypatch.setattr(mod, "_provider_models_cache_path", lambda: cache_path) + + with patch.object(mod, "_credential_fingerprint", return_value="fp1"): + mod.update_provider_cache_entry("openrouter", ["m1", "m2"]) + + cache = mod._load_provider_models_cache() + assert "openrouter" in cache + assert cache["openrouter"]["models"] == ["m1", "m2"] + assert cache["openrouter"]["fp"] == "fp1" + + def test_does_not_clobber_other_entries(self, tmp_path, monkeypatch): + """Concurrent writes to different providers don't lose entries.""" + import hermes_cli.models as mod + + cache_path = tmp_path / "provider_models_cache.json" + monkeypatch.setattr(mod, "_provider_models_cache_path", lambda: cache_path) + + # Seed with one entry + with patch.object(mod, "_credential_fingerprint", return_value="fp_a"): + mod.update_provider_cache_entry("provider_a", ["a1"]) + + # Write a second entry + with patch.object(mod, "_credential_fingerprint", return_value="fp_b"): + mod.update_provider_cache_entry("provider_b", ["b1"]) + + cache = mod._load_provider_models_cache() + assert "provider_a" in cache + assert cache["provider_a"]["models"] == ["a1"] + assert "provider_b" in cache + assert cache["provider_b"]["models"] == ["b1"] + + def test_skips_empty_models(self, tmp_path, monkeypatch): + """Empty model lists are not written to cache.""" + import hermes_cli.models as mod + + cache_path = tmp_path / "provider_models_cache.json" + monkeypatch.setattr(mod, "_provider_models_cache_path", lambda: cache_path) + + mod.update_provider_cache_entry("empty_provider", []) + cache = mod._load_provider_models_cache() + assert "empty_provider" not in cache + + def test_concurrent_writes_no_lost_entries(self, tmp_path, monkeypatch): + """Multiple threads writing different providers concurrently — all land.""" + import hermes_cli.models as mod + import concurrent.futures + + cache_path = tmp_path / "provider_models_cache.json" + monkeypatch.setattr(mod, "_provider_models_cache_path", lambda: cache_path) + + providers = [f"prov_{i}" for i in range(10)] + + with patch.object(mod, "_credential_fingerprint", side_effect=lambda p: f"fp_{p}"): + with concurrent.futures.ThreadPoolExecutor(max_workers=5) as executor: + list(executor.map( + lambda p: mod.update_provider_cache_entry(p, [f"model_{p}"]), + providers, + )) + + cache = mod._load_provider_models_cache() + for p in providers: + assert p in cache, f"{p} was lost in concurrent write" + assert cache[p]["models"] == [f"model_{p}"] + + +# --------------------------------------------------------------------------- +# Parallel prefetch (hermes_cli/model_switch.py) +# --------------------------------------------------------------------------- + +class TestPrefetchProviderModelsParallel: + """Verify ``_prefetch_provider_models_parallel`` fetches concurrently.""" + + def test_skips_all_fresh_entries(self, monkeypatch): + """When all cache entries are fresh, no fetch is made.""" + from hermes_cli.model_switch import _prefetch_provider_models_parallel + + fresh_cache = { + "openrouter": {"fp": "fp", "at": time.time(), "models": ["m1"]}, + "anthropic": {"fp": "fp", "at": time.time(), "models": ["m2"]}, + } + + with patch("hermes_cli.models._load_provider_models_cache", return_value=fresh_cache), \ + patch("hermes_cli.models._credential_fingerprint", return_value="fp"), \ + patch("hermes_cli.models.cached_provider_model_ids") as fetch: + _prefetch_provider_models_parallel(["openrouter", "anthropic"]) + + fetch.assert_not_called() + + def test_fetches_only_stale_entries(self, monkeypatch): + """Only providers with stale/missing cache entries are fetched.""" + from hermes_cli.model_switch import _prefetch_provider_models_parallel + + cache = { + "fresh_prov": {"fp": "fp_f", "at": time.time(), "models": ["m1"]}, + } + + fetch_calls = [] + + def mock_fetch(slug, force_refresh=False): + fetch_calls.append(slug) + return [f"model_{slug}"] + + with patch("hermes_cli.models._load_provider_models_cache", return_value=cache), \ + patch("hermes_cli.models._credential_fingerprint", return_value="fp_f"), \ + patch("hermes_cli.models.cached_provider_model_ids", side_effect=mock_fetch), \ + patch("hermes_cli.models.update_provider_cache_entry"): + _prefetch_provider_models_parallel(["fresh_prov", "stale_prov"]) + + assert "fresh_prov" not in fetch_calls + assert "stale_prov" in fetch_calls + + def test_fetches_in_parallel(self, monkeypatch): + """Multiple providers are fetched concurrently, not serially.""" + from hermes_cli.model_switch import _prefetch_provider_models_parallel + + # Track overlap: if serial, no two fetches should overlap in time. + active = [] + max_concurrent = [0] + lock = __import__("threading").Lock() + + def mock_fetch(slug, force_refresh=False): + with lock: + active.append(slug) + max_concurrent[0] = max(max_concurrent[0], len(active)) + time.sleep(0.05) # simulate network latency + with lock: + active.remove(slug) + return [f"model_{slug}"] + + slugs = [f"prov_{i}" for i in range(6)] + + with patch("hermes_cli.models._load_provider_models_cache", return_value={}), \ + patch("hermes_cli.models._credential_fingerprint", return_value="fp"), \ + patch("hermes_cli.models.cached_provider_model_ids", side_effect=mock_fetch), \ + patch("hermes_cli.models.update_provider_cache_entry"): + _prefetch_provider_models_parallel(slugs) + + assert max_concurrent[0] > 1, "fetches were serial, not parallel" + + def test_swallows_exceptions(self): + """A failing provider fetch doesn't raise — best-effort.""" + from hermes_cli.model_switch import _prefetch_provider_models_parallel + + def mock_fetch(slug, force_refresh=False): + raise ConnectionError("simulated network failure") + + with patch("hermes_cli.models._load_provider_models_cache", return_value={}), \ + patch("hermes_cli.models._credential_fingerprint", return_value="fp"), \ + patch("hermes_cli.models.cached_provider_model_ids", side_effect=mock_fetch), \ + patch("hermes_cli.models.update_provider_cache_entry"): + # Should not raise + _prefetch_provider_models_parallel(["failing_prov"]) + + def test_empty_list_is_noop(self): + """Empty provider list does nothing.""" + from hermes_cli.model_switch import _prefetch_provider_models_parallel + + with patch("hermes_cli.models.cached_provider_model_ids") as fetch: + _prefetch_provider_models_parallel([]) + fetch.assert_not_called() + + +# --------------------------------------------------------------------------- +# Integration: prefetch is called from list_authenticated_providers +# --------------------------------------------------------------------------- + +class TestPrefetchIntegration: + """Verify ``list_authenticated_providers`` triggers parallel prefetch.""" + + def test_prefetch_called_with_more_than_3_providers(self): + """When >3 providers are authed, parallel prefetch is invoked.""" + from hermes_cli import model_switch + + slugs = [f"prov_{i}" for i in range(5)] + captured_slugs = [] + + def mock_collect(data, curated, excluded): + return slugs + + with patch.object(model_switch, "_collect_authed_provider_slugs", side_effect=mock_collect), \ + patch.object(model_switch, "_prefetch_provider_models_parallel") as prefetch: + try: + model_switch.list_authenticated_providers() + except Exception: + pass # we only care about the prefetch call + captured_slugs = prefetch.call_args[0][0] if prefetch.called else [] + + assert prefetch.called + assert captured_slugs == slugs + + def test_prefetch_skipped_with_3_or_fewer_providers(self): + """When ≤3 providers are authed, parallel prefetch is skipped.""" + from hermes_cli import model_switch + + slugs = ["prov_a", "prov_b"] + + def mock_collect(data, curated, excluded): + return slugs + + with patch.object(model_switch, "_collect_authed_provider_slugs", side_effect=mock_collect), \ + patch.object(model_switch, "_prefetch_provider_models_parallel") as prefetch: + try: + model_switch.list_authenticated_providers() + except Exception: + pass + + prefetch.assert_not_called() + + def test_prefetch_skipped_on_refresh(self): + """When refresh=True, prefetch is skipped (serial path force-refreshes).""" + from hermes_cli import model_switch + + with patch.object(model_switch, "_collect_authed_provider_slugs") as collect, \ + patch.object(model_switch, "_prefetch_provider_models_parallel") as prefetch: + try: + model_switch.list_authenticated_providers(refresh=True) + except Exception: + pass + + collect.assert_not_called() + prefetch.assert_not_called() From 3806f8fc9a9bc5c9726ec05d14544a792d7965ef Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 23:12:21 -0700 Subject: [PATCH 336/376] fix(session-search): forward detail through the public session_search wrapper Main extracted a session_search() wrapper (owned-DB lifecycle) around _session_search_impl after #82595 was opened; the cherry-picked detail parameter landed on the impl only. Append it to the wrapper with the same positional-compatibility contract and pass it through. --- tools/session_search_tool.py | 3 +++ 1 file changed, 3 insertions(+) diff --git a/tools/session_search_tool.py b/tools/session_search_tool.py index 9bab7addfc576..c5752f5ca4adf 100644 --- a/tools/session_search_tool.py +++ b/tools/session_search_tool.py @@ -1076,6 +1076,8 @@ def session_search( sort: str = None, # Cross-profile (any shape) profile: str = None, + # Discovery result shaping (appended to preserve positional compatibility) + detail: str = "adaptive", ) -> str: """Run session search and close databases opened by this invocation.""" owned_dbs: List[Any] = [] @@ -1103,6 +1105,7 @@ def session_search( window=window, sort=sort, profile=profile, + detail=detail, _owned_dbs=owned_dbs, ) finally: From 7f84de277547aba2f059d64d86aff23bbe5e3313 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 14 Aug 2026 23:12:59 -0700 Subject: [PATCH 337/376] chore(contributors): map lepetitprince716@gmail.com -> lepetitprince716-prog --- contributors/emails/lepetitprince716@gmail.com | 1 + 1 file changed, 1 insertion(+) create mode 100644 contributors/emails/lepetitprince716@gmail.com diff --git a/contributors/emails/lepetitprince716@gmail.com b/contributors/emails/lepetitprince716@gmail.com new file mode 100644 index 0000000000000..568b68869a769 --- /dev/null +++ b/contributors/emails/lepetitprince716@gmail.com @@ -0,0 +1 @@ +lepetitprince716-prog From b587681afcd80cc859aeef71cf1c5c39f43244af Mon Sep 17 00:00:00 2001 From: LeonSGP43 Date: Mon, 8 Jun 2026 19:23:20 +0800 Subject: [PATCH 338/376] fix: keep historical thinking collapsed in tui (cherry picked from commit 354999388a4e465b6b7efa46cdc61d8d5b22b35f) --- ui-tui/src/__tests__/messages.test.ts | 75 ++++++++++++++++++++ ui-tui/src/__tests__/virtualHeights.test.ts | 22 ++++++ ui-tui/src/app/useMainApp.ts | 6 ++ ui-tui/src/components/messageLine.tsx | 4 ++ ui-tui/src/components/streamingAssistant.tsx | 1 + ui-tui/src/components/thinking.tsx | 11 ++- ui-tui/src/lib/virtualHeights.ts | 4 +- 7 files changed, 119 insertions(+), 4 deletions(-) diff --git a/ui-tui/src/__tests__/messages.test.ts b/ui-tui/src/__tests__/messages.test.ts index e572bd5b8c9c5..e83fe2d311320 100644 --- a/ui-tui/src/__tests__/messages.test.ts +++ b/ui-tui/src/__tests__/messages.test.ts @@ -128,6 +128,81 @@ describe('MessageLine', () => { expect(renderedLine).toContain('Ψ > Okay') }) + + it('keeps historical thinking blocks collapsed by default', () => { + const stdout = new PassThrough() + const stdin = new PassThrough() + const stderr = new PassThrough() + let output = '' + + Object.assign(stdout, { columns: 80, isTTY: false, rows: 24 }) + Object.assign(stdin, { isTTY: false }) + Object.assign(stderr, { isTTY: false }) + stdout.on('data', chunk => { + output += chunk.toString() + }) + + const instance = renderSync( + React.createElement(MessageLine, { + cols: 80, + msg: { kind: 'trail', role: 'system', text: '', thinking: 'step one\nstep two' }, + t: DEFAULT_THEME + }), + { + patchConsole: false, + stderr: stderr as NodeJS.WriteStream, + stdin: stdin as NodeJS.ReadStream, + stdout: stdout as NodeJS.WriteStream + } + ) + + instance.unmount() + instance.cleanup() + + const rendered = stripAnsi(output) + + expect(rendered).toContain('Thinking') + expect(rendered).not.toContain('step one') + expect(rendered).not.toContain('step two') + }) + + it('keeps live thinking blocks expanded while streaming', () => { + const stdout = new PassThrough() + const stdin = new PassThrough() + const stderr = new PassThrough() + let output = '' + + Object.assign(stdout, { columns: 80, isTTY: false, rows: 24 }) + Object.assign(stdin, { isTTY: false }) + Object.assign(stderr, { isTTY: false }) + stdout.on('data', chunk => { + output += chunk.toString() + }) + + const instance = renderSync( + React.createElement(MessageLine, { + cols: 80, + liveDetails: true, + msg: { kind: 'trail', role: 'system', text: '', thinking: 'step one\nstep two' }, + t: DEFAULT_THEME + }), + { + patchConsole: false, + stderr: stderr as NodeJS.WriteStream, + stdin: stdin as NodeJS.ReadStream, + stdout: stdout as NodeJS.WriteStream + } + ) + + instance.unmount() + instance.cleanup() + + const rendered = stripAnsi(output) + + expect(rendered).toContain('Thinking') + expect(rendered).toContain('step one') + expect(rendered).toContain('step two') + }) }) describe('upsert', () => { diff --git a/ui-tui/src/__tests__/virtualHeights.test.ts b/ui-tui/src/__tests__/virtualHeights.test.ts index 17cd32fec8d36..6440154771294 100644 --- a/ui-tui/src/__tests__/virtualHeights.test.ts +++ b/ui-tui/src/__tests__/virtualHeights.test.ts @@ -82,6 +82,28 @@ describe('virtual height estimates', () => { ).toBe(estimatedMsgHeight(toolsOnly, 80, { compact: false, details: false })) }) + it('treats historical thinking blocks as collapsed unless explicitly expanded', () => { + const msg: Msg = { role: 'assistant', text: 'ok', thinking: 'line 1\nline 2\nline 3' } + + expect( + estimatedMsgHeight(msg, 80, { + compact: false, + details: true, + thinkingExpanded: false, + thinkingVisible: true, + toolsVisible: false + }) + ).toBeLessThan( + estimatedMsgHeight(msg, 80, { + compact: false, + details: true, + thinkingExpanded: true, + thinkingVisible: true, + toolsVisible: false + }) + ) + }) + it('reserves two extra rows for the inter-turn separator on non-first user messages', () => { const msg: Msg = { role: 'user', text: 'follow-up question' } const base = estimatedMsgHeight(msg, 80, { compact: false, details: false }) diff --git a/ui-tui/src/app/useMainApp.ts b/ui-tui/src/app/useMainApp.ts index 0756c2fd0ef11..555d66f073fff 100644 --- a/ui-tui/src/app/useMainApp.ts +++ b/ui-tui/src/app/useMainApp.ts @@ -353,6 +353,10 @@ export function useMainApp(gw: GatewayClient) { const [thinkingDetailsMode, toolsDetailsMode] = detailsLayoutKey.split(':') const thinkingDetailsVisible = thinkingDetailsMode !== 'hidden' const toolsDetailsVisible = toolsDetailsMode !== 'hidden' + + const historyThinkingExpanded = + thinkingDetailsVisible && (ui.detailsModeCommandOverride || ui.sections.thinking === 'expanded') + const detailsVisible = thinkingDetailsVisible || toolsDetailsVisible const userPromptWidth = composerPromptWidth(ui.theme.brand.prompt) const heightCacheKey = `${ui.sid ?? 'draft'}:${cols}:${userPromptWidth}:${ui.compact ? '1' : '0'}:${detailsLayoutKey}` @@ -390,6 +394,7 @@ export function useMainApp(gw: GatewayClient) { }), virtualRows[index]!.msg ), + thinkingExpanded: historyThinkingExpanded, thinkingVisible: thinkingDetailsVisible, toolsVisible: toolsDetailsVisible, userPrompt: ui.theme.brand.prompt, @@ -399,6 +404,7 @@ export function useMainApp(gw: GatewayClient) { cols, detailsVisible, firstUserIdx, + historyThinkingExpanded, thinkingDetailsVisible, toolsDetailsVisible, ui.compact, diff --git a/ui-tui/src/components/messageLine.tsx b/ui-tui/src/components/messageLine.tsx index 09b1c78a1ad44..1e7c273a2bb9f 100644 --- a/ui-tui/src/components/messageLine.tsx +++ b/ui-tui/src/components/messageLine.tsx @@ -34,6 +34,7 @@ export const MessageLine = memo(function MessageLine({ detailsMode = 'collapsed', detailsModeCommandOverride = false, isStreaming = false, + liveDetails = false, msg, prev, sections, @@ -81,6 +82,7 @@ export const MessageLine = memo(function MessageLine({ Date.now()) // Local toggles own the open state once mounted. Init from the resolved // section visibility so default-expanded sections (thinking/tools) render @@ -735,7 +740,7 @@ export const ToolTrail = memo(function ToolTrail({ // label. This only affects the initial mount value; the re-sync effect // below deliberately does NOT re-apply it, so a manual collapse still // sticks (see the no-OR-at-effect-time warning above, #14968). - const [openThinking, setOpenThinking] = useState(visible.thinking === 'expanded' || reasoningAlwaysVisible) + const [openThinking, setOpenThinking] = useState(thinkingDefaultExpanded || reasoningAlwaysVisible) const [openTools, setOpenTools] = useState(visible.tools === 'expanded') const [openSubagents, setOpenSubagents] = useState(visible.subagents === 'expanded') const [deepSubagents, setDeepSubagents] = useState(visible.subagents === 'expanded') @@ -766,11 +771,11 @@ export const ToolTrail = memo(function ToolTrail({ return } - setOpenThinking(visible.thinking === 'expanded') + setOpenThinking(thinkingDefaultExpanded) setOpenTools(visible.tools === 'expanded') setOpenSubagents(visible.subagents === 'expanded') setOpenMeta(visible.activity === 'expanded') - }, [visible]) + }, [thinkingDefaultExpanded, visible]) const cot = useMemo(() => thinkingPreview(reasoning, 'full', THINKING_COT_MAX), [reasoning]) diff --git a/ui-tui/src/lib/virtualHeights.ts b/ui-tui/src/lib/virtualHeights.ts index bb470da892321..cf1ebd95d7f62 100644 --- a/ui-tui/src/lib/virtualHeights.ts +++ b/ui-tui/src/lib/virtualHeights.ts @@ -74,6 +74,7 @@ export const estimatedMsgHeight = ( details, leadGap = false, thinkingVisible = details, + thinkingExpanded = thinkingVisible, toolsVisible = details, userPrompt = '', withSeparator = false @@ -81,6 +82,7 @@ export const estimatedMsgHeight = ( compact: boolean details: boolean leadGap?: boolean + thinkingExpanded?: boolean thinkingVisible?: boolean toolsVisible?: boolean userPrompt?: string @@ -124,7 +126,7 @@ export const estimatedMsgHeight = ( if (hasVisibleDetails) { h += (hasVisibleTools ? (msg.tools?.length ?? 0) : 0) + - (hasVisibleThinking ? wrappedLines(msg.thinking ?? '', bodyWidth) : 0) + (hasVisibleThinking ? (thinkingExpanded ? wrappedLines(msg.thinking ?? '', bodyWidth) : 1) : 0) if (msg.role === 'assistant' && /\S/.test(msg.text)) { h += 2 From 2b0b4a219195e9203e83efb9f1b87cdaabf45f76 Mon Sep 17 00:00:00 2001 From: Jordy Elfferich Date: Sat, 8 Aug 2026 23:50:13 +0200 Subject: [PATCH 339/376] feat(tui): auto-collapse reasoning blocks only when the reasoning phase ends MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Under display.sections.thinking: collapsed, the TUI now keeps the LIVE reasoning panel open while reasoning streams and collapses it the moment the reasoning phase ends (first tool call, final answer, or new turn). Previously 'collapsed' meant the panel was always collapsed — including the currently-streaming reasoning — and there was no way to get 'expanded while live, collapsed when done'. This makes 'collapsed' an auto preference: - turnController tags the open reasoning segment isLiveReasoning and seals the tag in endReasoningPhase/closeReasoningSegment - streamingAssistant passes reasoningActive only to the live segment, so sealed reasoning segments from earlier phases stay collapsed - ToolTrail auto-opens while reasoningActive under collapsed mode; expanded/hidden/MoA-reference semantics are unchanged Adds thinkingLiveCollapse.test.tsx covering open-on-stream, close-on- finish (including mid-turn rerender), and the expanded-mode no-op. (cherry picked from commit 6ef4ef77d3a0837c8ef5c5dac3de74f352bd637d) --- .../__tests__/thinkingLiveCollapse.test.tsx | 118 ++++++++++++++++++ ui-tui/src/app/turnController.ts | 13 +- ui-tui/src/components/messageLine.tsx | 4 + ui-tui/src/components/streamingAssistant.tsx | 1 + ui-tui/src/components/thinking.tsx | 14 +++ ui-tui/src/types.ts | 5 + 6 files changed, 152 insertions(+), 3 deletions(-) create mode 100644 ui-tui/src/__tests__/thinkingLiveCollapse.test.tsx diff --git a/ui-tui/src/__tests__/thinkingLiveCollapse.test.tsx b/ui-tui/src/__tests__/thinkingLiveCollapse.test.tsx new file mode 100644 index 0000000000000..0209f02fe32b9 --- /dev/null +++ b/ui-tui/src/__tests__/thinkingLiveCollapse.test.tsx @@ -0,0 +1,118 @@ +import { PassThrough } from 'stream' + +import { renderSync } from '@hermes/ink' +import React from 'react' +import { describe, expect, it } from 'vitest' + +import { ToolTrail } from '../components/thinking.js' +import { stripAnsi } from '../lib/text.js' +import { DEFAULT_THEME } from '../theme.js' + +const flushEffects = async () => { + // Passive effects + the re-render they trigger need a few macrotask + // turns (React's scheduler uses MessageChannel) before the next frame + // paints — setTimeout(0)-class waits, not setImmediate (which can land + // in the wrong phase and observe the pre-effect frame). + for (let i = 0; i < 10; i++) { + await new Promise(resolve => setTimeout(resolve, 5)) + } +} + +const mountTrail = (reasoningActive: boolean, sections?: Record) => { + const stdout = new PassThrough() + const stdin = new PassThrough() + const stderr = new PassThrough() + let output = '' + + Object.assign(stdout, { columns: 60, isTTY: false, rows: 20 }) + Object.assign(stdin, { isTTY: false }) + Object.assign(stderr, { isTTY: false }) + stdout.on('data', chunk => { + output += chunk.toString() + }) + + const instance = renderSync( + , + { + patchConsole: false, + stderr: stderr as NodeJS.WriteStream, + stdin: stdin as NodeJS.ReadStream, + stdout: stdout as NodeJS.WriteStream + } + ) + + // The PassThrough accumulates every repaint, and a collapsed panel stops + // repainting entirely once settled — so assert on the FINAL chevron state + // in the accumulated output rather than the tail after a clear(). + const finalChevronOpen = () => stripAnsi(output).lastIndexOf('▾ ') > stripAnsi(output).lastIndexOf('▸ ') + + return { finalChevronOpen, instance } +} + +describe('ToolTrail — collapsed mode auto-expands while reasoning is live', () => { + it('opens (▾) when reasoningActive is true under sections.thinking: collapsed', async () => { + const { finalChevronOpen, instance } = mountTrail(true) + + await flushEffects() + + expect(finalChevronOpen()).toBe(true) + + instance.unmount() + instance.cleanup() + }) + + it('collapses (▸) when reasoningActive is false under sections.thinking: collapsed', async () => { + const { finalChevronOpen, instance } = mountTrail(false) + + await flushEffects() + + expect(finalChevronOpen()).toBe(false) + + instance.unmount() + instance.cleanup() + }) + + it('closes the panel when the reasoning phase ends mid-turn (rerender)', async () => { + const { finalChevronOpen, instance } = mountTrail(true) + + await flushEffects() + + expect(finalChevronOpen()).toBe(true) + + // Reasoning phase finished (final answer / tool call started) — the + // turn's reasoningActive drops and the panel must collapse. + instance.rerender( + + ) + + await flushEffects() + + expect(finalChevronOpen()).toBe(false) + + instance.unmount() + instance.cleanup() + }) + + it('leaves expanded-mode panels fully manual (no forced collapse)', async () => { + const { finalChevronOpen, instance } = mountTrail(false, { thinking: 'expanded' }) + + await flushEffects() + + // `expanded` is a manual preference: reasoningActive=false must NOT + // force it closed (the auto behavior only applies to `collapsed`). + expect(finalChevronOpen()).toBe(true) + + instance.unmount() + instance.cleanup() + }) +}) diff --git a/ui-tui/src/app/turnController.ts b/ui-tui/src/app/turnController.ts index 91380466487da..3314c3554516c 100644 --- a/ui-tui/src/app/turnController.ts +++ b/ui-tui/src/app/turnController.ts @@ -266,6 +266,12 @@ class TurnController { endReasoningPhase() { this.reasoningStreamingTimer = clear(this.reasoningStreamingTimer) + // Seal any open reasoning segment so its isLiveReasoning flag drops the + // moment the reasoning phase ends — the panel must stop tracking the + // turn's global reasoningActive, not stay "live" for the rest of the turn. + if (this.reasoningSegmentIndex !== null) { + this.syncReasoningSegment(false) + } patchTurnState({ reasoningActive: false, reasoningStreaming: false }) } @@ -359,7 +365,7 @@ class TurnController { }) } - private syncReasoningSegment() { + private syncReasoningSegment(live = true) { const thinking = this.activeReasoningText.trim() if (!thinking) { @@ -372,7 +378,8 @@ class TurnController { text: '', thinking, thinkingTokens: estimateTokensRough(thinking), - toolTokens: this.toolTokenAcc || undefined + toolTokens: this.toolTokenAcc || undefined, + ...(live ? { isLiveReasoning: true } : {}) } if (this.reasoningSegmentIndex === null) { @@ -386,7 +393,7 @@ class TurnController { } private closeReasoningSegment() { - this.syncReasoningSegment() + this.syncReasoningSegment(false) this.activeReasoningText = '' this.reasoningSegmentIndex = null } diff --git a/ui-tui/src/components/messageLine.tsx b/ui-tui/src/components/messageLine.tsx index 1e7c273a2bb9f..ba59f6a343dfb 100644 --- a/ui-tui/src/components/messageLine.tsx +++ b/ui-tui/src/components/messageLine.tsx @@ -37,6 +37,7 @@ export const MessageLine = memo(function MessageLine({ liveDetails = false, msg, prev, + reasoningActive = false, sections, t, tools = [] @@ -84,6 +85,7 @@ export const MessageLine = memo(function MessageLine({ detailsMode={detailsMode} preferExpandedThinking={liveDetails} reasoning={thinking} + reasoningActive={reasoningActive} reasoningAlwaysVisible={msg.isMoaReference} reasoningTokens={msg.thinkingTokens} sections={sections} @@ -249,6 +251,7 @@ export const MessageLine = memo(function MessageLine({ detailsMode={detailsMode} preferExpandedThinking={liveDetails} reasoning={thinking} + reasoningActive={reasoningActive} reasoningTokens={msg.thinkingTokens} sections={sections} t={t} @@ -309,6 +312,7 @@ interface MessageLineProps { // lead gap (see domain/blockLayout.ts::hasLeadGap). Undefined at the top of // the transcript or when spacing is irrelevant. prev?: Msg + reasoningActive?: boolean sections?: SectionVisibility t: Theme tools?: ActiveTool[] diff --git a/ui-tui/src/components/streamingAssistant.tsx b/ui-tui/src/components/streamingAssistant.tsx index 9c60bd5077bc0..2cf259a959468 100644 --- a/ui-tui/src/components/streamingAssistant.tsx +++ b/ui-tui/src/components/streamingAssistant.tsx @@ -79,6 +79,7 @@ export const StreamingAssistant = memo(function StreamingAssistant({ liveDetails msg={block.msg} prev={prev} + reasoningActive={block.msg.isLiveReasoning === true} sections={sections} t={ui.theme} {...(block.tools ? { tools: block.tools } : {})} diff --git a/ui-tui/src/components/thinking.tsx b/ui-tui/src/components/thinking.tsx index fdf8ac11d900a..d3225bee6a7ea 100644 --- a/ui-tui/src/components/thinking.tsx +++ b/ui-tui/src/components/thinking.tsx @@ -777,6 +777,20 @@ export const ToolTrail = memo(function ToolTrail({ setOpenMeta(visible.activity === 'expanded') }, [thinkingDefaultExpanded, visible]) + // `collapsed` is an auto preference: keep the panel open while reasoning + // is live (stream pulses keep `reasoningActive` true) and collapse it the + // moment the reasoning phase ends (`endReasoningPhase` flips it false). + // `expanded` stays fully manual, `hidden` never renders content, and MoA + // reference panels (reasoningAlwaysVisible) are left alone. + const thinkingAuto = visible.thinking === 'collapsed' && !reasoningAlwaysVisible + useEffect(() => { + if (!thinkingAuto) { + return + } + + setOpenThinking(reasoningActive) + }, [thinkingAuto, reasoningActive]) + const cot = useMemo(() => thinkingPreview(reasoning, 'full', THINKING_COT_MAX), [reasoning]) // Spawn-tree derivations must live above any early return so React's diff --git a/ui-tui/src/types.ts b/ui-tui/src/types.ts index 49c4cef4189a5..c976913522d6a 100644 --- a/ui-tui/src/types.ts +++ b/ui-tui/src/types.ts @@ -125,6 +125,11 @@ export interface Msg { // user-facing mixture-of-agents process the user opted into, so it stays // visible even when `display.sections.thinking` is hidden. isMoaReference?: boolean + // True only while this trail segment's reasoning is being streamed live by + // the current turn (see turnController's syncReasoningSegment). Sealed + // reasoning segments from earlier in the turn carry no flag, so the TUI can + // tell "the reasoning happening right now" apart from finished blocks. + isLiveReasoning?: boolean thinkingTokens?: number toolTokens?: number tools?: string[] From 10bf145e303e4ad8e49f743b52cb9925f0144331 Mon Sep 17 00:00:00 2001 From: Isak du Plessis Date: Wed, 22 Jul 2026 11:22:19 +0200 Subject: [PATCH 340/376] feat(desktop): collapse thinking by default (cherry picked from commit 14fd1aace2c76c31f3b48e4b9bf9943c74582d21) --- .../src/app/settings/appearance-settings.tsx | 20 +++++++++++++++++++ .../assistant-ui/thread/message-parts.tsx | 13 ++++++------ .../assistant-ui/thread/streaming.test.tsx | 18 +++++++++++++++++ apps/desktop/src/i18n/ar.ts | 2 ++ apps/desktop/src/i18n/en.ts | 2 ++ apps/desktop/src/i18n/ja.ts | 2 ++ apps/desktop/src/i18n/runtime.test.ts | 6 ++++++ apps/desktop/src/i18n/types.ts | 2 ++ apps/desktop/src/i18n/zh-hant.ts | 2 ++ apps/desktop/src/i18n/zh.ts | 2 ++ .../desktop/src/store/reasoning-disclosure.ts | 16 +++++++++++++++ 11 files changed, 79 insertions(+), 6 deletions(-) create mode 100644 apps/desktop/src/store/reasoning-disclosure.ts diff --git a/apps/desktop/src/app/settings/appearance-settings.tsx b/apps/desktop/src/app/settings/appearance-settings.tsx index 998f84bb6cfcc..a53757f0fb603 100644 --- a/apps/desktop/src/app/settings/appearance-settings.tsx +++ b/apps/desktop/src/app/settings/appearance-settings.tsx @@ -16,6 +16,7 @@ import { $backdrop, setBackdrop } from '@/store/backdrop' import { $embedAllowed, $embedMode, clearEmbedAllowed, type EmbedMode, setEmbedMode } from '@/store/embed-consent' import { $activeGatewayProfile, $profiles, normalizeProfileKey } from '@/store/profile' import { $reactionsEnabled, setReactionsEnabled } from '@/store/reactions-enabled' +import { $reasoningCollapsedByDefault, setReasoningCollapsedByDefault } from '@/store/reasoning-disclosure' import { $toolViewMode, setToolViewMode } from '@/store/tool-view' import { $translucency, setTranslucency } from '@/store/translucency' import { $zoomPercent, setZoomPercent } from '@/store/zoom' @@ -248,6 +249,7 @@ export function AppearanceSettings() { const { t, isSavingLocale } = useI18n() const { themeName, mode, resolvedMode, availableThemes, setTheme, setMode } = useTheme() const toolViewMode = useStore($toolViewMode) + const reasoningCollapsedByDefault = useStore($reasoningCollapsedByDefault) const zoomPercent = useStore($zoomPercent) const embedMode = useStore($embedMode) const embedAllowed = useStore($embedAllowed) @@ -510,6 +512,24 @@ export function AppearanceSettings() { title={a.toolViewTitle} /> + { + triggerHaptic('selection') + setReasoningCollapsedByDefault(id === 'on') + }} + options={[ + { id: 'off', label: t.common.off }, + { id: 'on', label: t.common.on } + ]} + value={reasoningCollapsedByDefault ? 'on' : 'off'} + /> + } + description={a.reasoningCollapsedDesc} + title={a.reasoningCollapsedTitle} + /> + diff --git a/apps/desktop/src/components/assistant-ui/thread/message-parts.tsx b/apps/desktop/src/components/assistant-ui/thread/message-parts.tsx index dd0a2e2204f42..e328867af9e97 100644 --- a/apps/desktop/src/components/assistant-ui/thread/message-parts.tsx +++ b/apps/desktop/src/components/assistant-ui/thread/message-parts.tsx @@ -4,6 +4,7 @@ import { useAuiState, useMessagePartReasoning } from '@assistant-ui/react' +import { useStore } from '@nanostores/react' import { type ComponentProps, type FC, type ReactNode, useEffect, useRef, useState } from 'react' import { ClarifyTool } from '@/components/assistant-ui/clarify-tool' @@ -21,6 +22,7 @@ import { generatedImageFromResult } from '@/lib/generated-images' import { separateGluedReasoningBlocks } from '@/lib/reasoning-blocks' import { useEnterAnimation } from '@/lib/use-enter-animation' import { cn } from '@/lib/utils' +import { $reasoningCollapsedByDefault } from '@/store/reasoning-disclosure' const ImageGenerateTool: FC = props => { const { args, result } = props @@ -103,10 +105,9 @@ const ThinkingDisclosure: FC<{ timerKey: string }> = ({ children, messageRunning = false, pending = false, timerKey }) => { const { t } = useI18n() - // `null` = no explicit user toggle yet, defer to the streaming default. - // The default is "auto-open while streaming, auto-collapse when done" so - // reasoning surfaces a live preview without manual interaction. The first - // explicit toggle wins from then on. + const reasoningCollapsedByDefault = useStore($reasoningCollapsedByDefault) + // `null` = no explicit user toggle yet. Live reasoning remains visible by + // default, unless the user opts into the low-jitter collapsed presentation. const [userOpen, setUserOpen] = useState(null) const elapsed = useElapsedSeconds(pending, timerKey) const thoughtFor = useMeasuredDuration(pending, timerKey) @@ -114,8 +115,8 @@ const ThinkingDisclosure: FC<{ const contentRef = useRef(null) const enterRef = useEnterAnimation(messageRunning, timerKey) - const open = userOpen ?? pending - const isPreview = pending && userOpen === null + const open = userOpen ?? (pending && !reasoningCollapsedByDefault) + const isPreview = pending && userOpen === null && !reasoningCollapsedByDefault // Three ways a finished block can report itself. With a measured duration it // says so, unless the timer's whole seconds round it to "0s" — accurate and diff --git a/apps/desktop/src/components/assistant-ui/thread/streaming.test.tsx b/apps/desktop/src/components/assistant-ui/thread/streaming.test.tsx index 0cedda1e7b1f0..566e5b04e8168 100644 --- a/apps/desktop/src/components/assistant-ui/thread/streaming.test.tsx +++ b/apps/desktop/src/components/assistant-ui/thread/streaming.test.tsx @@ -3,6 +3,8 @@ import { act, fireEvent, render, screen, waitFor, within } from '@testing-librar import { useEffect, useState } from 'react' import { beforeEach, describe, expect, it, vi } from 'vitest' +import { $reasoningCollapsedByDefault } from '@/store/reasoning-disclosure' + import { Thread } from '.' const createdAt = new Date('2026-05-01T00:00:00.000Z') @@ -475,6 +477,7 @@ function DismissibleErrorHarness({ onDismissError }: { onDismissError: (messageI describe('assistant-ui streaming renderer', () => { beforeEach(() => { resizeObservers.clear() + $reasoningCollapsedByDefault.set(false) }) it('renders assistant text incrementally before completion', async () => { @@ -602,6 +605,21 @@ describe('assistant-ui streaming renderer', () => { expect(container.textContent).not.toContain('```ts') }) + it('keeps streaming reasoning collapsed by default when the preference is enabled', () => { + $reasoningCollapsedByDefault.set(true) + + const { container } = render() + const thinkingToggle = within(container).getByRole('button', { name: /thinking/i }) + + expect(thinkingToggle.getAttribute('aria-expanded')).toBe('false') + expect(container.querySelector('[data-slot="aui_reasoning-text"]')).toBeNull() + + fireEvent.click(thinkingToggle) + + expect(thinkingToggle.getAttribute('aria-expanded')).toBe('true') + expect(container.querySelector('[data-slot="aui_reasoning-text"]')?.textContent).toContain('const answer = 42') + }) + it('renders reasoning text without a leading token space', () => { const { container } = render() const ui = within(container) diff --git a/apps/desktop/src/i18n/ar.ts b/apps/desktop/src/i18n/ar.ts index 1397b6f4aa208..87ef6febddf94 100644 --- a/apps/desktop/src/i18n/ar.ts +++ b/apps/desktop/src/i18n/ar.ts @@ -397,6 +397,8 @@ export const ar = defineLocale({ colorModeDesc: 'اختر الوضع الفاتح أو الداكن أو اتبع النظام.', toolViewTitle: 'عرض الأدوات', toolViewDesc: 'تحكم في كيفية عرض نشاط الأدوات داخل المحادثة.', + reasoningCollapsedTitle: 'طي التفكير افتراضيًا', + reasoningCollapsedDesc: 'أبقِ التفكير المتدفق متاحًا دون توسيعه حتى تفتحه.', translucencyTitle: 'شفافية النافذة', translucencyDesc: 'إظهار سطح المكتب من خلال النافذة بالكامل. متاح على macOS وWindows فقط.', backdropTitle: 'خلفية النافذة', diff --git a/apps/desktop/src/i18n/en.ts b/apps/desktop/src/i18n/en.ts index f5f021424c04c..27cd9c38bc54e 100644 --- a/apps/desktop/src/i18n/en.ts +++ b/apps/desktop/src/i18n/en.ts @@ -458,6 +458,8 @@ export const en: Translations = { colorModeDesc: 'Pick a fixed mode or let Hermes follow your system setting.', toolViewTitle: 'Tool Call Display', toolViewDesc: 'Product hides raw tool payloads; Technical shows full input/output.', + reasoningCollapsedTitle: 'Collapse thinking by default', + reasoningCollapsedDesc: 'Keep streamed reasoning available without expanding it until you open it.', uiScaleTitle: 'UI Scale', uiScaleDesc: (percent: number) => `Scales text and controls across the whole app. Cmd/Ctrl with +, - and 0 also works. Current: ${percent}%.`, diff --git a/apps/desktop/src/i18n/ja.ts b/apps/desktop/src/i18n/ja.ts index db3799c83b2da..ddfac57f4c8a6 100644 --- a/apps/desktop/src/i18n/ja.ts +++ b/apps/desktop/src/i18n/ja.ts @@ -317,6 +317,8 @@ export const ja = defineLocale({ colorModeDesc: '固定モードを選ぶか、Hermes をシステム設定に合わせます。', toolViewTitle: 'ツール呼び出しの表示', toolViewDesc: 'プロダクト表示は生のツールペイロードを隠し、テクニカル表示は入出力をすべて表示します。', + reasoningCollapsedTitle: '思考ブロックをデフォルトで折りたたむ', + reasoningCollapsedDesc: 'ストリーミング中の推論を、開くまで折りたたんだまま利用できるようにします。', uiScaleTitle: 'UI スケール', uiScaleDesc: (percent: number) => `アプリ全体の文字と UI を拡大縮小します。Cmd/Ctrl と +、-、0 でも変更できます。現在: ${percent}%`, diff --git a/apps/desktop/src/i18n/runtime.test.ts b/apps/desktop/src/i18n/runtime.test.ts index 499fc1de6c9ca..dcf743dc5a652 100644 --- a/apps/desktop/src/i18n/runtime.test.ts +++ b/apps/desktop/src/i18n/runtime.test.ts @@ -44,6 +44,12 @@ describe('desktop i18n runtime translator', () => { setRuntimeI18nLocale('zh-hant') expect(translateNow('settings.appearance.title')).toBe('外觀') expect(translateNow('settings.nav.providerApiKeys')).toBe('API 金鑰') + + setRuntimeI18nLocale('ar') + expect(translateNow('settings.appearance.reasoningCollapsedTitle')).toBe('طي التفكير افتراضيًا') + expect(translateNow('settings.appearance.reasoningCollapsedDesc')).toBe( + 'أبقِ التفكير المتدفق متاحًا دون توسيعه حتى تفتحه.' + ) }) it('keeps translated settings field copy addressable from schema keys', () => { diff --git a/apps/desktop/src/i18n/types.ts b/apps/desktop/src/i18n/types.ts index ac803a37821c6..fd2fda62ee5f9 100644 --- a/apps/desktop/src/i18n/types.ts +++ b/apps/desktop/src/i18n/types.ts @@ -361,6 +361,8 @@ export interface Translations { colorModeDesc: string toolViewTitle: string toolViewDesc: string + reasoningCollapsedTitle: string + reasoningCollapsedDesc: string uiScaleTitle: string uiScaleDesc: (percent: number) => string terminalFontTitle: string diff --git a/apps/desktop/src/i18n/zh-hant.ts b/apps/desktop/src/i18n/zh-hant.ts index d3100740451a8..8ee30ed1f2192 100644 --- a/apps/desktop/src/i18n/zh-hant.ts +++ b/apps/desktop/src/i18n/zh-hant.ts @@ -309,6 +309,8 @@ export const zhHant = defineLocale({ colorModeDesc: '選擇固定模式,或讓 Hermes 跟隨系統設定。', toolViewTitle: '工具呼叫顯示', toolViewDesc: '產品模式會隱藏原始工具 payload;技術模式會顯示完整輸入/輸出。', + reasoningCollapsedTitle: '預設摺疊推理過程', + reasoningCollapsedDesc: '保留串流推理內容,但在您開啟前維持摺疊。', uiScaleTitle: '介面縮放', uiScaleDesc: (percent: number) => `縮放整個應用程式的文字與介面。也可使用 Cmd/Ctrl 加 +、- 或 0 調整。目前:${percent}%`, diff --git a/apps/desktop/src/i18n/zh.ts b/apps/desktop/src/i18n/zh.ts index a1b03f9e450c2..b8e0696903e9a 100644 --- a/apps/desktop/src/i18n/zh.ts +++ b/apps/desktop/src/i18n/zh.ts @@ -448,6 +448,8 @@ export const zh: Translations = { colorModeDesc: '选择固定模式,或让 Hermes 跟随系统设置。', toolViewTitle: '工具调用显示', toolViewDesc: '产品模式隐藏原始工具数据;技术模式显示完整输入/输出。', + reasoningCollapsedTitle: '默认折叠推理过程', + reasoningCollapsedDesc: '保留流式推理内容,但在您打开前保持折叠。', uiScaleTitle: '界面缩放', uiScaleDesc: (percent: number) => `缩放整个应用的文字和界面。也可使用 Cmd/Ctrl 加 +、- 或 0 调整。当前:${percent}%`, diff --git a/apps/desktop/src/store/reasoning-disclosure.ts b/apps/desktop/src/store/reasoning-disclosure.ts new file mode 100644 index 0000000000000..08821f803411b --- /dev/null +++ b/apps/desktop/src/store/reasoning-disclosure.ts @@ -0,0 +1,16 @@ +import { atom } from 'nanostores' + +import { persistBoolean, storedBoolean } from '@/lib/storage' + +const REASONING_COLLAPSED_BY_DEFAULT_STORAGE_KEY = 'hermes.desktop.reasoning.collapsedByDefault' + +/** Desktop-local presentation preference; shared backend config must not be changed by a single window. */ +export const $reasoningCollapsedByDefault = atom( + storedBoolean(REASONING_COLLAPSED_BY_DEFAULT_STORAGE_KEY, false) +) + +$reasoningCollapsedByDefault.subscribe(value => persistBoolean(REASONING_COLLAPSED_BY_DEFAULT_STORAGE_KEY, value)) + +export function setReasoningCollapsedByDefault(value: boolean) { + $reasoningCollapsedByDefault.set(value) +} From ac0d0f889750e31e8344f64f55f69ac8e1f767fe Mon Sep 17 00:00:00 2001 From: chillerno1 Date: Wed, 29 Jul 2026 18:45:52 +1000 Subject: [PATCH 341/376] fix(desktop): flip zone collapse chevron to action direction MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Collapsed tool zones (terminal/logs) kept a down chevron after minimize, so the restore affordance looked like another collapse. Point the icon in the action direction — down when expanded, up when collapsed — matching master-detail collapsible detail headers. Same fix for floating panes. (cherry picked from commit 3392aeb9dba0ed9e9cabfb96b6010e46fa453ca1) --- .../pane-shell/tree/renderer/floating-panes.test.tsx | 5 +++++ .../components/pane-shell/tree/renderer/floating-panes.tsx | 2 +- .../src/components/pane-shell/tree/renderer/tree-group.tsx | 6 ++++-- 3 files changed, 10 insertions(+), 3 deletions(-) diff --git a/apps/desktop/src/components/pane-shell/tree/renderer/floating-panes.test.tsx b/apps/desktop/src/components/pane-shell/tree/renderer/floating-panes.test.tsx index b5e97640dfa29..62b825659f1b6 100644 --- a/apps/desktop/src/components/pane-shell/tree/renderer/floating-panes.test.tsx +++ b/apps/desktop/src/components/pane-shell/tree/renderer/floating-panes.test.tsx @@ -167,6 +167,10 @@ describe('FloatingPanes (live DOM)', () => { const before = card()!.style.left const toggle = card()!.querySelector('button')! + const chevron = () => toggle.querySelector('i')! + + // Expanded: down chevron (fold). Collapsed: up chevron (restore). + expect(chevron().className).toContain('codicon-chevron-down') // The button is inside the drag handle — [data-floating-no-drag] must // stop it starting a drag. @@ -182,6 +186,7 @@ describe('FloatingPanes (live DOM)', () => { expect(document.querySelector('[data-testid="hud-body"]')).toBeNull() expect(card()!.style.height).toBe('') + expect(chevron().className).toContain('codicon-chevron-up') }) it('renders one card per floating contribution', () => { diff --git a/apps/desktop/src/components/pane-shell/tree/renderer/floating-panes.tsx b/apps/desktop/src/components/pane-shell/tree/renderer/floating-panes.tsx index a2344c4dad88f..b4fea56fb7c01 100644 --- a/apps/desktop/src/components/pane-shell/tree/renderer/floating-panes.tsx +++ b/apps/desktop/src/components/pane-shell/tree/renderer/floating-panes.tsx @@ -172,7 +172,7 @@ function FloatingPane({ pane }: { pane: Contribution }) { onClick={toggleCollapsed} type="button" > - + diff --git a/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.tsx b/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.tsx index 6e7abf05ccf3e..cdf9081dd3a59 100644 --- a/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.tsx +++ b/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.tsx @@ -138,7 +138,9 @@ function ZoneMenu({ })} {minimizable && renderActionItem(kit, { - icon: minimized ? 'chevron-down' : 'chevron-up', + // Same action-direction contract as the strip button below: the + // icon points where the zone will GO (restore opens upward). + icon: minimized ? 'chevron-up' : 'chevron-down', label: minimized ? t.zones.restore : t.zones.minimize, onSelect: () => setTreeGroupMinimized(nodeId, !minimized) })} @@ -436,7 +438,7 @@ export function TreeGroup({ onPointerDown={e => e.stopPropagation()} type="button" > - + )} From c6ec8a95e6ed95963ed618856d5ec827ba04aeb7 Mon Sep 17 00:00:00 2001 From: chillerno1 Date: Fri, 31 Jul 2026 07:48:27 +1000 Subject: [PATCH 342/376] test(desktop): cover docked zone chevron direction (cherry picked from commit 95358428f95175ef57376da334b69279618cce6a) --- .../tree/renderer/tree-group.test.tsx | 76 +++++++++++++++++++ 1 file changed, 76 insertions(+) create mode 100644 apps/desktop/src/components/pane-shell/tree/renderer/tree-group.test.tsx diff --git a/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.test.tsx b/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.test.tsx new file mode 100644 index 0000000000000..8241ae1b50f8f --- /dev/null +++ b/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.test.tsx @@ -0,0 +1,76 @@ +import { act, type ReactNode } from 'react' +import { createRoot, type Root } from 'react-dom/client' +import { afterEach, describe, expect, it, vi } from 'vitest' + +import { registry } from '@/contrib/registry' + +import type { GroupNode } from '../model' + +import { TreeGroup } from './tree-group' + +let root: null | Root = null +let container: HTMLDivElement | null = null +let disposePane: (() => void) | null = null + +function render(ui: ReactNode) { + if (!container) { + container = globalThis.document.createElement('div') + globalThis.document.body.append(container) + root = createRoot(container) + } + + act(() => { + root!.render(ui) + }) +} + +function terminalGroup(minimized: boolean): GroupNode { + return { + active: 'terminal', + headerHidden: false, + id: 'terminal-zone', + minimized, + panes: ['terminal'], + type: 'group' + } +} + +const toggle = (label: string) => + globalThis.document.querySelector( + `[data-tree-group="terminal-zone"] button[aria-label="${label}"]` + )! + +afterEach(() => { + if (root) { + act(() => root!.unmount()) + } + + container?.remove() + disposePane?.() + root = null + container = null + disposePane = null + vi.unstubAllGlobals() +}) + +describe('TreeGroup', () => { + it('points the docked-zone chevron in the collapse or restore action direction', () => { + disposePane = registry.register({ + area: 'panes', + data: { height: '12rem' }, + id: 'terminal', + render: () =>
Terminal
, + title: 'Terminal' + }) + // jsdom does not implement CSS.escape, which the real tab-strip effect uses. + vi.stubGlobal('CSS', { escape: (value: string) => value }) + + render() + + expect(toggle('Minimize').querySelector('i')!.className).toContain('codicon-chevron-down') + + render() + + expect(toggle('Restore').querySelector('i')!.className).toContain('codicon-chevron-up') + }) +}) From 505f289d37dc1ce42d64a22a8e40d85c0154020d Mon Sep 17 00:00:00 2001 From: Eros <94871490+Eros-ITA@users.noreply.github.com> Date: Mon, 10 Aug 2026 12:41:51 +0200 Subject: [PATCH 343/376] feat(web): add collapse toggle for the chat side panel The right-hand chat panel (model picker + session list) is a fixed 240px column on desktop with no way to hide it. Add a collapse button (X) in the panel header and a floating 'panel' button over the terminal to reopen it, mirroring the collapsible app sidebar. The choice is persisted in localStorage (hermes-chat-panel-collapsed) so it survives reloads. (cherry picked from commit 854f1325a5490bcc968723a82746c939118f6b67) --- web/src/pages/ChatPage.test.tsx | 69 +++++++++++++++++++++++++++++++++ web/src/pages/ChatPage.tsx | 53 ++++++++++++++++++++++++- 2 files changed, 121 insertions(+), 1 deletion(-) diff --git a/web/src/pages/ChatPage.test.tsx b/web/src/pages/ChatPage.test.tsx index 907147048a776..c078aee434926 100644 --- a/web/src/pages/ChatPage.test.tsx +++ b/web/src/pages/ChatPage.test.tsx @@ -158,6 +158,25 @@ type CloseEventLike = { let container: HTMLDivElement; let root: Root; +// jsdom runs without an origin here (per-file @vitest-environment jsdom on a +// node-default config), so localStorage is undefined. Stub it so components +// that persist UI state (side panel collapse) can be exercised. +const localStorageMock = (() => { + let store: Record = {}; + return { + getItem: (key: string) => store[key] ?? null, + setItem: (key: string, value: string) => { + store[key] = String(value); + }, + removeItem: (key: string) => { + delete store[key]; + }, + clear: () => { + store = {}; + }, + }; +})(); + async function render(ui: ReactNode) { container = document.createElement("div"); document.body.append(container); @@ -220,6 +239,8 @@ beforeEach(() => { }, }); sessionStorage.clear(); + vi.stubGlobal("localStorage", localStorageMock); + localStorageMock.clear(); }); afterEach(async () => { @@ -250,6 +271,54 @@ describe("ChatPage", () => { }); }); +describe("ChatPage side panel collapse", () => { + async function renderChat() { + const { default: ChatPage } = await import("./ChatPage"); + await render( + + + , + ); + } + + it("collapses the desktop side panel and persists the choice", async () => { + localStorage.clear(); + await renderChat(); + await vi.waitFor(() => expect(FakeWebSocket.instances).toHaveLength(1)); + + const collapseButton = container.querySelector( + '[aria-label="Collapse chat side panel"]', + ); + expect(collapseButton).not.toBeNull(); + + await act(async () => { + collapseButton!.dispatchEvent( + new MouseEvent("click", { bubbles: true }), + ); + }); + + expect(localStorage.getItem("hermes-chat-panel-collapsed")).toBe("1"); + expect( + container.querySelector('[aria-label="Collapse chat side panel"]'), + ).toBeNull(); + expect( + container.querySelector('[aria-label="Show chat side panel"]'), + ).not.toBeNull(); + + // Reopening restores the panel and clears the persisted flag. + await act(async () => { + container + .querySelector('[aria-label="Show chat side panel"]')! + .dispatchEvent(new MouseEvent("click", { bubbles: true })); + }); + + expect(localStorage.getItem("hermes-chat-panel-collapsed")).toBe("0"); + expect( + container.querySelector('[aria-label="Collapse chat side panel"]'), + ).not.toBeNull(); + }); +}); + // The gated-mode ticket request runs before any socket exists, so a rejection // or a hang emits no `close` event and never arms PTY_CONNECTING_TIMEOUT_MS // (that timer is set after `new WebSocket`). Without its own deadline the tab diff --git a/web/src/pages/ChatPage.tsx b/web/src/pages/ChatPage.tsx index cfda855f39073..46ae92bb41a3f 100644 --- a/web/src/pages/ChatPage.tsx +++ b/web/src/pages/ChatPage.tsx @@ -290,6 +290,19 @@ export default function ChatPage({ isActive = true }: { isActive?: boolean }) { // tabs because the dep wouldn't change on tab switch. const [mobilePanelOpenRaw, setMobilePanelOpenRaw] = useState(false); const mobilePanelOpen = isActive && mobilePanelOpenRaw; + + // Collapse toggle for the desktop chat side panel (model + sessions), + // persisted in localStorage so the choice survives reloads. + const [chatPanelCollapsed, setChatPanelCollapsed] = useState( + () => localStorage.getItem("hermes-chat-panel-collapsed") === "1", + ); + const toggleChatPanel = useCallback(() => { + setChatPanelCollapsed((prev) => { + const next = !prev; + localStorage.setItem("hermes-chat-panel-collapsed", next ? "1" : "0"); + return next; + }); + }, []); const { setEnd, setTitle } = usePageHeader(); const [sessionTitleState, setSessionTitleState] = useState<{ scope: string; @@ -1717,15 +1730,53 @@ export default function ChatPage({ isActive = true }: { isActive?: boolean }) { + + {chatPanelCollapsed && ( + + )}
- {!narrow && ( + {!narrow && !chatPanelCollapsed && (