diff --git a/CHANGELOG.md b/CHANGELOG.md index 79535c04444..c991145ac3e 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -5,6 +5,8 @@ ### Fixed +- **Regenerating a response no longer refetches the entire conversation, so regenerate on a long chat is fast instead of stalling (and silently cancelling).** Clicking "regenerate" used to pull the whole transcript back from the server before it could rebuild the request, which on a large session caused a long UI freeze and could time out and silently drop the regeneration. The regeneration authority now reads only a bounded, sidecar-anchored tail of recent turns instead of the full transcript — and it is exact by construction: it takes the tail's prefix proof, keys, and rows from a single WAL-consistent database snapshot (a `data_version` check forces a full read if the database is written mid-snapshot), reuses the canonical projection and durable ordering so the bounded rows match the full reader byte-for-byte, and falls back to a full read whenever the skipped prefix isn't provably identical or the tail contains a duplicated turn key. If any of those invariants can't be met it reads the whole transcript exactly as before, so a concurrent edit or an unusual session can never produce a stale or reordered regeneration. Thanks @webtecnica. (#7204, closes #6826) + - **Loading the session sidebar is faster on installs with many delegated-subagent sessions — the subagent classification no longer runs a separate database probe per row.** Building `/api/sessions` used to open a `state.db` connection for every session that might be a delegated child, an N+1 pattern that grew with session count. The classification now reads the authoritative `state.db` source for all candidate rows through the existing batched overlay (one read-only connection, chunked 500-ID `IN` queries), and a rich-read failure recovers the remaining IDs in a single bounded source-only pass rather than falling back to per-row queries. Behavior is unchanged — a delegated child whose index row still says `webui`/`fork` while its `state.db` source is `subagent` is classified read-only exactly as before. Thanks @starship-s. (#7231) - **Reopening a conversation that contains pasted or generated images no longer shows the same turn twice.** A native-image user turn is stored two ways — the rich WebUI sidecar row (with the actual image) and the Hermes Agent state row (the same turn as model-facing text, with each image replaced by `[screenshot]`) — and transcript reconciliation treated them as two different messages, so reopening the chat rendered a duplicate `[screenshot]` bubble under the real one. Reconciliation now recognizes the scalar state row as a mirror of the rich image row (matching on the exact `[screenshot]` projection, full-precision timestamp, role/tool shape, and provenance) and keeps only the rich row, so the image and its metadata survive and the duplicate is gone. It fails closed: a literal `[screenshot]` typed by the user, an ambiguous or conflicting identity match, a timestamp mismatch, or malformed content all preserve both rows rather than risk collapsing two genuinely different turns. Thanks @starship-s. (#7230) diff --git a/api/models.py b/api/models.py index fa57cf68544..2ad051b21b0 100644 --- a/api/models.py +++ b/api/models.py @@ -8178,6 +8178,41 @@ def _state_db_active_rows_digest(rows) -> str: return digest.hexdigest() if stamped else '' +def _project_state_db_message(row, available, id_col, optional): + """Authoritative state.db row → WebUI message projection (#6826 r4). + + Shared by ``get_state_db_session_messages`` and the regeneration + single-snapshot helper so the bounded tail can never drift from the + canonical reader: JSON-decode tool_calls/reasoning payloads, omit + empty fields, keep durable row id private (``_state_db_row_id`` only for + real Agent api_content replays), and apply ``tool_name → name``. + """ + msg = { + 'role': row['role'], + 'content': row['content'], + 'timestamp': row['timestamp'], + } + for col in optional: + if col not in row.keys(): + continue + value = row[col] + if value in (None, ''): + continue + if col in {'tool_calls', 'reasoning_details', 'codex_reasoning_items', 'codex_message_items'}: + value = _json_loads_if_string(value) + msg[col] = value + if ( + id_col + and row['id'] is not None + and isinstance(msg.get('api_content'), str) + and msg['api_content'] + ): + msg['_state_db_row_id'] = row['id'] + if msg.get('role') == 'tool' and msg.get('tool_name') and not msg.get('name'): + msg['name'] = msg['tool_name'] + return msg + + @overload def get_state_db_session_messages( sid, @@ -8418,38 +8453,9 @@ def get_state_db_session_messages( msgs = [] for row in rows: - msg = { - 'role': row['role'], - 'content': row['content'], - 'timestamp': row['timestamp'], - } - # ``id`` is the durable SQLite row identity, not the WebUI's - # session-local stable message id. Keep it in a private - # provenance field so duplicate reconciliation can align a - # state.db sidecar without changing the existing ``id`` key - # used by WebUI transcript merge/dedup logic. - for col in optional: - if col not in row.keys(): - continue - value = row[col] - if value in (None, ''): - continue - if col in {'tool_calls', 'reasoning_details', 'codex_reasoning_items', 'codex_message_items'}: - value = _json_loads_if_string(value) - msg[col] = value - # Keep durable provenance only alongside a real Agent replay - # sidecar. Ordinary state.db rows must retain their historic - # shape and must not acquire internal bookkeeping fields. - if ( - id_col - and row['id'] is not None - and isinstance(msg.get('api_content'), str) - and msg['api_content'] - ): - msg['_state_db_row_id'] = row['id'] - if msg.get('role') == 'tool' and msg.get('tool_name') and not msg.get('name'): - msg['name'] = msg['tool_name'] - msgs.append(msg) + msgs.append( + _project_state_db_message(row, available, bool(id_col), optional) + ) except Exception: return _state_db_session_messages_result([], None, with_revision=with_revision) return _state_db_session_messages_result(msgs, revision, with_revision=with_revision) @@ -8598,6 +8604,157 @@ def get_state_db_session_message_keys_before_timestamp( return None +def get_state_db_regeneration_tail_snapshot( + sid, + floor, + *, + profile=None, +): + """Return prefix proof + bounded tail from ONE read transaction (#6826 r3). + + The regeneration guard must not suffer TOCTOU: the prefix summary, the + ordered prefix keys, and the bounded tail must come from the same SQLite + snapshot, otherwise a row inserted between the proof and the data read + can be silently omitted while the proof still authorizes the bounded path. + + Returns ``None`` when a stable single-connection snapshot cannot be + obtained (callers must fall back to the full read). Otherwise returns:: + + { + "prefix": {"count": N, "null_timestamp_count": M}, + "prefix_keys": [visible-key, ...], # rows < floor, db order + "tail": [message-dict, ...], # rows >= floor (bounded) + "tail_keys": [visible-key, ...], # rows >= floor, db order + } + """ + try: + import sqlite3 + except ImportError: + return None + + if not sid: + return None + try: + floor_ts = float(floor) + except (TypeError, ValueError): + return None + if isinstance(profile, str) and profile: + db_path = _get_profile_home(profile) / 'state.db' + if not db_path.exists(): + db_path = _active_state_db_path() + else: + db_path = _active_state_db_path() + if not db_path.exists(): + return {"prefix": {"count": 0, "null_timestamp_count": 0}, "prefix_keys": [], "tail": [], "tail_keys": []} + + try: + with closing(open_state_db_readonly(db_path)) as conn: + conn.row_factory = sqlite3.Row + cur = conn.cursor() + cur.execute("PRAGMA data_version") + dv_before = cur.fetchone()[0] + cur.execute("BEGIN") + cur.execute("PRAGMA table_info(messages)") + available = {str(row['name']) for row in cur.fetchall()} + if not {'session_id', 'role', 'content', 'timestamp'}.issubset(available): + cur.execute("ROLLBACK") + return None + active_clause = "" + if 'active' in available: + active_clause = " AND (active IS NULL OR active != 0)" + # durable order: id ASC when present (matches the canonical reader), + # else timestamp ASC + durable_order = 'id' if 'id' in available else 'timestamp' + # 1) prefix summary (rows with timestamp < floor) + cur.execute( + f""" + SELECT + COUNT(CASE WHEN timestamp IS NOT NULL AND timestamp < ? THEN 1 END) AS count, + COUNT(CASE WHEN timestamp IS NULL THEN 1 END) AS null_timestamp_count + FROM messages + WHERE session_id = ? {active_clause} + """, + (floor_ts, str(sid)), + ) + row = cur.fetchone() + prefix = { + "count": int(row["count"]) if row else 0, + "null_timestamp_count": int(row["null_timestamp_count"]) if row else 0, + } + # 2) ordered prefix keys (rows < floor) in durable order + prefix_key_cols = "COALESCE(role,'') AS role, COALESCE(content,'') AS content" + if 'tool_calls' in available: + prefix_key_cols += ", tool_calls" + if 'api_content' in available: + prefix_key_cols += ", api_content" + prefix_key_sql = ( + f"SELECT {prefix_key_cols} FROM messages " + "WHERE session_id = ? AND timestamp IS NOT NULL AND timestamp < ? " + f"{active_clause} ORDER BY {durable_order} ASC" + ) + try: + cur.execute(prefix_key_sql, (str(sid), floor_ts)) + except Exception: + cur.execute("ROLLBACK") + return None + prefix_keys = [ + _session_message_visible_key({ + "role": r["role"], + "content": r["content"], + "tool_calls": _json_loads_if_string(r["tool_calls"]) if "tool_calls" in r.keys() and r["tool_calls"] is not None else None, + "api_content": r["api_content"] if "api_content" in r.keys() else None, + }, normalize_workspace_prefix=True) + for r in cur.fetchall() + ] + # 3) bounded tail (rows >= floor) with the canonical projection + optional = [ + 'tool_call_id', 'tool_calls', 'tool_name', 'reasoning', + 'reasoning_details', 'codex_reasoning_items', 'reasoning_content', + 'codex_message_items', 'api_content', + ] + tail_select = ['id', 'role', 'content', 'timestamp'] if 'id' in available else ['role', 'content', 'timestamp'] + for col in optional + (['active'] if 'active' in available else []): + if col in available and col not in tail_select: + tail_select.append(col) + tail_sql = ( + f"SELECT {', '.join(tail_select)} FROM messages " + f"WHERE session_id = ? AND (timestamp IS NULL OR timestamp >= ?) {active_clause} " + f"ORDER BY {durable_order} ASC" + ) + cur.execute(tail_sql, (str(sid), floor_ts)) + tail_rows = [ + _project_state_db_message(r, available, 'id' in available, optional) + for r in cur.fetchall() + ] + tail_keys = [ + _session_message_visible_key( + { + "role": r.get("role"), + "content": r.get("content"), + "tool_calls": r.get("tool_calls"), + "api_content": r.get("api_content"), + }, + normalize_workspace_prefix=True, + ) + for r in tail_rows + ] + cur.execute("COMMIT") + cur.execute("PRAGMA data_version") + dv_after = cur.fetchone()[0] + if dv_before != dv_after: + # a concurrent WAL commit landed during the proof → stale + # snapshot; refuse the bounded path (TOCTOU). + return None + return { + "prefix": prefix, + "prefix_keys": prefix_keys, + "tail": tail_rows, + "tail_keys": tail_keys, + } + except Exception: + return None + + def get_state_db_session_summary(sid, *, profile=None) -> dict: """Return a cheap message count/timestamp summary for one state.db session.""" try: diff --git a/api/session_ops.py b/api/session_ops.py index 3a4f985ae96..c03fc09f787 100644 --- a/api/session_ops.py +++ b/api/session_ops.py @@ -120,7 +120,7 @@ def plan_regeneration(session, *, expected_revision=None, lock_held=False): """Prepare one canonical display/context pair for a locked regeneration.""" lock_context = nullcontext() if lock_held else _get_session_agent_lock(session.session_id) with lock_context: - rows, context = regeneration_state(session) + rows, context = regeneration_state(session, use_sidecar=True) revision = regeneration_revision_for(rows, session=session, context=context) if expected_revision is not None and expected_revision != revision: raise RegenerationUnavailable("stale_regeneration_revision") @@ -217,17 +217,148 @@ def regeneration_context(session): return regeneration_state(session)[1] -def regeneration_state(session): - """Read one immutable state.db snapshot and reconcile both transcript views.""" +_REGENERATION_SIDECAR_ANCHOR_BUDGET = 200 + + +def _sidecar_regeneration_read_floor(session): + """Return a state.db tail-read floor anchored by the already-loaded sidecar. + + #6826: regenerating a large session must not re-materialize the full + state.db transcript (a >1min stall on big sessions). When the in-memory + sidecar is a usable reconciliation base — append-only session with no + active truncation markers and timestamped rows — return the timestamp + floor for a bounded ``since_timestamp`` tail read. Rows at/after the + floor (including any gateway/server-applied tail the sidecar has not seen + yet) are re-read and merged, and rows the sidecar already carries are + deduplicated by the append-only merge, so the #6611 reconciliation + authority is preserved on the fast path. + + Returns ``None`` when the sidecar cannot anchor a tail read; callers then + fall back to the full state.db read (unchanged behavior). + """ + if getattr(session, "truncation_watermark", None) not in (None, ""): + return None + if getattr(session, "truncation_boundary", None) not in (None, ""): + return None + messages = getattr(session, "messages", None) + if not isinstance(messages, list) or not messages: + return None + from api.models import _message_timestamp_as_float + + timestamps = [_message_timestamp_as_float(message) for message in messages] + if any(timestamp is None for timestamp in timestamps): + return None + # Conservative anchor: re-read a bounded tip window so sub-second/clock + # drift near the sidecar tip cannot hide a concurrently appended state.db + # row, while the raw read stays tiny for huge sessions. + return min(timestamps[-_REGENERATION_SIDECAR_ANCHOR_BUDGET:]) + + +def _bounded_tail_snapshot_if_safe(session, read_floor): + """Return the bounded tail rows ONLY when it is provably identical to the + full read; otherwise None (caller must fall back to the full read). + + #6826 r3: the skipped state.db prefix (rows older than the floor) must be + represented identically in the sidecar — same count AND same ordered + visible identity — and the bounded tail must not repeat any skipped key + (occurrence-count collision: a new tail turn repeating an older prompt + would be mistaken for the old sidecar duplicate and dropped). The prefix + proof and the tail data come from ONE read transaction (no TOCTOU). + + Any mismatch, missing database, or uncertainty returns None, so the #6611 + regeneration authority never operates on an unreconciled view. + """ + sid = getattr(session, "session_id", None) + if not sid: + return None + profile = getattr(session, "profile", None) + from api.models import ( + _session_message_visible_key, + get_state_db_regeneration_tail_snapshot, + ) + + snap = get_state_db_regeneration_tail_snapshot(sid, read_floor, profile=profile) + if snap is None: + return None # cannot obtain a stable single-snapshot → full read + # Compression-anchor coverage: if the anchor predates the floor the bounded + # read can drop compacted-tail context rows (display may still match). + anchor = getattr(session, "compression_anchor_message_key", None) + if isinstance(anchor, dict): + try: + anchor_ts = float(anchor.get("ts")) + except (TypeError, ValueError): + anchor_ts = None + if anchor_ts is None or anchor_ts < read_floor: + return None + prefix = snap["prefix"] + if prefix.get("count") == 0 and prefix.get("null_timestamp_count") == 0: + # Empty skipped prefix: the bounded read already covers every row. + return snap["tail"] + # Non-empty skipped prefix: prove identical ordered visible identity. + sidecar_keys = [] + for message in getattr(session, "messages", None) or []: + if not isinstance(message, dict): + continue + try: + ts = float(message.get("timestamp")) + except (TypeError, ValueError): + ts = None + if ts is not None and ts < read_floor: + key = _session_message_visible_key(message) + if key is None: + return None + sidecar_keys.append(key) + if list(snap["prefix_keys"]) != sidecar_keys: + return None # mismatch → full read + # Occurrence-count collision (#6826 r3 #1): if any bounded-tail key ALSO + # occurs in the skipped prefix, the reconciler may drop the repeated tail + # row — fall back conservatively. + prefix_key_set = set(snap["prefix_keys"]) + for key in snap["tail_keys"]: + if key in prefix_key_set: + return None + # In-tail duplicates (#6826 r5): a repeated message wholly inside the + # bounded tail makes the reconciler's context dedup diverge from the full + # read (display may still match) → refuse the bounded path. + if len(snap["tail_keys"]) != len(set(snap["tail_keys"])): + return None + return snap["tail"] + + +def regeneration_state(session, *, use_sidecar=False): + """Read one immutable state.db snapshot and reconcile both transcript views. + + ``use_sidecar=True`` (#6826) anchors the state.db read to the already + loaded in-memory sidecar: only a bounded tail (``since_timestamp`` floor) + is re-read instead of the full transcript, and both views still route + through :func:`reconciled_state_db_messages_for_session`, so the #6611 + reconciliation authority (recovered display/context pair survives local + and gateway apply) is preserved on the fast path. + + The bounded tail is only trusted when + :func:`_bounded_tail_snapshot_if_safe` proves the skipped state.db prefix + is identical in the sidecar (count + ordered visible identity + no + occurrence collision + compression anchor coverage), and the tail rows + come from the SAME single read transaction as the proof (no TOCTOU); + otherwise the read falls back to the full transcript. + """ from api.models import ( get_state_db_session_messages, reconciled_state_db_messages_for_session, ) - state_messages = get_state_db_session_messages( - getattr(session, "session_id", None), - profile=getattr(session, "profile", None), - ) + bounded_tail = None + if use_sidecar: + read_floor = _sidecar_regeneration_read_floor(session) + if read_floor is not None: + bounded_tail = _bounded_tail_snapshot_if_safe(session, read_floor) + if bounded_tail is not None: + state_messages = bounded_tail + else: + state_messages = get_state_db_session_messages( + getattr(session, "session_id", None), + profile=getattr(session, "profile", None), + ) return ( reconciled_state_db_messages_for_session( session, @@ -242,7 +373,7 @@ def regeneration_state(session): def regeneration_revision(session) -> str: - rows, context = regeneration_state(session) + rows, context = regeneration_state(session, use_sidecar=True) return regeneration_revision_for( rows, session=session, diff --git a/static/ui.js b/static/ui.js index 56af87daeaf..e54a3ba38a5 100644 --- a/static/ui.js +++ b/static/ui.js @@ -13503,6 +13503,117 @@ function _updateLiveAnchorReasoningRowForFallback(turn, text, opts){ if(typeof scrollIfPinned==='function') scrollIfPinned(); return true; } +// -- Live activity-scene repaint memo (streaming frame budget) -------------- +// renderLiveAnchorActivityScene repaints the live worklog by tearing the whole +// row list down (list.innerHTML='' / node.remove()) and rebuilding every row, +// bracketed by _captureMessageScrollSnapshot() before and +// _restoreMessageScrollSnapshotSameFrame() + _restoreLiveAnchorScrollSnapshot- +// AfterRebuild() after. Both brackets READ layout (scrollHeight / clientHeight / +// getBoundingClientRect) immediately around those DOM writes, so every call +// forces a synchronous layout of the ENTIRE transcript -- cost that scales with +// the whole session, not with what changed. +// +// During a live turn the same scene is REQUESTED far more often than it +// changes. One _doRender asks twice (_renderLiveThinking -> updateThinking -> +// appendThinking -> here, then _upsertAnchorProcessProse -> _renderAnchorLive- +// Scene -> here), and every `reasoning` SSE event asks again. Measured on a +// 2000-message session with a live streaming turn: 26.4 repaints/s against a +// 15 fps render throttle, 49% of wall-clock inside this function, of which +// 39% was the scroll capture/restore reflow pairs alone -- and roughly half of +// those repaints produced byte-identical DOM. +// +// The repaint is a pure function of (mode, ids, rendering rows, _showThinking, +// worklog-open default, turn start). When those inputs match the scene we last +// PAINTED, the rebuild is provably a no-op, so skip it and return the same +// result the caller got last time (callers such as _upsertAnchorReasoning use +// the boolean to decide whether to fall back to the legacy thinking card, so +// the return value must stay truthful). +// +// Fail closed: the skip is only taken when the exact turn element and row +// container we painted are still connected, still owned by this stream, and +// still hold the same number of children. Anything that touches the live DOM +// behind our back (removeThinking(), _dedupeLiveProcessedWorklogAnchors(), a +// renderMessages() rebuild, a session switch, a snapshot restore) invalidates +// the memo and the full repaint runs. +let _liveAnchorSceneRepaintMemo=null; +function _liveAnchorSceneRepaintKey(sceneMode, streamId, opts){ + return [ + String(sceneMode||''), + String(streamId||''), + String((opts&&opts.sessionId)||''), + String((typeof S!=='undefined'&&S.session&&S.session.session_id)||''), + String((typeof S!=='undefined'&&S.activeStreamId)||''), + (typeof window!=='undefined'&&window._showThinking===false)?'0':'1', + (typeof _worklogDetailsExpandedDefault==='function'&&_worklogDetailsExpandedDefault())?'1':'0', + String((typeof S!=='undefined'&&S.session&&S.session.pending_started_at)||''), + ].join('|'); +} +// Structural compare rather than a hand-picked field list: the renderers read a +// wide, mode-dependent slice of each row (text, tool.*, thinking.*, payload.*, +// group.*, timestamps), and an omitted field would silently freeze a live row. +// Row text/payload strings are shared by reference with the anchor registry, so +// the common case short-circuits on `===`. Depth-bounded and fail-closed: an +// unexpectedly deep or non-plain value forces the repaint. +function _liveAnchorSceneValueUnchanged(a, b, depth){ + if(a===b) return true; + if(depth>8) return false; + if(a===null||b===null||a===undefined||b===undefined) return false; + const type=typeof a; + if(type!==typeof b) return false; + if(type!=='object') return type==='number'&&Number.isNaN(a)&&Number.isNaN(b); + const aIsArray=Array.isArray(a); + if(aIsArray!==Array.isArray(b)) return false; + if(aIsArray){ + if(a.length!==b.length) return false; + for(let i=0;i the teardown/rebuild below would + // reproduce the same DOM at the cost of two full-transcript reflows. Skip it. + const repaintKey=(typeof _liveAnchorSceneRepaintKey==='function') + ? _liveAnchorSceneRepaintKey(sceneMode,streamId,opts) : ''; + const repaintSkip=(typeof _liveAnchorSceneRepaintSkip==='function') + ? _liveAnchorSceneRepaintSkip(repaintKey,rows) : undefined; + if(repaintSkip!==undefined) return repaintSkip; $('emptyState').style.display='none'; let turn=$('liveAssistantTurn'); if(!turn){ @@ -13579,7 +13697,10 @@ function renderLiveAnchorActivityScene(streamId, scene, opts){ _restoreMessageScrollSnapshotSameFrame(scrollSnapshot); _restoreLiveAnchorScrollSnapshotAfterRebuild(scrollSnapshot,scrollRebuildGuard); if(!scrollRebuildGuard.readerAwayFromBottom&&typeof scrollIfPinned==='function') scrollIfPinned(); - return true; + // Memo the element the rows actually live in (the worklog list), so a row + // removed behind our back changes childElementCount and invalidates the skip. + if(typeof _liveAnchorSceneRepaintRemember!=='function') return true; + return _liveAnchorSceneRepaintRemember(repaintKey,rows,streamId,turn,(group&&_toolWorklogListEl(group))||group,true); } function _renderLiveAnchorActivitySceneTransparent(streamId, scene, opts){ opts=opts||{}; @@ -13588,6 +13709,13 @@ function _renderLiveAnchorActivitySceneTransparent(streamId, scene, opts){ if(streamId&&S.activeStreamId!==streamId) return false; const rows=_anchorSceneRowsForRendering(scene,{settled:false}); if(!rows.length) return false; + // Same repaint memo as the compact path -- both modes rebuild from the same + // rows and pay the same capture/restore reflow pair, so both need the guard. + const repaintKey=(typeof _liveAnchorSceneRepaintKey==='function') + ? _liveAnchorSceneRepaintKey('transparent_stream',streamId,opts) : ''; + const repaintSkip=(typeof _liveAnchorSceneRepaintSkip==='function') + ? _liveAnchorSceneRepaintSkip(repaintKey,rows) : undefined; + if(repaintSkip!==undefined) return repaintSkip; $('emptyState').style.display='none'; let turn=$('liveAssistantTurn'); if(!turn){ @@ -13685,7 +13813,8 @@ function _renderLiveAnchorActivitySceneTransparent(streamId, scene, opts){ _restoreMessageScrollSnapshotSameFrame(scrollSnapshot); _restoreLiveAnchorScrollSnapshotAfterRebuild(scrollSnapshot,scrollRebuildGuard); if(!scrollRebuildGuard.readerAwayFromBottom&&typeof scrollIfPinned==='function') scrollIfPinned(); - return !!renderedRows.length; + if(typeof _liveAnchorSceneRepaintRemember!=='function') return !!renderedRows.length; + return _liveAnchorSceneRepaintRemember(repaintKey,rows,streamId,turn,blocks,!!renderedRows.length); } function _transparentLiveRowKey(node, streamId){ diff --git a/tests/test_issue6611_regeneration_authority.py b/tests/test_issue6611_regeneration_authority.py index a519f8c78d1..2a42f6ae387 100644 --- a/tests/test_issue6611_regeneration_authority.py +++ b/tests/test_issue6611_regeneration_authority.py @@ -412,25 +412,517 @@ def test_terminal_payload_embeds_the_rows_it_hashes(monkeypatch): def test_recovered_display_context_pair_survives_local_and_gateway_apply(monkeypatch): - session = _session() + """#6611: the recovered display/context pair survives apply on BOTH paths. + + The state.db authority path (empty sidecar, recovery) and the #6826 + sidecar-anchored path (pair already loaded in memory) must both plan the + same canonical pair, and the recovered pair must survive local and + gateway apply on each path. + """ + from api import models as models_api + canonical_rows = [ - {"role": "user", "content": "recovered", "id": "u-recovered", "_source": "webui"}, - {"role": "assistant", "content": "failed"}, + {"role": "user", "content": "recovered", "id": "u-recovered", "_source": "webui", "timestamp": 100.0}, + {"role": "assistant", "content": "failed", "timestamp": 101.0}, ] canonical_context = [ - {"role": "system", "content": "recovered context only"}, + {"role": "system", "content": "recovered context only", "timestamp": 99.0}, *canonical_rows, ] monkeypatch.setattr( - "api.session_ops.regeneration_state", - lambda _session: (canonical_rows, canonical_context), + models_api, + "get_state_db_session_messages", + lambda *_args, **_kwargs: [dict(row) for row in canonical_rows], ) from api.session_ops import apply_regeneration_plan, plan_regeneration + # --- state.db authority path: empty sidecar (recovery), the authority + # reconciles the state.db snapshot into the canonical pair. + session = _session() + session.messages = [] + session.context_messages = [] plan = plan_regeneration(session) + assert plan.canonical_rows == canonical_rows + assert plan.canonical_context == canonical_rows + payload = _session_payload_with_full_messages(session) + assert payload["messages"] == canonical_rows + assert payload["message_count"] == len(canonical_rows) assert apply_regeneration_plan(session, plan) assert session.messages == canonical_rows[:1] assert any(row.get("content") == "recovered" for row in session.context_messages) + + # --- sidecar-anchored path: the recovered pair is already loaded in + # memory; the bounded state.db tail read must not lose it, and the same + # canonical pair must survive local + gateway apply. + session = _session() + session.messages = [dict(row) for row in canonical_rows] + session.context_messages = [dict(row) for row in canonical_context] + plan = plan_regeneration(session) + assert plan.canonical_rows == canonical_rows + assert plan.canonical_context == canonical_context payload = _session_payload_with_full_messages(session) assert payload["messages"] == canonical_rows assert payload["message_count"] == len(canonical_rows) + assert apply_regeneration_plan(session, plan) + assert session.messages == canonical_rows[:1] + assert any(row.get("content") == "recovered" for row in session.context_messages) + + +def test_regeneration_sidecar_path_avoids_full_transcript_refetch(monkeypatch): + """#6826: regenerate reads a bounded state.db tail, not the full transcript. + + The full authority read materializes the entire state.db transcript (the + >1min stall on large sessions). The sidecar-anchored regeneration path + must pass ``since_timestamp`` so the SQL scan is bounded, while still + reconciling through ``reconciled_state_db_messages_for_session`` — a + state.db-only gateway tail row must be picked up identically on both + paths (authority preserved). + """ + from api import models as models_api + from api.session_ops import regeneration_state + + session = _session() + session.messages = [ + {"role": "user", "content": f"m{index}", "timestamp": float(index)} + for index in range(500) + ] + session.context_messages = [dict(row) for row in session.messages] + # state.db has the same transcript plus a gateway-applied tail the sidecar + # has not seen yet. + state_rows = [dict(row) for row in session.messages] + state_rows.append( + {"role": "assistant", "content": "gateway applied turn", "timestamp": 500.0} + ) + + calls = [] + + def fake_get_state_db_session_messages(*_args, **_kwargs): + calls.append(dict(_kwargs)) + since = _kwargs.get("since_timestamp") + if since is None: + return [dict(row) for row in state_rows] + return [ + dict(row) + for row in state_rows + if (row.get("timestamp") or 0) >= since + ] + + monkeypatch.setattr( + models_api, + "get_state_db_session_messages", + fake_get_state_db_session_messages, + ) + + # The bounded guard must prove the skipped prefix (rows < floor) is + # identical in the sidecar and return the tail from ONE snapshot. The fake + # state.db carries the same 300 prefix rows as the sidecar. + from api.models import _session_message_visible_key + + floor = min(row["timestamp"] for row in session.messages[-200:]) + prefix_rows = [r for r in session.messages if (r.get("timestamp") or 0) < floor] + tail_rows = [r for r in state_rows if (r.get("timestamp") or 0) >= floor] + monkeypatch.setattr( + models_api, + "get_state_db_regeneration_tail_snapshot", + lambda *_a, **_k: { + "prefix": {"count": len(prefix_rows), "null_timestamp_count": 0}, + "prefix_keys": [_session_message_visible_key(r) for r in prefix_rows], + "tail": [dict(r) for r in tail_rows], + "tail_keys": [_session_message_visible_key(r) for r in tail_rows], + }, + ) + + full_rows, full_context = regeneration_state(session) + assert calls[-1].get("since_timestamp") is None # full read + + # provably effective: the fast path does NOT consult the full reader — the + # bounded tail comes from the single snapshot + calls_before = len(calls) + fast_rows, fast_context = regeneration_state(session, use_sidecar=True) + assert len(calls) == calls_before, "bounded path must not touch the full reader" + + # authority preserved: identical reconciled transcript on both paths, + # including the state.db-only gateway tail row. + assert fast_rows == full_rows + assert fast_context == full_context + assert fast_rows[-1]["content"] == "gateway applied turn" + + +# ── #6826 r3: bounded tail-read guard ──────────────────────────────────────── + +def _sidecar_session(rows=3, *, anchor_key=None, anchor_ts=None): + """Session whose sidecar carries `rows` timestamped display messages.""" + messages = [ + {"role": "user" if i % 2 == 0 else "assistant", "content": f"m{i}", "timestamp": 100.0 + i * 100.0} + for i in range(rows) + ] + s = Session( + session_id="sidecar-guard-6826", + messages=messages, + context_messages=[dict(m) for m in messages], + ) + if anchor_key: + s.compression_anchor_message_key = anchor_key + if anchor_ts is not None: + s._anchor_ts = anchor_ts + return s + + +def _snap(prefix_count=0, prefix_keys=None, tail=None, tail_keys=None): + """Factory for a fake get_state_db_regeneration_tail_snapshot result.""" + tail = tail if tail is not None else [] + return { + "prefix": {"count": prefix_count, "null_timestamp_count": 0}, + "prefix_keys": prefix_keys or [], + "tail": tail, + "tail_keys": tail_keys if tail_keys is not None else [], + } + + +def test_bounded_guard_empty_prefix_allows_tail_read(monkeypatch): + from api import session_ops + + session = _sidecar_session(3) + monkeypatch.setattr( + "api.models.get_state_db_regeneration_tail_snapshot", + lambda *_a, **_k: _snap(tail=[{"role": "user", "content": "t", "timestamp": 300.0}]), + ) + seen = {} + monkeypatch.setattr( + "api.models.get_state_db_session_messages", + lambda *_a, **_k: _capture(seen, _k), + ) + session_ops.regeneration_state(session, use_sidecar=True) + # bounded path: full reader NOT consulted (tail comes from the snapshot) + assert seen == {}, "empty skipped prefix must take the bounded tail from the snapshot" + + +def test_bounded_guard_unrepresented_prefix_row_falls_back(monkeypatch): + from api import session_ops + + session = _sidecar_session(3) + # state.db has one row older than the floor that the sidecar does NOT carry + monkeypatch.setattr( + "api.models.get_state_db_regeneration_tail_snapshot", + lambda *_a, **_k: _snap( + prefix_count=1, + prefix_keys=[("user", "recoverable-only-in-db")], + ), + ) + seen = {} + monkeypatch.setattr( + "api.models.get_state_db_session_messages", + lambda *_a, **_k: _capture(seen, _k), + ) + session_ops.regeneration_state(session, use_sidecar=True) + assert seen != {}, "unrepresented prefix row must force the full read (full reader consulted)" + + +def test_bounded_guard_repeated_tail_key_falls_back(monkeypatch): + """#6826 r3 #1: a bounded-tail key that ALSO occurs in the skipped prefix + must force the full read (occurrence-count collision would drop the + repeated tail row).""" + from api import session_ops + + session = _sidecar_session(3) + repeated_key = ("user", "m0") # m0 lives below the floor AND repeats in the tail + monkeypatch.setattr( + "api.models.get_state_db_regeneration_tail_snapshot", + lambda *_a, **_k: _snap( + prefix_count=1, + prefix_keys=[repeated_key], + tail=[{"role": "user", "content": "m0", "timestamp": 300.0}], + tail_keys=[repeated_key], + ), + ) + seen = {} + monkeypatch.setattr( + "api.models.get_state_db_session_messages", + lambda *_a, **_k: _capture(seen, _k), + ) + session_ops.regeneration_state(session, use_sidecar=True) + assert seen != {}, "repeated tail key must force the full read" + + +def test_bounded_guard_compression_anchor_below_floor_falls_back(monkeypatch): + from api import session_ops + + session = _sidecar_session(3) + # anchor timestamp predates the floor → bounded read cannot cover it + session.compression_anchor_message_key = {"role": "user", "text": "m0", "ts": 100.0} + monkeypatch.setattr( + "api.models.get_state_db_regeneration_tail_snapshot", + lambda *_a, **_k: _snap(), + ) + # shrink the anchor budget window so the floor (300.0) sits above the anchor + monkeypatch.setattr(session_ops, "_REGENERATION_SIDECAR_ANCHOR_BUDGET", 1) + seen = {} + monkeypatch.setattr( + "api.models.get_state_db_session_messages", + lambda *_a, **_k: _capture(seen, _k), + ) + session_ops.regeneration_state(session, use_sidecar=True) + assert seen != {}, "compression anchor below floor must force the full read" + + +def test_bounded_guard_compression_anchor_covered_allows(monkeypatch): + from api import session_ops + + session = _sidecar_session(3) + session.compression_anchor_message_key = {"role": "user", "text": "m0", "ts": 100.0} + # budget 200 covers all 3 rows → floor = min(all) = 100 → anchor at floor is covered + monkeypatch.setattr( + "api.models.get_state_db_regeneration_tail_snapshot", + lambda *_a, **_k: _snap(tail=[{"role": "user", "content": "t", "timestamp": 300.0}]), + ) + seen = {} + monkeypatch.setattr( + "api.models.get_state_db_session_messages", + lambda *_a, **_k: _capture(seen, _k), + ) + session_ops.regeneration_state(session, use_sidecar=True) + assert seen == {}, "anchor covered by floor must keep the bounded tail read" + + +def test_bounded_guard_missing_snapshot_falls_back(monkeypatch): + """#6826 r3 #2 (TOCTOU): if a stable single-connection snapshot cannot be + obtained, the guard must refuse the bounded path entirely.""" + from api import session_ops + + session = _sidecar_session(3) + monkeypatch.setattr( + "api.models.get_state_db_regeneration_tail_snapshot", + lambda *_a, **_k: None, + ) + seen = {} + monkeypatch.setattr( + "api.models.get_state_db_session_messages", + lambda *_a, **_k: _capture(seen, _k), + ) + session_ops.regeneration_state(session, use_sidecar=True) + assert seen != {}, "missing snapshot must force the full read (no TOCTOU window)" + + +def _capture(seen, kwargs): + seen.update(kwargs) + return [] + + +def test_regeneration_tail_snapshot_reads_real_sqlite(monkeypatch, tmp_path): + """Real-SQLite: the single-connection snapshot returns prefix proof + tail + from one read, including the repeated-prompt collision shape from the + round-3 review (a prompt that lives both below and above the floor).""" + import sqlite3 + + from api import models + + db = tmp_path / "state.db" + conn = sqlite3.connect(db) + conn.execute( + "CREATE TABLE messages (id INTEGER PRIMARY KEY, session_id TEXT, role TEXT, " + "content TEXT, timestamp REAL, tool_calls TEXT, active INTEGER)" + ) + rows = [ + ("s1", "user", "prompt", 100.0), + ("s1", "assistant", "a1", 200.0), + ("s1", "user", "prompt", 300.0), # repeated prompt above the floor + ("s1", "assistant", "a2", 400.0), + ] + for sid, role, content, ts in rows: + conn.execute( + "INSERT INTO messages (session_id, role, content, timestamp, active) VALUES (?,?,?,?,1)", + (sid, role, content, ts), + ) + conn.commit() + conn.close() + + monkeypatch.setattr(models, "_active_state_db_path", lambda: db) + snap = models.get_state_db_regeneration_tail_snapshot("s1", 250.0) + assert snap is not None + assert snap["prefix"]["count"] == 2, "rows < floor must be counted" + assert len(snap["prefix_keys"]) == 2 + assert len(snap["tail"]) == 2, "rows >= floor must be the bounded tail" + # the repeated prompt key appears in BOTH the skipped prefix and the tail — + # exactly the occurrence-count collision the guard must refuse + assert snap["prefix_keys"][0] == snap["tail_keys"][0] + + +def _make_state_db(tmp_path, rows, with_id=True): + """Create a real state.db with the given rows (list of dicts).""" + import sqlite3 + + db = tmp_path / "state.db" + conn = sqlite3.connect(db) + cols = ["session_id TEXT", "role TEXT", "content TEXT", "timestamp REAL", + "tool_calls TEXT", "tool_name TEXT", "reasoning TEXT", "active INTEGER"] + if with_id: + cols.append("id INTEGER PRIMARY KEY") + conn.execute(f"CREATE TABLE messages ({', '.join(cols)})") + for i, r in enumerate(rows): + conn.execute( + "INSERT INTO messages (session_id, role, content, timestamp, tool_calls, tool_name, reasoning, active) " + "VALUES (?,?,?,?,?,?,?,1)", + ("s1", r.get("role"), r.get("content"), r.get("timestamp"), + r.get("tool_calls"), r.get("tool_name"), r.get("reasoning")), + ) + conn.commit() + conn.close() + return db + + +def test_snapshot_tail_uses_canonical_projection(monkeypatch, tmp_path): + """#6826 r4 #1: the snapshot tail must use the SAME projection as the + canonical reader — JSON tool_calls decoded, tool_name → name, no raw + id/active exposure (tool-call/gateway-row sessions).""" + from api import models + + db = _make_state_db(tmp_path, [ + {"role": "user", "content": "p", "timestamp": 100.0}, + {"role": "tool", "content": "result", "timestamp": 200.0, + "tool_calls": '[{"name": "x", "arguments": {"a": 1}}]', + "tool_name": "read_file", "reasoning": "r1"}, + {"role": "assistant", "content": "a", "timestamp": 300.0}, + ]) + monkeypatch.setattr(models, "_active_state_db_path", lambda: db) + snap = models.get_state_db_regeneration_tail_snapshot("s1", 50.0) + assert snap is not None + tool = [m for m in snap["tail"] if m.get("role") == "tool"][0] + assert isinstance(tool["tool_calls"], list), "tool_calls must be JSON-decoded" + assert tool["name"] == "read_file", "tool_name → name must be applied" + assert "id" not in tool and "active" not in tool, "raw id/active must not leak" + assert "reasoning" in tool and tool["reasoning"] == "r1" + + +def test_snapshot_uses_durable_id_order(monkeypatch, tmp_path): + """#6826 r4 #2: durable id ASC order (not timestamp) — a later-id user at + ts=600 followed by its assistant at ts=550 must stay user → assistant.""" + from api import models + + db = _make_state_db(tmp_path, [ + {"role": "user", "content": "later user", "timestamp": 600.0}, + {"role": "assistant", "content": "assistant 550", "timestamp": 550.0}, + ]) + monkeypatch.setattr(models, "_active_state_db_path", lambda: db) + snap = models.get_state_db_regeneration_tail_snapshot("s1", 50.0) + assert snap is not None + roles = [m.get("role") for m in snap["tail"]] + assert roles == ["user", "assistant"], f"durable id order violated: {roles}" + + +class _FakeCursor: + def __init__(self, conn): + self._c = conn._real.cursor() + self._conn = conn + self._dv = None + + def execute(self, sql, *args): + s = sql.strip() + if s.startswith("PRAGMA data_version"): + self._conn._dv_calls += 1 + self._dv = 42 if self._conn._dv_calls == 1 else 43 # changed on 2nd read + return self + return self._c.execute(sql, *args) + + def fetchone(self): + if self._dv is not None: + v, self._dv = self._dv, None + return (v,) + return self._c.fetchone() + + def fetchall(self): + return self._c.fetchall() + + +class _FakeConn: + def __init__(self, real): + self._real = real + self._dv_calls = 0 + + @property + def row_factory(self): + return self._real.row_factory + + @row_factory.setter + def row_factory(self, value): + # the real cursor must see the row_factory, or the interleaving path + # dies on tuple indexing instead of exercising data_version + self._real.row_factory = value + + def cursor(self): + return _FakeCursor(self) + + def close(self): + self._real.close() + + def __enter__(self): + return self + + def __exit__(self, *a): + self.close() + + +def test_snapshot_refuses_when_wal_data_version_changes(monkeypatch, tmp_path): + """#6826 r4 #3 (WAL TOCTOU): if PRAGMA data_version changes during the + proof (a concurrent WAL commit), the snapshot must return None so the + caller falls back to the full read.""" + import sqlite3 + + from api import models + + db = _make_state_db(tmp_path, [ + {"role": "user", "content": "p", "timestamp": 100.0}, + {"role": "assistant", "content": "a", "timestamp": 200.0}, + ]) + real_conn = sqlite3.connect(db) + fake = _FakeConn(real_conn) + monkeypatch.setattr(models, "open_state_db_readonly", lambda _p: fake) + snap = models.get_state_db_regeneration_tail_snapshot("s1", 50.0) + assert snap is None, "changed data_version must refuse the bounded snapshot" + + +def test_in_tail_duplicate_guard_refuses_bounded_and_full_revision_accepted(monkeypatch, tmp_path): + """#6826 r5: a repeated message wholly INSIDE the bounded tail must refuse + the fast path (full-read fallback), so the minted full-read revision is + accepted by plan_regeneration (no 409 stale_regeneration_revision).""" + from api import models + from api.session_ops import ( + plan_regeneration, + regeneration_revision_for, + regeneration_state, + ) + + # 212 rows (above the 200-row anchor budget); the user row at index 210 + # repeats the visible key of the user row at index 100 — a duplicate that + # lives entirely within the tail. The final assistant row (211) answers + # the current turn so the session is regenerable. + rows = [] + for i in range(212): + if i == 211: + rows.append({"role": "assistant", "content": "final answer", "timestamp": 100.0 + i}) + continue + role = "user" if i % 2 == 0 else "assistant" + content = "m100" if i == 210 else f"m{i}" + rows.append({"role": role, "content": content, "timestamp": 100.0 + i}) + db = _make_state_db(tmp_path, rows) + monkeypatch.setattr(models, "_active_state_db_path", lambda: db) + + session = Session( + session_id="s1", + messages=[dict(r) for r in rows], + context_messages=[dict(r) for r in rows], + ) + # guard must refuse the bounded path (in-tail duplicate) + floor = min(m["timestamp"] for m in session.messages[-200:]) + snap = models.get_state_db_regeneration_tail_snapshot("s1", floor) + assert snap is not None + assert len(snap["tail_keys"]) > len(set(snap["tail_keys"])), "fixture must repeat a key inside the tail" + from api import session_ops + assert session_ops._bounded_tail_snapshot_if_safe(session, floor) is None, \ + "in-tail duplicate must refuse the bounded path" + + # mint → validate: the full-read revision must be accepted + full_rows, full_context = regeneration_state(session) + full_rev = regeneration_revision_for(full_rows, session=session, context=full_context) + assert full_rev + plan = plan_regeneration(session, expected_revision=full_rev, lock_held=True) + assert plan is not None and plan.revision == full_rev diff --git a/tests/test_live_anchor_scene_repaint_memo.py b/tests/test_live_anchor_scene_repaint_memo.py new file mode 100644 index 00000000000..fe070769f80 --- /dev/null +++ b/tests/test_live_anchor_scene_repaint_memo.py @@ -0,0 +1,427 @@ +"""Streaming frame budget: the live activity scene must not repaint unchanged. + +`renderLiveAnchorActivityScene` (and its transparent_stream sibling) repaint the +live worklog by tearing the whole row list down and rebuilding every row, +bracketed by `_captureMessageScrollSnapshot()` before and +`_restoreMessageScrollSnapshotSameFrame()` after. Both brackets read layout +(`scrollHeight`/`clientHeight`/`getBoundingClientRect`) right around those DOM +writes, so each repaint forces a synchronous layout of the ENTIRE transcript — +cost proportional to the whole session, not to what changed. + +During a live turn the scene is *requested* far more often than it changes: one +`_doRender` asks twice (`_renderLiveThinking` -> `appendThinking`, then +`_upsertAnchorProcessProse` -> `_renderAnchorLiveScene`) and every `reasoning` +SSE event asks again. Measured on a 2000-message session with a live streaming +turn, that was 26.4 repaints/s against a 15 fps render throttle and ~49% of +wall-clock, with roughly half of the repaints producing byte-identical DOM. + +These tests drive the REAL production renderers over a mock DOM and count the +expensive work. They fail on the pre-fix code (every request repaints) and pass +only when an unchanged scene is skipped — while still repainting the moment the +scene, the mode, the stream, or the painted DOM changes (fail-closed). +""" + +from __future__ import annotations + +import json +import pathlib +import shutil +import subprocess +import tempfile + +import pytest + +ROOT = pathlib.Path(__file__).parent.parent +UI_JS_PATH = ROOT / "static" / "ui.js" +NODE = shutil.which("node") + +pytestmark = pytest.mark.skipif(NODE is None, reason="node not on PATH") + +# Helpers introduced by the fix. They are extracted when present; when absent +# (pre-fix tree) the harness substitutes inert stubs so the production renderer +# still runs and the test fails on the repaint COUNT rather than on an import +# error. +MEMO_HELPERS = ( + "_liveAnchorSceneRepaintKey", + "_liveAnchorSceneValueUnchanged", + "_liveAnchorSceneRepaintSkip", + "_liveAnchorSceneRepaintRemember", +) +RENDERERS = ( + "renderLiveAnchorActivityScene", + "_renderLiveAnchorActivitySceneTransparent", +) + + +def _run_node(source: str) -> dict: + with tempfile.NamedTemporaryFile( + "w", suffix=".cjs", encoding="utf-8", dir=ROOT, delete=False + ) as script: + script.write(source) + script_path = pathlib.Path(script.name) + try: + result = subprocess.run( + [NODE, str(script_path)], + cwd=str(ROOT), + capture_output=True, + text=True, + timeout=60, + ) + finally: + script_path.unlink(missing_ok=True) + if result.returncode != 0: + raise RuntimeError(result.stderr) + return json.loads(result.stdout.strip()) + + +def _extract_js_function(js: str, name: str) -> str | None: + """Extract a top-level production function body without eval'ing the file.""" + marker = f"function {name}(" + start = js.find(marker) + if start < 0: + return None + depth = 0 + quote = None + escaped = False + line_comment = False + block_comment = False + index = js.index("{", start) + while index < len(js): + char = js[index] + next_char = js[index + 1] if index + 1 < len(js) else "" + if line_comment: + if char == "\n": + line_comment = False + index += 1 + continue + if block_comment: + if char == "*" and next_char == "/": + block_comment = False + index += 2 + continue + index += 1 + continue + if quote: + if escaped: + escaped = False + elif char == "\\": + escaped = True + elif char == quote: + quote = None + index += 1 + continue + if char == "/" and next_char == "/": + line_comment = True + index += 2 + continue + if char == "/" and next_char == "*": + block_comment = True + index += 2 + continue + if char in "'\"`": + quote = char + index += 1 + continue + if char == "{": + depth += 1 + elif char == "}": + depth -= 1 + if depth == 0: + return js[start : index + 1] + index += 1 + raise AssertionError(f"JavaScript function {name} did not close") + + +_STUBS = { + "_liveAnchorSceneRepaintKey": "function _liveAnchorSceneRepaintKey(){return '';}", + "_liveAnchorSceneValueUnchanged": "function _liveAnchorSceneValueUnchanged(){return false;}", + "_liveAnchorSceneRepaintSkip": "function _liveAnchorSceneRepaintSkip(){return undefined;}", + "_liveAnchorSceneRepaintRemember": ( + "function _liveAnchorSceneRepaintRemember(k,r,s,t,c,result){return result;}" + ), +} + + +def _production() -> str: + js = UI_JS_PATH.read_text(encoding="utf-8") + parts = [] + for name in MEMO_HELPERS: + extracted = _extract_js_function(js, name) + parts.append(extracted if extracted else _STUBS[name]) + for name in RENDERERS: + extracted = _extract_js_function(js, name) + assert extracted, f"{name} missing from static/ui.js" + parts.append(extracted) + return "\n".join(parts) + + +_HARNESS = r""" +let _liveAnchorSceneRepaintMemo=null; + +// ---- minimal DOM ---------------------------------------------------------- +let nodeSeq=0; +function el(tag){ + const node={ + tag, id:'', _id:++nodeSeq, isConnected:true, children:[], parent:null, + attrs:Object.create(null), dataset:{}, style:{}, hidden:false, + classList:{add(){},remove(){},toggle(){},contains(){return false;}}, + setAttribute(k,v){this.attrs[k]=String(v);}, + getAttribute(k){return Object.prototype.hasOwnProperty.call(this.attrs,k)?this.attrs[k]:null;}, + removeAttribute(k){delete this.attrs[k];}, + appendChild(child){child.parent=this;this.children.push(child);return child;}, + removeChild(child){const i=this.children.indexOf(child);if(i>=0)this.children.splice(i,1);child.isConnected=false;}, + remove(){if(this.parent)this.parent.removeChild(this);this.isConnected=false;}, + querySelector(){return null;}, + querySelectorAll(){return [];}, + closest(){return null;}, + matches(){return false;}, + get childElementCount(){return this.children.length;}, + }; + return node; +} +const messagesEl=el('div'); +const msgInner=el('div'); +const emptyState=el('div'); +const turn=el('div'); +turn.id='liveAssistantTurn'; +const blocks=el('div'); +const group=el('div'); +const worklogList=el('div'); +group.appendChild(worklogList); +msgInner.appendChild(turn); +turn.appendChild(blocks); +function $(id){ + if(id==='messages')return messagesEl; + if(id==='msgInner')return msgInner; + if(id==='emptyState')return emptyState; + if(id==='liveAssistantTurn')return turn; + return null; +} +const window={}; +const document={getElementById:$}; + +// ---- instrumented seams (what the fix is supposed to stop paying for) ----- +const counters={rebuilds:0,captures:0,restores:0,transparentRows:0}; +function _captureMessageScrollSnapshot(){counters.captures++;return {pinned:true,bottom:0,top:0};} +function _restoreMessageScrollSnapshotSameFrame(){counters.restores++;} +function _prepareLiveAnchorScrollRebuildGuard(){return {readerAwayFromBottom:false,release:null};} +function _restoreLiveAnchorScrollSnapshotAfterRebuild(){} +function _renderAnchorSceneRowsIntoWorklog(g,rows){ + counters.rebuilds++; + worklogList.children.length=0; + for(const row of rows) worklogList.appendChild(el('div')); + return true; +} +function _toolWorklogListEl(){return worklogList;} +function _anchorSceneWorklogGroup(){blocks.children.includes(group)||blocks.appendChild(group);return group;} +function _assistantTurnBlocks(){return blocks;} +function _createAssistantTurn(){return turn;} +function _captureWorklogDetailDisclosureState(){return null;} +function _restoreWorklogDetailDisclosureState(){} +function _startActivityElapsedTimer(){} +function _dedupeLiveProcessedWorklogAnchors(){} +function _moveLiveRunStatusToTurnEnd(){} +function _syncToolCallGroupSummary(){} +function _syncTransparentEventControls(){} +function _resetMismatchedLiveAssistantTurnForSession(){return true;} +function _worklogDetailsExpandedDefault(){return false;} +function scrollIfPinned(){} +function isSimplifiedToolCalling(){return true;} +function _anchorSceneRowsForRendering(scene){return (scene&&scene.activity_rows)||[];} +function _transparentLiveRowKey(){return '';} +function _transparentLiveRowsCompatible(){return false;} +function _refreshTransparentLiveRow(existing){return existing;} +function _anchorSceneRowTimestampSeconds(){return null;} +function _anchorSceneTransparentNodeForRow(){counters.transparentRows++;return el('div');} + +let MODE='compact_worklog'; +function chatActivityMode(){return MODE;} + +const S={session:{session_id:'sid-1',pending_started_at:11},activeStreamId:'stream-1'}; + +// A fresh-but-equal projection of the same scene, exactly like the real +// projectAssistantTurnAnchorActivityScene() which rebuilds frozen row objects +// on every call. Reference equality must NOT be what makes the memo hit. +function scene(proseText, toolDone){ + return {activity_rows:[ + {row_id:'r-prose', local_id:'live-prose:stream-1:1', role:'prose', kind:'process_prose', + status:'running', source_event_type:'token', display_hint:'main_prose', + text:proseText, thinking:null, tool:null, + payload:{text:proseText, activitySegmentSeq:1, activityBurstId:0}, + group:{group_key:'segment:1', activity_burst_id:0, activity_segment_seq:1, assistant_msg_idx:null}}, + {row_id:'r-tool', local_id:'live-tool:1', role:'tool', kind:'tool_started', + status: toolDone?'completed':'running', source_event_type:'tool', display_hint:'tool_row', + text:'', thinking:null, + tool:{id:'t1', name:'read_file', args:{path:'a.js'}, preview:'reading a.js', + snippet:'', done:!!toolDone, is_error:false, duration:null, + started_at:5, signature:'read_file|t1|{"path":"a.js"}'}, + payload:{name:'read_file', tid:'t1', args:{path:'a.js'}}, + group:{group_key:'segment:1', activity_burst_id:0, activity_segment_seq:1, assistant_msg_idx:null}}, + ]}; +} +function render(sc){return renderLiveAnchorActivityScene('stream-1',sc,{sessionId:'sid-1'});} +""" + + +def _script(body: str) -> str: + return _production() + _HARNESS + body + + +def test_identical_scene_is_not_repainted_and_still_reports_rendered(): + """Two back-to-back requests for the SAME scene must repaint once, not twice. + + The second call is what `_doRender` issues via `_renderLiveThinking` when no + reasoning arrived; pre-fix it tore the worklog down and rebuilt it, paying a + second full-transcript capture/restore reflow pair for identical DOM. + """ + out = _run_node( + _script( + """ +const first=render(scene('hello world')); +const second=render(scene('hello world')); +const third=render(scene('hello world')); +console.log(JSON.stringify({first,second,third,...counters, + paintedRows:worklogList.childElementCount})); +""" + ) + ) + assert out["first"] is True + # Callers (e.g. _upsertAnchorReasoning) branch on this boolean to decide + # whether to fall back to the legacy thinking card — a skip must not lie. + assert out["second"] is True + assert out["third"] is True + assert out["rebuilds"] == 1 + assert out["captures"] == 1 + assert out["restores"] == 1 + assert out["paintedRows"] == 2 + + +def test_changed_prose_still_repaints(): + """The memo must never freeze a live turn: new streamed text repaints.""" + out = _run_node( + _script( + """ +render(scene('hello')); +render(scene('hello')); +render(scene('hello world')); +render(scene('hello world')); +render(scene('hello world again')); +console.log(JSON.stringify({...counters})); +""" + ) + ) + assert out["rebuilds"] == 3 + assert out["captures"] == 3 + + +def test_tool_completion_repaints(): + """A row field change other than text (tool done/status) must repaint.""" + out = _run_node( + _script( + """ +render(scene('x')); +render(scene('x')); +render(scene('x', true)); +console.log(JSON.stringify({...counters})); +""" + ) + ) + assert out["rebuilds"] == 2 + + +def test_dom_tampered_behind_the_memo_repaints_fail_closed(): + """If anything removes painted rows, an identical scene must repaint.""" + out = _run_node( + _script( + """ +render(scene('x')); +worklogList.children.pop(); // e.g. removeThinking() / a dedupe pass +const second=render(scene('x')); +console.log(JSON.stringify({second,...counters,paintedRows:worklogList.childElementCount})); +""" + ) + ) + assert out["second"] is True + assert out["rebuilds"] == 2 + assert out["paintedRows"] == 2 + + +def test_detached_turn_repaints_fail_closed(): + """A live turn replaced behind our back (reload/session restore) repaints.""" + out = _run_node( + _script( + """ +render(scene('x')); +turn.isConnected=false; // e.g. renderMessages() rebuilt #msgInner +render(scene('x')); +console.log(JSON.stringify({...counters})); +""" + ) + ) + assert out["rebuilds"] == 2 + + +def test_stream_switch_repaints(): + """A different stream id must not inherit the previous stream's paint.""" + out = _run_node( + _script( + """ +render(scene('x')); +S.activeStreamId='stream-2'; +renderLiveAnchorActivityScene('stream-2',scene('x'),{sessionId:'sid-1'}); +console.log(JSON.stringify({...counters})); +""" + ) + ) + assert out["rebuilds"] == 2 + + +def test_session_switch_repaints(): + """A different session must not inherit the previous session's paint.""" + out = _run_node( + _script( + """ +render(scene('x')); +S.session={session_id:'sid-2',pending_started_at:11}; +renderLiveAnchorActivityScene('stream-1',scene('x'),{sessionId:'sid-2'}); +console.log(JSON.stringify({...counters})); +""" + ) + ) + assert out["rebuilds"] == 2 + + +def test_transparent_stream_mode_is_guarded_too(): + """The sibling renderer pays the same reflow pair and needs the same guard.""" + out = _run_node( + _script( + """ +MODE='transparent_stream'; +const first=render(scene('hello')); +const second=render(scene('hello')); +const third=render(scene('hello there')); +console.log(JSON.stringify({first,second,third,...counters})); +""" + ) + ) + assert out["first"] is True + assert out["second"] is True + assert out["third"] is True + # Two paints for three requests: the identical middle request is skipped. + assert out["captures"] == 2 + assert out["restores"] == 2 + + +def test_mode_switch_repaints(): + """Switching activity display mode must repaint, not reuse the other mode.""" + out = _run_node( + _script( + """ +render(scene('x')); +MODE='transparent_stream'; +render(scene('x')); +console.log(JSON.stringify({...counters})); +""" + ) + ) + assert out["captures"] == 2