diff --git a/DEVJOURNAL.md b/DEVJOURNAL.md index 66184f54f6a11..26ea60456855c 100644 --- a/DEVJOURNAL.md +++ b/DEVJOURNAL.md @@ -1,5 +1,38 @@ # Development Journal +## 2026-03-12: Payload Visualization Feature + +### Goal +Visualize what goes into each API payload: system prompt, tool definitions, user messages, assistant messages, tool results. Show actual cached tokens. Display as stacked bar (per-turn detail) and area chart (session growth over turns). + +### Changes + +**`run_agent.py` — instrumentation:** +- `_compute_payload_breakdown(api_kwargs)` — uses tiktoken (`gpt-4o` encoding) to count tokens per component. Handles both Codex Responses and Chat Completions API modes. ~3ms overhead. +- Breakdown computed after `_build_api_kwargs()` + preflight at main loop only (not memory flush or compression side tasks). +- `breakdown` dict merged into JSONL entry: `{"system": N, "tool_defs": N, "user": N, "assistant": N, "tool_results": N}` + +**`dashboard/data.py` — endpoint:** +- `get_payload_breakdown(session_id)` — reads JSONL, filters by session, returns turns with breakdown. Old entries without breakdown gracefully skipped. + +**`dashboard/server.py` — route:** +- `GET /api/payload-breakdown?session_id=` + +**`dashboard/static/index.html` — visualization:** +- New "Payload" tab with session selector +- Canvas area chart: stacked token layers (system blue, tool_defs purple, user green, assistant orange, tool_results red) with dashed cached-tokens overlay line +- Horizontal stacked bars: per-turn proportions, click any turn for detailed breakdown +- Per-turn drill-down: component bars with token counts, percentages, actual vs estimated vs cached +- Version display (v0.2.0) added to topbar + +### Design decisions +- tiktoken for estimated counts based on JSON serialization (more accurate than the rough estimator in `model_metadata.py`, but not exact chat-ml counts) +- Extend existing JSONL rather than new file or DB table (old entries without breakdown handled gracefully) +- Instrument only main agent loop, not side tasks (memory flush, compression) +- `message_extras` extension table planned for reasoning content (avoids upstream schema conflicts on merge) + +--- + ## 2026-03-09: OpenAI Codex Caching Improvements ### Problem diff --git a/dashboard/data.py b/dashboard/data.py index 5726bae4c1f0b..eddd8a7dbaf69 100644 --- a/dashboard/data.py +++ b/dashboard/data.py @@ -222,6 +222,45 @@ def get_usage(days: int = 30, source: Optional[str] = None) -> Dict[str, Any]: return result +def get_payload_breakdown(session_id: str) -> List[Dict[str, Any]]: + """Get per-turn payload breakdown for a session from the JSONL token log. + + Returns entries that have a 'breakdown' field, ordered by api_call number. + Entries without breakdown (older data) are skipped. + """ + if not JSONL_PATH.exists(): + return [] + results = [] + try: + with open(JSONL_PATH, "r") as f: + for line in f: + line = line.strip() + if not line: + continue + try: + entry = json.loads(line) + except json.JSONDecodeError: + continue + if entry.get("session_id") != session_id: + continue + if "breakdown" not in entry: + continue + results.append({ + "api_call": entry.get("api_call", 0), + "prompt_tokens": entry.get("prompt_tokens", 0), + "completion_tokens": entry.get("completion_tokens", 0), + "cached_tokens": entry.get("cached_tokens", 0), + "cache_write_tokens": entry.get("cache_write_tokens", 0), + "breakdown": entry["breakdown"], + "tools": entry.get("tools", []), + "model": entry.get("model", ""), + }) + except Exception: + return [] + results.sort(key=lambda x: x["api_call"]) + return results + + def get_insights(days: int = 30, source: Optional[str] = None) -> Dict[str, Any]: """Generate insights report using InsightsEngine.""" if not DB_PATH.exists(): diff --git a/dashboard/server.py b/dashboard/server.py index 11f74034ff9eb..9c4486cf4feec 100644 --- a/dashboard/server.py +++ b/dashboard/server.py @@ -112,6 +112,13 @@ async def handle_memory(request): return json_response(data.get_memory()) +async def handle_payload_breakdown(request): + session_id = request.query.get("session_id", "") + if not session_id: + return json_response({"error": "session_id required"}, status=400) + return json_response(data.get_payload_breakdown(session_id)) + + async def handle_insights(request): days = int(request.query.get("days", 30)) source = request.query.get("source") @@ -143,6 +150,7 @@ def create_app() -> web.Application: app.router.add_get("/api/activity", handle_activity) app.router.add_get("/api/memory", handle_memory) app.router.add_get("/api/insights", handle_insights) + app.router.add_get("/api/payload-breakdown", handle_payload_breakdown) # Static files app.router.add_static("/static/", STATIC_DIR, name="static") diff --git a/dashboard/static/index.html b/dashboard/static/index.html index 68bfa8292953a..0a16de086d10b 100644 --- a/dashboard/static/index.html +++ b/dashboard/static/index.html @@ -335,6 +335,35 @@ min-width: 2px; } +/* ── Payload viz ── */ +.payload-bar-row { + display: flex; + align-items: center; + gap: 8px; + margin-bottom: 6px; + font-size: 12px; + cursor: pointer; +} +.payload-bar-row:hover { opacity: 0.85; } +.payload-bar-row .bar-label { width: 60px; text-align: right; color: var(--text-dim); flex-shrink: 0; } +.payload-bar-row .bar-track { flex: 1; height: 22px; display: flex; border-radius: 3px; overflow: hidden; } +.payload-bar-row .bar-segment { height: 100%; display: flex; align-items: center; justify-content: center; font-size: 10px; color: #fff; min-width: 0; } +.payload-bar-row .bar-value { width: 70px; font-family: var(--mono); font-size: 11px; text-align: right; flex-shrink: 0; } + +.stacked-detail { + background: var(--bg-card); + border: 1px solid var(--border); + border-radius: var(--radius); + padding: 16px; + margin-top: 12px; +} +.stacked-detail .seg-row { display: flex; align-items: center; gap: 8px; margin-bottom: 6px; font-size: 13px; } +.stacked-detail .seg-color { width: 14px; height: 14px; border-radius: 3px; flex-shrink: 0; } +.stacked-detail .seg-label { width: 100px; color: var(--text-dim); } +.stacked-detail .seg-bar { flex: 1; height: 18px; border-radius: 3px; } +.stacked-detail .seg-value { width: 80px; text-align: right; font-family: var(--mono); font-size: 12px; } +.stacked-detail .seg-pct { width: 50px; text-align: right; font-family: var(--mono); font-size: 11px; color: var(--text-dim); } + /* ── Loading ── */ .loading { color: var(--text-dim); padding: 40px; text-align: center; } .empty { color: var(--text-dim); padding: 40px; text-align: center; font-style: italic; } @@ -351,6 +380,7 @@

Hermes Dashboard

+ v0.2.0
@@ -360,6 +390,7 @@

Hermes Dashboard

+ @@ -486,6 +517,26 @@

Per-Call Detail (JSONL)

+ +
+
+ + +
+ + +
Select a session with payload data to visualize
+
+
@@ -604,6 +655,7 @@

Per-Call Detail (JSONL)

activity: loadActivity, sessions: loadSessions, usage: loadUsage, + payload: loadPayloadTab, cron: loadCron, logs: loadLogs, search: () => {}, @@ -1227,6 +1279,255 @@

Per-Call Detail (JSONL)

: '
No files found
'; } +// ═══════════════════════════════════════════════════════════════════ +// Payload Breakdown +// ═══════════════════════════════════════════════════════════════════ +const PAYLOAD_COLORS = { + system: '#58a6ff', // blue + tool_defs: '#bc8cff', // purple + user: '#3fb950', // green + assistant: '#d29922', // orange + tool_results: '#f85149', // red +}; +const PAYLOAD_LABELS = { + system: 'System', + tool_defs: 'Tool Defs', + user: 'User', + assistant: 'Assistant', + tool_results: 'Tool Results', +}; +const PAYLOAD_KEYS = ['system', 'tool_defs', 'user', 'assistant', 'tool_results']; + +async function loadPayloadTab() { + // Populate session dropdown with recent sessions + const sessions = await api('sessions?limit=50'); + const sel = $('#payload-session'); + const prev = sel.value; + sel.innerHTML = ''; + for (const s of sessions) { + const label = (s.title ? truncate(s.title, 40) + ' — ' : '') + + truncate(s.id, 16) + ' (' + (s.source || '?') + ', ' + fmtTime(s.started_at) + ')'; + sel.innerHTML += ``; + } + if (prev) sel.value = prev; +} + +async function loadPayload() { + const sessionId = $('#payload-session').value; + if (!sessionId) return; + + const data = await api('payload-breakdown?session_id=' + encodeURIComponent(sessionId)); + + if (!data.length) { + $('#payload-area-wrap').style.display = 'none'; + $('#payload-bar-wrap').style.display = 'none'; + $('#payload-empty').style.display = ''; + $('#payload-empty').textContent = 'No payload breakdown data for this session (pre-instrumentation data)'; + return; + } + + $('#payload-empty').style.display = 'none'; + $('#payload-area-wrap').style.display = ''; + $('#payload-bar-wrap').style.display = ''; + + drawPayloadAreaChart(data); + drawPayloadBars(data); +} + +function drawPayloadAreaChart(data) { + const canvas = $('#payload-area-chart'); + const ctx = canvas.getContext('2d'); + const dpr = window.devicePixelRatio || 1; + const W = canvas.clientWidth; + const H = canvas.clientHeight; + canvas.width = W * dpr; + canvas.height = H * dpr; + ctx.scale(dpr, dpr); + + const pad = { top: 20, right: 20, bottom: 40, left: 65 }; + const plotW = W - pad.left - pad.right; + const plotH = H - pad.top - pad.bottom; + + // Compute stacked totals per turn + const turns = data.map(d => { + const b = d.breakdown || {}; + let cumulative = 0; + const stacked = {}; + for (const k of PAYLOAD_KEYS) { + cumulative += b[k] || 0; + stacked[k] = cumulative; + } + return { ...d, stacked, total: cumulative }; + }); + + const maxY = Math.max(...turns.map(t => t.total), 1); + const n = turns.length; + + // Clear + ctx.clearRect(0, 0, W, H); + + // X/Y scale + const xScale = i => pad.left + (n > 1 ? (i / (n - 1)) * plotW : plotW / 2); + const yScale = v => pad.top + plotH - (v / maxY) * plotH; + + // Draw stacked areas (bottom to top) + for (let ki = PAYLOAD_KEYS.length - 1; ki >= 0; ki--) { + const key = PAYLOAD_KEYS[ki]; + ctx.beginPath(); + ctx.moveTo(xScale(0), yScale(0)); + for (let i = 0; i < n; i++) { + ctx.lineTo(xScale(i), yScale(turns[i].stacked[key])); + } + ctx.lineTo(xScale(n - 1), yScale(0)); + ctx.closePath(); + ctx.fillStyle = PAYLOAD_COLORS[key] + '99'; + ctx.fill(); + } + + // Draw cached overlay line + ctx.beginPath(); + ctx.strokeStyle = '#ffffff88'; + ctx.lineWidth = 2; + ctx.setLineDash([5, 3]); + for (let i = 0; i < n; i++) { + const y = yScale(turns[i].cached_tokens || 0); + i === 0 ? ctx.moveTo(xScale(i), y) : ctx.lineTo(xScale(i), y); + } + ctx.stroke(); + ctx.setLineDash([]); + + // Axes + ctx.strokeStyle = '#30363d'; + ctx.lineWidth = 1; + ctx.beginPath(); + ctx.moveTo(pad.left, pad.top); + ctx.lineTo(pad.left, pad.top + plotH); + ctx.lineTo(pad.left + plotW, pad.top + plotH); + ctx.stroke(); + + // Y-axis labels + ctx.fillStyle = '#8b949e'; + ctx.font = '11px -apple-system, sans-serif'; + ctx.textAlign = 'right'; + const yTicks = 5; + for (let i = 0; i <= yTicks; i++) { + const v = (maxY / yTicks) * i; + const y = yScale(v); + ctx.fillText(fmtTokens(Math.round(v)), pad.left - 8, y + 4); + if (i > 0) { + ctx.strokeStyle = '#30363d44'; + ctx.beginPath(); + ctx.moveTo(pad.left, y); + ctx.lineTo(pad.left + plotW, y); + ctx.stroke(); + } + } + + // X-axis labels + ctx.textAlign = 'center'; + ctx.fillStyle = '#8b949e'; + const step = Math.max(1, Math.floor(n / 10)); + for (let i = 0; i < n; i += step) { + ctx.fillText('T' + turns[i].api_call, xScale(i), pad.top + plotH + 20); + } + if (n > 1) { + ctx.fillText('T' + turns[n - 1].api_call, xScale(n - 1), pad.top + plotH + 20); + } + + // Click handler for turn drill-down + canvas.onclick = function(e) { + const rect = canvas.getBoundingClientRect(); + const mx = e.clientX - rect.left; + let closest = 0; + let minDist = Infinity; + for (let i = 0; i < n; i++) { + const d = Math.abs(xScale(i) - mx); + if (d < minDist) { minDist = d; closest = i; } + } + showPayloadDetail(turns[closest]); + }; + + // Legend + const legendEl = $('#payload-legend'); + legendEl.innerHTML = PAYLOAD_KEYS.map(k => + `${PAYLOAD_LABELS[k]}` + ).join('') + 'Cached'; +} + +function drawPayloadBars(data) { + const container = $('#payload-bar-detail'); + const maxTotal = Math.max(...data.map(d => { + const b = d.breakdown || {}; + return PAYLOAD_KEYS.reduce((s, k) => s + (b[k] || 0), 0); + }), 1); + + container.innerHTML = data.map((d, idx) => { + const b = d.breakdown || {}; + const total = PAYLOAD_KEYS.reduce((s, k) => s + (b[k] || 0), 0); + const segments = PAYLOAD_KEYS.map(k => { + const v = b[k] || 0; + const pct = total > 0 ? (v / total * 100) : 0; + const widthPct = total > 0 ? (v / maxTotal * 100) : 0; + return pct > 3 + ? `
${fmtTokens(v)}
` + : (pct > 0 ? `
` : ''); + }).join(''); + const tools = (d.tools || []).join(', '); + return ` +
+
T${d.api_call}
+
${segments}
+
${fmtTokens(total)}
+
`; + }).join(''); + + // Store data for click handler + window.payloadData = data; +} + +function showPayloadDetail(turn) { + const wrap = $('#payload-bar-wrap'); + wrap.style.display = ''; + $('#payload-bar-title').textContent = `— Turn ${turn.api_call}` + + (turn.tools?.length ? ` (${turn.tools.join(', ')})` : ''); + + const b = turn.breakdown || {}; + const total = PAYLOAD_KEYS.reduce((s, k) => s + (b[k] || 0), 0); + const cached = turn.cached_tokens || 0; + const actual = turn.prompt_tokens || 0; + + let html = '
'; + + // Per-component bars + for (const k of PAYLOAD_KEYS) { + const v = b[k] || 0; + const pct = total > 0 ? (v / total * 100) : 0; + html += ` +
+
+
${PAYLOAD_LABELS[k]}
+
+
${fmtNum(v)}
+
${pct.toFixed(1)}%
+
`; + } + + // Summary row + html += ` +
+ Estimated: ${fmtNum(total)} + Actual prompt: ${fmtNum(actual)} + Cached: ${fmtNum(cached)}${actual > 0 ? ` (${(cached/actual*100).toFixed(0)}%)` : ''} +
`; + + html += '
'; + + // Replace the bar detail area or append below it + const existing = wrap.querySelector('.stacked-detail'); + if (existing) existing.remove(); + wrap.insertAdjacentHTML('beforeend', html); +} + // ═══════════════════════════════════════════════════════════════════ // Init // ═══════════════════════════════════════════════════════════════════ diff --git a/run_agent.py b/run_agent.py index 33bc269b9a09d..835bc65f96001 100644 --- a/run_agent.py +++ b/run_agent.py @@ -686,7 +686,9 @@ def __init__( self.session_completion_tokens = 0 self.session_total_tokens = 0 self.session_api_calls = 0 - + self._last_payload_breakdown = {} + self._tiktoken_enc = None + if not self.quiet_mode: if compression_enabled: print(f"📊 Context limit: {self.context_compressor.context_length:,} tokens (compress at {int(compression_threshold*100)}% = {self.context_compressor.threshold_tokens:,})") @@ -2437,6 +2439,93 @@ def _build_api_kwargs(self, api_messages: list) -> dict: return api_kwargs + def _compute_payload_breakdown(self, api_kwargs: dict) -> dict: + """Estimate tokens per payload component using tiktoken. + + Returns a dict like {"system": N, "tool_defs": N, "user": N, + "assistant": N, "tool_results": N} where each value is the + estimated token count for that component. These are approximations + based on JSON serialization, not exact chat-ml token counts. + """ + if self._tiktoken_enc is None: + try: + import tiktoken + except ImportError: + logger.debug("tiktoken not installed; payload breakdown disabled") + return {} + try: + self._tiktoken_enc = tiktoken.encoding_for_model("gpt-4o") + except Exception: + self._tiktoken_enc = tiktoken.get_encoding("cl100k_base") + + enc = self._tiktoken_enc + + def _count(obj): + if obj is None: + return 0 + if isinstance(obj, str): + return len(enc.encode(obj)) + return len(enc.encode(json.dumps(obj, default=str, ensure_ascii=False))) + + breakdown = {} + + if self.api_mode == "codex_responses": + # Codex Responses: instructions, input, tools + breakdown["system"] = _count(api_kwargs.get("instructions")) + breakdown["tool_defs"] = _count(api_kwargs.get("tools")) + user_tokens = 0 + assistant_tokens = 0 + tool_results_tokens = 0 + for item in (api_kwargs.get("input") or []): + if isinstance(item, str): + user_tokens += _count(item) + continue + # Codex input items use "type" for function_call / + # function_call_output, and "role" for regular messages. + item_type = item.get("type", "") if isinstance(item, dict) else getattr(item, "type", "") + if item_type == "function_call": + assistant_tokens += _count(item) + elif item_type == "function_call_output": + tool_results_tokens += _count(item) + else: + role = item.get("role", "") if isinstance(item, dict) else getattr(item, "role", "") + if role == "user": + user_tokens += _count(item) + elif role == "assistant": + assistant_tokens += _count(item) + elif role in ("tool", "function"): + tool_results_tokens += _count(item) + else: + user_tokens += _count(item) + breakdown["user"] = user_tokens + breakdown["assistant"] = assistant_tokens + breakdown["tool_results"] = tool_results_tokens + else: + # Chat Completions: messages (with system first), tools + breakdown["tool_defs"] = _count(api_kwargs.get("tools")) + system_tokens = 0 + user_tokens = 0 + assistant_tokens = 0 + tool_results_tokens = 0 + for msg in (api_kwargs.get("messages") or []): + role = msg.get("role", "") if isinstance(msg, dict) else getattr(msg, "role", "") + if role == "system": + system_tokens += _count(msg) + elif role == "user": + user_tokens += _count(msg) + elif role == "assistant": + assistant_tokens += _count(msg) + elif role == "tool": + tool_results_tokens += _count(msg) + else: + user_tokens += _count(msg) + breakdown["system"] = system_tokens + breakdown["user"] = user_tokens + breakdown["assistant"] = assistant_tokens + breakdown["tool_results"] = tool_results_tokens + + return breakdown + def _build_assistant_message(self, assistant_message, finish_reason: str) -> dict: """Build a normalized assistant message dict from an API response message. @@ -3561,6 +3650,12 @@ def run_conversation( if self.api_mode == "codex_responses": api_kwargs = self._preflight_codex_api_kwargs(api_kwargs, allow_stream=False) + # Compute per-component token breakdown for payload viz + try: + self._last_payload_breakdown = self._compute_payload_breakdown(api_kwargs) + except Exception: + self._last_payload_breakdown = {} + if os.getenv("HERMES_DUMP_REQUESTS", "").strip().lower() in {"1", "true", "yes", "on"}: self._dump_api_request_debug(api_kwargs, reason="preflight") @@ -3860,6 +3955,8 @@ def run_conversation( } if _tool_names: _entry["tools"] = _tool_names + if getattr(self, '_last_payload_breakdown', None): + _entry["breakdown"] = self._last_payload_breakdown _usage_line = _json.dumps(_entry) with open(_hermes_home / "token_usage.jsonl", "a") as _tf: _tf.write(_usage_line + "\n")