diff --git a/.github/workflows/agent-quality-lane.yml b/.github/workflows/agent-quality-lane.yml index 1cb3aa399..7eca71862 100644 --- a/.github/workflows/agent-quality-lane.yml +++ b/.github/workflows/agent-quality-lane.yml @@ -4,6 +4,9 @@ on: paths: - "she/metrics/**" - "tests/test_she_agent_throughput.py" + - "tests/test_emit_agent_throughput.py" + - "scripts/ci/emit_agent_throughput.py" + - ".github/workflows/agent-throughput-evidence.yml" - "tests/test_tdqs.py" - "docs/ops/AGENT-THROUGHPUT-*.md" - "docs/ops/AGENT-OBSERVABILITY-PRIORITY-DECISION.md" @@ -57,7 +60,7 @@ jobs: ':(glob)**/*.sh' - name: Compile measurement Python sources - run: python -m compileall -q she/metrics scripts/ci tests/test_she_agent_throughput.py tests/test_tdqs.py + run: python -m compileall -q she/metrics scripts/ci tests/test_she_agent_throughput.py tests/test_emit_agent_throughput.py tests/test_tdqs.py - name: Validate event schema JSON run: python -m json.tool docs/ops/AGENT-THROUGHPUT-EVENT.schema.json >/dev/null @@ -71,6 +74,7 @@ jobs: - name: Run focused ATES reducer tests with coverage run: | coverage run --branch --source=she/metrics -m unittest tests.test_she_agent_throughput -v + python -m unittest tests.test_emit_agent_throughput -v coverage run --branch --source=she/metrics -a -m unittest tests.test_tdqs -v coverage xml -o coverage.xml diff --git a/.github/workflows/agent-throughput-evidence.yml b/.github/workflows/agent-throughput-evidence.yml new file mode 100644 index 000000000..87b7e84a4 --- /dev/null +++ b/.github/workflows/agent-throughput-evidence.yml @@ -0,0 +1,222 @@ +name: Agent Throughput Evidence + +# Phase B ATES evidence producer. +# This workflow is an observer: it reads completed runs/jobs and PR metadata, +# emits sanitized JSONL, and publishes an immutable receipt. It never checks out +# or executes the triggering workflow's SHA and never writes source/repository +# state. workflow_run intentionally runs from the default branch; keep this file +# on master before expecting live completion events. +on: + workflow_run: + workflows: + - "🔀 Gemini Dispatch (free-tier agentic)" + - "Jules on Issues (label / @jules)" + - "Agent review → auto Jules" + - "DeepSeek CI – Agentic Automation" + types: [completed] + +permissions: + actions: read + contents: read + pull-requests: read + +concurrency: + group: agent-throughput-evidence-${{ github.event.workflow_run.id || github.run_id }}-${{ github.event.workflow_run.run_attempt || github.run_attempt }} + cancel-in-progress: false + +jobs: + emit: + runs-on: ubuntu-latest + timeout-minutes: 10 + steps: + - name: Checkout observer implementation + uses: actions/checkout@fbc6f3992d24b796d5a048ff273f7fcc4a7b6c09 # v5.1.0 + with: + fetch-depth: 1 + persist-credentials: false + + - name: Collect trusted workflow metadata + uses: actions/github-script@f28e40c7f34bde8b3046d885e986cb6290c5673b # v7.1.0 + env: + METADATA_PATH: ${{ runner.temp }}/agent-throughput-metadata.json + with: + script: | + const fs = require('fs'); + const run = context.payload.workflow_run; + if (!run) { + core.setFailed('workflow_dispatch is supported only as a validation surface; no target run was supplied'); + return; + } + + const { owner, repo } = context.repo; + const jobs = await github.paginate( + github.rest.actions.listJobsForWorkflowRun, + { owner, repo, run_id: run.id, per_page: 100 } + ); + + let pulls = Array.isArray(run.pull_requests) ? run.pull_requests : []; + if (!pulls.length) { + try { + const response = await github.rest.repos.listPullRequestsAssociatedWithCommit({ + owner, + repo, + commit_sha: run.head_sha, + per_page: 20 + }); + pulls = response.data || []; + } catch (error) { + core.warning(`PR association lookup unavailable: ${error.message}`); + } + } + + let taskComplexity = null; + let pullRequest = null; + if (pulls.length === 1) { + pullRequest = pulls[0]; + try { + const { data: pr } = await github.rest.pulls.get({ + owner, + repo, + pull_number: pullRequest.number + }); + taskComplexity = { + additions: pr.additions, + deletions: pr.deletions, + files_changed: pr.changed_files + }; + } catch (error) { + core.warning(`PR structural evidence unavailable: ${error.message}`); + } + } else if (pulls.length > 1) { + core.notice(`Multiple PR associations (${pulls.length}); structural complexity remains missing.`); + } + + const workflowName = String(run.name || ''); + const identity = [ + ['Gemini Dispatch', 'gemini'], + ['Jules on Issues', 'jules'], + ['Agent review → auto Jules', 'jules'], + ['DeepSeek CI', 'deepseek'] + ].find(([needle]) => workflowName.includes(needle)); + + if (!identity) { + core.setFailed(`workflow is outside the explicit agent identity map: ${workflowName}`); + return; + } + + const metadata = { + agent_id: identity[1], + workflow_run: { + id: run.id, + run_attempt: run.run_attempt || 1, + head_sha: run.head_sha, + created_at: run.created_at, + conclusion: run.conclusion, + name: run.name, + event: run.event, + html_url: run.html_url + }, + jobs: jobs.map(job => ({ + id: job.id, + name: job.name, + started_at: job.started_at, + completed_at: job.completed_at, + conclusion: job.conclusion + })), + task_complexity: taskComplexity, + pull_request_number: pullRequest?.number || null + }; + + fs.writeFileSync(process.env.METADATA_PATH, JSON.stringify(metadata, null, 2) + '\n', 'utf8'); + await core.summary + .addHeading('ATES evidence observation') + .addRaw(`- workflow: ${run.name}\n`) + .addRaw(`- agent: ${identity[1]}\n`) + .addRaw(`- run: ${run.id} / attempt ${run.run_attempt || 1}\n`) + .addRaw(`- SHA: ${run.head_sha}\n`) + .addRaw(`- conclusion: ${run.conclusion}\n`) + .addRaw(`- jobs observed: ${jobs.length}\n`) + .addRaw(`- structural complexity: ${taskComplexity ? 'available' : 'missing'}\n`) + .write(); + + - name: Move metadata into workspace + run: cp "${{ runner.temp }}/agent-throughput-metadata.json" ./agent-throughput-metadata.json + + - name: Emit sanitized ATES JSONL + run: | + python3 scripts/ci/emit_agent_throughput.py \ + --metadata agent-throughput-metadata.json \ + --output agent-throughput.ndjson + + - name: Validate emitter tests + run: python3 -m unittest tests.test_emit_agent_throughput -v + + - name: Reduce observed ATES metrics + run: | + python3 - <<'PY' + import json + from pathlib import Path + from she.metrics.agent_throughput import reduce_events + + events = [ + json.loads(line) + for line in Path("agent-throughput.ndjson").read_text(encoding="utf-8").splitlines() + if line.strip() + ] + metrics = reduce_events(events) + Path("agent-throughput-metrics.json").write_text( + json.dumps(metrics.to_dict(), sort_keys=True, indent=2) + "\n", + encoding="utf-8", + ) + PY + + - name: Build evidence receipt + env: + SOURCE_RUN_ID: ${{ github.event.workflow_run.id || '0' }} + SOURCE_RUN_ATTEMPT: ${{ github.event.workflow_run.run_attempt || '1' }} + run: | + python3 - <<'PY' + import hashlib + import json + import os + from pathlib import Path + + files = {} + for name in ("agent-throughput.ndjson", "agent-throughput-metrics.json"): + data = Path(name).read_bytes() + files[name] = { + "bytes": len(data), + "sha256": hashlib.sha256(data).hexdigest(), + } + + receipt = { + "schema_version": "ates.evidence.v1", + "producer": "agent-throughput-evidence", + "source_run_id": int(os.environ["SOURCE_RUN_ID"]), + "source_run_attempt": int(os.environ["SOURCE_RUN_ATTEMPT"]), + "files": files, + "privacy": { + "prompts": False, + "completions": False, + "credentials": False, + "tool_payloads": False, + }, + "ates_gate": "observational-only", + "baseline": "missing-unless-explicitly-declared", + } + Path("agent-throughput-receipt.json").write_text( + json.dumps(receipt, sort_keys=True, indent=2) + "\n", + encoding="utf-8", + ) + PY + + - name: Upload ATES evidence + uses: actions/upload-artifact@v4.6.2 + with: + name: agent-throughput-${{ github.event.workflow_run.id }}-${{ github.event.workflow_run.run_attempt }} + path: | + agent-throughput.ndjson + agent-throughput-metrics.json + agent-throughput-receipt.json + if-no-files-found: error + retention-days: 14 diff --git a/docs/ops/AGENT-OBSERVABILITY-PRIORITY-DECISION.md b/docs/ops/AGENT-OBSERVABILITY-PRIORITY-DECISION.md index 53297540d..65dc8ccaa 100644 --- a/docs/ops/AGENT-OBSERVABILITY-PRIORITY-DECISION.md +++ b/docs/ops/AGENT-OBSERVABILITY-PRIORITY-DECISION.md @@ -9,7 +9,7 @@ The observability stack is reorganized by dependency rather than by product cate - OpenTelemetry: neutral trace/span transport at the agent invocation boundary. - Docker: reproducible execution substrate for CI and agent jobs; useful for controlled experiments, isolated tooling, environment fingerprints, and reproducible failure reproduction. - Complexity scoring: explicit complexity plus the structural fallback is part of the measurement foundation. Tree-sitter is the preferred language-neutral structural expansion. -- ATES Phase A: the pure reducer, event schema, focused tests, and quality verification are already implemented; subsequent phases instrument the runtime around this foundation. +- ATES Phase A + B: the pure reducer, event schema, focused tests, quality verification, and a read-only `workflow_run` runtime evidence observer are implemented; ATES is the primary execution-measurement spine for the quality-first telemetry plane. **P1 — parallel evaluation and reproduction surfaces** @@ -47,9 +47,9 @@ ATES is an observation inside this quality-first frame, not the objective functi Pure ATES/WTCV reducer, sanitized event schema, focused fixtures, missing-evidence invariants, structural complexity fallback, and a visible Actions quality check. No external telemetry dependency. -### Phase B — emit +### Phase B — emit **[IMPLEMENTED]** -Teach eligible agent workflows to emit sanitized JSONL and publish a receipt alongside existing Action artifacts. Add immutable run/attempt/SHA linkage and environment fingerprints where available. +The read-only `agent-throughput-evidence` observer watches eligible completed agent workflows, emits sanitized JSONL, reduces the observations through ATES, and publishes a receipt with immutable run/attempt/SHA linkage. It does not execute triggering code and does not infer missing complexity or a sequential baseline. ### Phase C — correlate diff --git a/docs/ops/AGENT-OBSERVABILITY-RESEARCH-MATRIX.md b/docs/ops/AGENT-OBSERVABILITY-RESEARCH-MATRIX.md index fde60504d..33a9effda 100644 --- a/docs/ops/AGENT-OBSERVABILITY-RESEARCH-MATRIX.md +++ b/docs/ops/AGENT-OBSERVABILITY-RESEARCH-MATRIX.md @@ -11,6 +11,7 @@ The research candidates should be compared as **parallel adapters/providers agai | Candidate | Primary contribution | Integration target | Priority | Promotion stance | |---|---|---|---|---| | OpenTelemetry | neutral trace/span transport | GitHub Actions + agent invocation boundary | P0 | canonical interoperability layer | +| **ATES runtime evidence** | run/job/task throughput observations | `workflow_run` observer + canonical JSONL | **P0** | primary execution-measurement spine; observational only | | Langfuse | traces, datasets, experiments, scores | optional experiment/eval adapter | P1 | observational; no source-of-truth authority | | Phoenix | open-source tracing, evals, datasets, experiments | optional experiment/eval adapter; Docker-friendly lab | P1 | observational; no source-of-truth authority | | **Glama TDQS** | MCP/connector tool-definition quality | AEF tool-definition evaluation lane | **P1** | observational; no merge-quality authority | diff --git a/docs/ops/AGENT-THROUGHPUT-EVENT.schema.json b/docs/ops/AGENT-THROUGHPUT-EVENT.schema.json index 6c55f1c90..335cb8a9c 100644 --- a/docs/ops/AGENT-THROUGHPUT-EVENT.schema.json +++ b/docs/ops/AGENT-THROUGHPUT-EVENT.schema.json @@ -3,34 +3,111 @@ "$id": "https://github.com/timerloggedout-spec/termux-monorepo/blob/master/docs/ops/AGENT-THROUGHPUT-EVENT.schema.json", "title": "Agent Throughput Event", "type": "object", - "required": ["timestamp", "event"], + "required": [ + "timestamp", + "event" + ], "properties": { - "timestamp": {"type": "string", "format": "date-time"}, - "event": {"enum": ["task_started", "task_completed", "tool_call", "tool_retry", "handoff", "active_window"]}, - "agent_id": {"type": "string", "minLength": 1}, - "task_id": {"type": "string", "minLength": 1}, - "tool": {"type": "string", "minLength": 1}, - "status": {"enum": ["success", "failed", "error", "skipped"]}, - "duration_ms": {"type": "number", "minimum": 0}, - "latency_ms": {"type": "number", "minimum": 0}, - "active_seconds": {"type": "number", "minimum": 0}, - "inference_seconds": {"type": "number", "minimum": 0}, - "tokens_in": {"type": "number", "minimum": 0}, - "tokens_out": {"type": "number", "minimum": 0}, - "complexity_score": {"type": "number", "minimum": 0}, + "timestamp": { + "type": "string", + "format": "date-time" + }, + "event": { + "enum": [ + "task_started", + "task_completed", + "tool_call", + "tool_retry", + "handoff", + "active_window" + ] + }, + "agent_id": { + "type": "string", + "minLength": 1 + }, + "task_id": { + "type": "string", + "minLength": 1 + }, + "tool": { + "type": "string", + "minLength": 1 + }, + "status": { + "enum": [ + "success", + "failed", + "error", + "skipped" + ] + }, + "duration_ms": { + "type": "number", + "minimum": 0 + }, + "latency_ms": { + "type": "number", + "minimum": 0 + }, + "active_seconds": { + "type": "number", + "minimum": 0 + }, + "inference_seconds": { + "type": "number", + "minimum": 0 + }, + "tokens_in": { + "type": "number", + "minimum": 0 + }, + "tokens_out": { + "type": "number", + "minimum": 0 + }, + "complexity_score": { + "type": "number", + "minimum": 0 + }, "metrics": { "type": "object", "properties": { - "additions": {"type": "number", "minimum": 0}, - "deletions": {"type": "number", "minimum": 0}, - "files_changed": {"type": "number", "minimum": 0}, - "complexity_score": {"type": "number", "minimum": 0} + "additions": { + "type": "number", + "minimum": 0 + }, + "deletions": { + "type": "number", + "minimum": 0 + }, + "files_changed": { + "type": "number", + "minimum": 0 + }, + "complexity_score": { + "type": "number", + "minimum": 0 + } }, "additionalProperties": false }, - "gha_run_id": {"type": "integer", "minimum": 1}, - "gha_run_attempt": {"type": "integer", "minimum": 1}, - "source_sha": {"type": "string", "pattern": "^[0-9a-f]{40}$"} + "gha_run_id": { + "type": "integer", + "minimum": 1 + }, + "gha_run_attempt": { + "type": "integer", + "minimum": 1 + }, + "source_sha": { + "type": "string", + "pattern": "^[0-9a-f]{40}$" + }, + "event_id": { + "type": "string", + "pattern": "^[0-9a-f]{32}$" + } }, "additionalProperties": false } diff --git a/docs/ops/AGENT-THROUGHPUT-METRICS.md b/docs/ops/AGENT-THROUGHPUT-METRICS.md index 14a6ccb55..cfc3a951c 100644 --- a/docs/ops/AGENT-THROUGHPUT-METRICS.md +++ b/docs/ops/AGENT-THROUGHPUT-METRICS.md @@ -7,7 +7,7 @@ SHE already reconstructs durable GitHub Actions timing from run/job timestamps. This lane extends that reducer model to multi-agent execution telemetry without introducing a hosted observability dependency. -ATES is **not a future/fancy add-on**: Phase A has already landed as a pure reducer and test contract. The remaining phases add evidence emission, longitudinal correlation, provider experiments, and manager-level use around that foundation. +ATES is **not a future/fancy add-on**: Phase A has already landed as a pure reducer and test contract. Phase B now adds runtime evidence emission; the remaining phases add longitudinal correlation, provider experiments, and manager-level use around that foundation. The design accepts Gemini's ATES concepts, but makes three corrections: @@ -112,6 +112,9 @@ The repository should measure the adapters against the same event schema, then r - Phase A reducer: `she/metrics/agent_throughput.py` - Phase A tests: `tests/test_she_agent_throughput.py` - Phase A quality gate: `.github/workflows/agent-quality-lane.yml` + `scripts/ci/verify_agent_quality.py` +- Phase B emitter: `scripts/ci/emit_agent_throughput.py` +- Phase B observer: `.github/workflows/agent-throughput-evidence.yml` +- Phase B tests: `tests/test_emit_agent_throughput.py` - Hex-compatible sanitizer: `scripts/hex_moneyball_export.py` - Actions timing reducer: `she/metrics/job_timestamps.py` - Longitudinal corpus: `workspace/llm_map/context_relationships/` @@ -123,9 +126,9 @@ The repository should measure the adapters against the same event schema, then r Land schema + pure reducer + fixtures + quality verification. No external telemetry dependency. This is the ATES foundation already present on the active integration branch. -### Phase B — evidence emission **[NEXT]** +### Phase B — evidence emission **[IMPLEMENTED]** -Teach eligible agent workflows to emit sanitized JSONL and publish a receipt alongside existing Action artifacts. Every event must retain run/attempt/SHA linkage where available. +An observer workflow now watches eligible completed agent workflows, emits sanitized JSONL, and publishes a receipt alongside the reducer output. Every emitted event retains run/attempt/SHA linkage; ambiguous complexity and missing job timing remain absent. ### Phase C — longitudinal correlation diff --git a/docs/ops/AGENT-THROUGHPUT-PHASE-B.md b/docs/ops/AGENT-THROUGHPUT-PHASE-B.md new file mode 100644 index 000000000..b8fc19b46 --- /dev/null +++ b/docs/ops/AGENT-THROUGHPUT-PHASE-B.md @@ -0,0 +1,61 @@ +# ATES Phase B — Runtime Evidence Emission + +**Status:** implemented as an observer-only Phase B producer +**Priority:** P0 telemetry / P2 presentation +**Contract:** `docs/ops/AGENT-THROUGHPUT-EVENT.schema.json` + +## Why this is the next compounding step + +Phase A already provides the pure ATES/WTCV/TCV reducer and quality gate. The missing high-value boundary was runtime evidence: a schema without real run/attempt/SHA-linked events cannot support longitudinal measurement. + +Phase B closes that gap without rewriting every agent workflow. + +## Producer + +`.github/workflows/agent-throughput-evidence.yml` observes completed runs for the explicitly admitted agent workflows: + +- Gemini Dispatch +- Jules on Issues +- Agent review → auto Jules +- DeepSeek CI + +It uses the GitHub `workflow_run` observer boundary and reads completed job timings through the Actions API. The producer checks out only its own trusted default-branch implementation; it never executes the triggering run's code. + +The resulting evidence contains: + +- `task_started` +- `active_window` +- `task_completed` +- run ID / attempt +- source SHA +- agent identity from the explicit workflow allowlist +- structural complexity when exactly one associated PR supplies additions/deletions/files-changed + +Unknown or ambiguous evidence stays missing. + +## ATES semantics + +The emitted events are fed through the existing `she.metrics.agent_throughput.reduce_events` reducer. The observer does **not** invent a sequential baseline. Therefore `parallel_yield` and `ates` remain `null` until a declared baseline is supplied. + +That is intentional: the system now measures actual agent execution while preserving the repository invariant that missing evidence is not a fabricated zero or inferred comparison. + +### Next baseline increment + +Create a bounded baseline cohort using the same task definition and explicit serial execution, then attach its measured wall-clock duration to the cohort record. The baseline must be task-comparable and independently attributable; it must not be inferred from the parallel run itself. + +## Security / privacy + +No prompts, completions, credentials, tool payloads, arbitrary comments, or repository contents enter the evidence artifact. + +`workflow_run` is treated as a privileged observer boundary. The observer never checks out or executes the triggering SHA. The workflow remains read-only and treats all triggering-run data as untrusted metadata. + +## Acceptance + +- real completed agent runs produce JSONL events; +- every emitted event is bound to run/attempt/SHA; +- missing job timing does not manufacture duration; +- ambiguous PR associations do not manufacture complexity; +- artifacts contain sanitized JSONL + reducer output + receipt; +- ATES remains observational and cannot block merge; +- re-runs remain distinct via `run_attempt`; +- downstream consumers can deduplicate using the deterministic `event_id`. diff --git a/docs/ops/EVIDENCE-PROJECTION-SURFACE.md b/docs/ops/EVIDENCE-PROJECTION-SURFACE.md index 564796c1c..3c2d4e8f9 100644 --- a/docs/ops/EVIDENCE-PROJECTION-SURFACE.md +++ b/docs/ops/EVIDENCE-PROJECTION-SURFACE.md @@ -28,7 +28,9 @@ EPS closes that gap incrementally without requiring every agent workflow to be r ### Phase 2 producers -Eligible agent workflows add agent/task/provider/manager events using the same envelope. The existing Hex Moneyball grain contracts become EPS-compatible consumers rather than a parallel telemetry format. +`.github/workflows/agent-throughput-evidence.yml` now observes the explicitly admitted Gemini/Jules/DeepSeek agent workflows and emits sanitized ATES JSONL plus reducer/receipt artifacts. The existing Hex Moneyball grain contracts become EPS-compatible consumers rather than a parallel telemetry format. + +The observer is read-only and source-independent: it does not checkout or execute the triggering SHA. Structural complexity is attached only when exactly one associated PR supplies concrete additions/deletions/files-changed evidence. ## Envelope diff --git a/scripts/ci/emit_agent_throughput.py b/scripts/ci/emit_agent_throughput.py new file mode 100644 index 000000000..6f16e644a --- /dev/null +++ b/scripts/ci/emit_agent_throughput.py @@ -0,0 +1,227 @@ +#!/usr/bin/env python3 +"""Emit sanitized ATES events from trusted GitHub Actions run metadata. + +The producer is deliberately offline: GitHub API access belongs to the +workflow observer, while this module converts already-fetched metadata into +the repository's closed throughput-event contract. It never copies prompts, +completions, credentials, comments, or arbitrary payloads. +""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import math +from datetime import datetime +from pathlib import Path +from typing import Any, Mapping + + +EVENTS = { + "task_started", + "task_completed", + "active_window", + "tool_call", + "tool_retry", + "handoff", +} +STATUSES = {"success", "failed", "error", "skipped"} +SHA_LENGTH = 40 + + +def _iso(value: Any) -> str: + """Require an explicit timezone-bearing ISO-8601 timestamp.""" + if not isinstance(value, str) or not value: + raise ValueError("timestamp must be a non-empty string") + parsed = datetime.fromisoformat(value.replace("Z", "+00:00")) + if parsed.tzinfo is None or parsed.utcoffset() is None: + raise ValueError(f"timestamp must include a timezone: {value!r}") + return parsed.isoformat().replace("+00:00", "Z") + + +def _sha(value: Any) -> str: + """Require a full lowercase Git SHA.""" + if not isinstance(value, str) or len(value) != SHA_LENGTH: + raise ValueError("source_sha must be a 40-character lowercase SHA") + if any(ch not in "0123456789abcdef" for ch in value): + raise ValueError("source_sha must be hexadecimal") + return value + + +def _non_negative(value: Any) -> float: + """Normalize a finite non-negative number.""" + try: + number = float(value) + except (TypeError, ValueError) as exc: + raise ValueError(f"expected non-negative number, got {value!r}") from exc + if not math.isfinite(number) or number < 0: + raise ValueError(f"expected finite non-negative number, got {value!r}") + return number + + +def _status(conclusion: Any) -> str: + """Map an Actions conclusion into the closed telemetry status enum.""" + value = str(conclusion or "").lower() + if value in {"success", "neutral"}: + return "success" + if value in {"skipped"}: + return "skipped" + if value in {"failure", "timed_out", "startup_failure"}: + return "failed" + if value in {"cancelled", "action_required", "stale"}: + return "error" + return "error" + + +def _event_id(event: Mapping[str, Any]) -> str: + """Create a deterministic idempotency key without retaining arbitrary text.""" + stable = "|".join(str(event.get(key, "")) for key in ( + "source_sha", "gha_run_id", "gha_run_attempt", "agent_id", + "task_id", "event", "timestamp", + )) + return hashlib.sha256(stable.encode("utf-8")).hexdigest()[:32] + + +def _base(metadata: Mapping[str, Any], agent_id: str, task_id: str) -> dict[str, Any]: + """Build the common provenance fields for one sanitized event.""" + run = metadata["workflow_run"] + return { + "timestamp": _iso(run["created_at"]), + "event": "task_started", + "agent_id": agent_id, + "task_id": task_id, + "gha_run_id": int(run["id"]), + "gha_run_attempt": int(run.get("run_attempt", 1)), + "source_sha": _sha(run["head_sha"]), + } + + +def build_events(metadata: Mapping[str, Any]) -> list[dict[str, Any]]: + """Convert trusted run/job metadata into bounded throughput events. + + One Actions job is represented as one bounded task. Complexity is attached + only when the observer found exactly one associated PR and therefore has a + concrete structural evidence source. Missing complexity remains missing. + """ + run = metadata.get("workflow_run") + if not isinstance(run, Mapping): + raise ValueError("workflow_run metadata is required") + agent_id = metadata.get("agent_id") + if not isinstance(agent_id, str) or not agent_id: + raise ValueError("agent_id is required") + source_sha = _sha(run.get("head_sha")) + jobs = metadata.get("jobs", []) + if not isinstance(jobs, list): + raise ValueError("jobs must be a list") + + complexity = metadata.get("task_complexity") + if complexity is not None: + if not isinstance(complexity, Mapping): + raise ValueError("task_complexity must be an object") + complexity = { + "additions": _non_negative(complexity.get("additions")), + "deletions": _non_negative(complexity.get("deletions")), + "files_changed": _non_negative(complexity.get("files_changed")), + } + + events: list[dict[str, Any]] = [] + for job in jobs: + if not isinstance(job, Mapping): + raise ValueError("job entries must be objects") + job_id = int(job["id"]) + started = job.get("started_at") + completed = job.get("completed_at") + if not started or not completed: + continue + task_id = f"gha:{run['id']}:{run.get('run_attempt', 1)}:job:{job_id}" + start = _iso(started) + finish = _iso(completed) + status = _status(job.get("conclusion")) + + started_event = _base(metadata, agent_id, task_id) + started_event.update({"timestamp": start, "event": "task_started"}) + started_event["source_sha"] = source_sha + events.append(started_event) + + start_dt = datetime.fromisoformat(start.replace("Z", "+00:00")) + finish_dt = datetime.fromisoformat(finish.replace("Z", "+00:00")) + duration = max(0.0, (finish_dt - start_dt).total_seconds()) + + active_event = _base(metadata, agent_id, task_id) + active_event.update({"timestamp": start, "event": "active_window", "active_seconds": duration}) + active_event["source_sha"] = source_sha + events.append(active_event) + + completed_event = _base(metadata, agent_id, task_id) + completed_event.update({"timestamp": finish, "event": "task_completed", "status": status}) + completed_event["source_sha"] = source_sha + if complexity is not None: + completed_event["metrics"] = dict(complexity) + events.append(completed_event) + + for event in events: + event["event_id"] = _event_id(event) + return events + + +def validate_event(event: Mapping[str, Any]) -> None: + """Validate the closed event shape without requiring third-party packages.""" + allowed = { + "timestamp", "event", "agent_id", "task_id", "tool", "status", + "duration_ms", "latency_ms", "active_seconds", "inference_seconds", + "tokens_in", "tokens_out", "complexity_score", "metrics", + "gha_run_id", "gha_run_attempt", "source_sha", "event_id", + } + unknown = set(event) - allowed + if unknown: + raise ValueError(f"unsupported telemetry fields: {sorted(unknown)}") + if not isinstance(event.get("timestamp"), str) or not event["timestamp"]: + raise ValueError("event timestamp is required") + if event.get("event") not in EVENTS: + raise ValueError("invalid event type") + if event.get("status") is not None and event["status"] not in STATUSES: + raise ValueError("invalid status") + if event.get("agent_id") is not None and not isinstance(event["agent_id"], str): + raise ValueError("agent_id must be a string") + if event.get("task_id") is not None and not isinstance(event["task_id"], str): + raise ValueError("task_id must be a string") + if event.get("gha_run_id") is not None and int(event["gha_run_id"]) < 1: + raise ValueError("gha_run_id must be positive") + if event.get("gha_run_attempt") is not None and int(event["gha_run_attempt"]) < 1: + raise ValueError("gha_run_attempt must be positive") + if event.get("source_sha") is not None: + _sha(event["source_sha"]) + if event.get("metrics") is not None: + metrics = event["metrics"] + if not isinstance(metrics, Mapping): + raise ValueError("metrics must be an object") + if set(metrics) - {"additions", "deletions", "files_changed", "complexity_score"}: + raise ValueError("unsupported structural metric") + for value in metrics.values(): + _non_negative(value) + + +def write_events(events: list[dict[str, Any]], output: Path) -> None: + """Validate and write deterministic JSONL evidence.""" + output.parent.mkdir(parents=True, exist_ok=True) + with output.open("w", encoding="utf-8") as handle: + for event in events: + validate_event(event) + handle.write(json.dumps(event, sort_keys=True, separators=(",", ":")) + "\n") + + +def main() -> int: + """Run the metadata-to-JSONL producer.""" + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--metadata", required=True, type=Path) + parser.add_argument("--output", required=True, type=Path) + args = parser.parse_args() + metadata = json.loads(args.metadata.read_text(encoding="utf-8")) + events = build_events(metadata) + write_events(events, args.output) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/she/metrics/agent_throughput.py b/she/metrics/agent_throughput.py index 64a51fc8e..48b878178 100644 --- a/she/metrics/agent_throughput.py +++ b/she/metrics/agent_throughput.py @@ -173,7 +173,7 @@ def reduce_events( density = action_count / active_sec if active_sec else 0.0 eta = None - if sequential_baseline_sec is not None and agents and workflow_sec > 0: + if sequential_baseline_sec is not None and len(agents) > 0 and workflow_sec > 0: # Sequential baseline is only valid when explicitly supplied; never infer # one from the observed parallel run because that would bias the metric. eta = _positive(sequential_baseline_sec) / (len(agents) * workflow_sec) @@ -203,7 +203,7 @@ def reduce_events( tool_actions=action_count, failed_actions=len(failed), retries=len(retries), - agents=len(agents) or 1, + agents=len(agents), tcv_tasks_per_min=round(tcv, 6), wtcv_per_min=round(wtcv, 6) if wtcv is not None else None, retry_penalty_ratio=round(rpi, 6), diff --git a/tests/test_emit_agent_throughput.py b/tests/test_emit_agent_throughput.py new file mode 100644 index 000000000..620689a22 --- /dev/null +++ b/tests/test_emit_agent_throughput.py @@ -0,0 +1,80 @@ +import json +import tempfile +import unittest +from pathlib import Path + +from scripts.ci.emit_agent_throughput import build_events, validate_event, write_events + + +class AgentThroughputEmitterTests(unittest.TestCase): + def metadata(self, **overrides): + payload = { + "agent_id": "gemini", + "workflow_run": { + "id": 12345, + "run_attempt": 2, + "head_sha": "a" * 40, + "created_at": "2026-09-25T18:00:00Z", + "conclusion": "success", + }, + "jobs": [{ + "id": 99, + "name": "agent", + "started_at": "2026-09-25T18:01:00Z", + "completed_at": "2026-09-25T18:03:30Z", + "conclusion": "success", + }], + "task_complexity": { + "additions": 12, + "deletions": 3, + "files_changed": 2, + }, + } + payload.update(overrides) + return payload + + def test_emits_start_active_and_completion_with_provenance(self): + events = build_events(self.metadata()) + self.assertEqual([e["event"] for e in events], [ + "task_started", "active_window", "task_completed", + ]) + self.assertEqual(events[-1]["status"], "success") + self.assertEqual(events[-1]["metrics"]["files_changed"], 2) + self.assertEqual(events[-1]["gha_run_attempt"], 2) + self.assertEqual(events[-1]["source_sha"], "a" * 40) + self.assertTrue(events[-1]["event_id"]) + + def test_missing_complexity_stays_missing(self): + events = build_events(self.metadata(task_complexity=None)) + self.assertNotIn("metrics", events[-1]) + + def test_missing_job_timing_does_not_fabricate_duration(self): + data = self.metadata() + data["jobs"][0]["started_at"] = None + self.assertEqual(build_events(data), []) + + def test_failed_job_is_observed_as_failed_task(self): + data = self.metadata() + data["jobs"][0]["conclusion"] = "failure" + events = build_events(data) + self.assertEqual(events[-1]["status"], "failed") + + def test_unknown_fields_are_rejected(self): + event = build_events(self.metadata())[0] + event["prompt"] = "must never enter evidence" + with self.assertRaises(ValueError): + validate_event(event) + + def test_jsonl_is_deterministic_and_validated(self): + events = build_events(self.metadata()) + with tempfile.TemporaryDirectory() as directory: + output = Path(directory) / "events.ndjson" + write_events(events, output) + rows = [json.loads(line) for line in output.read_text().splitlines()] + self.assertEqual(len(rows), 3) + self.assertEqual(rows[1]["active_seconds"], 150.0) + self.assertTrue(all("prompt" not in row for row in rows)) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_she_agent_throughput.py b/tests/test_she_agent_throughput.py index 1bea5f824..ce197f8e7 100644 --- a/tests/test_she_agent_throughput.py +++ b/tests/test_she_agent_throughput.py @@ -78,6 +78,15 @@ def test_offset_naive_timestamps_are_excluded(self): ]) self.assertEqual(metrics.workflow_minutes, 0.0) + def test_missing_agent_identity_does_not_fabricate_agent_count(self): + metrics = reduce_events([ + {"timestamp": "2026-09-14T10:00:00Z", "event": "task_completed", "complexity_score": 2}, + {"timestamp": "2026-09-14T10:01:00Z", "event": "task_started"}, + ], sequential_baseline_sec=120) + self.assertEqual(metrics.agents, 0) + self.assertIsNone(metrics.parallel_yield) + self.assertIsNone(metrics.ates) + def test_missing_baseline_does_not_fabricate_ates(self): metrics = reduce_events([ {"timestamp": "2026-09-14T10:00:00Z", "agent_id": "a", "event": "task_completed", "complexity_score": 2},