Skip to content
Merged
21 changes: 21 additions & 0 deletions agent/auxiliary_client.py
Original file line number Diff line number Diff line change
Expand Up @@ -2851,6 +2851,7 @@ def wrapped(*args, **kwargs):
"attempt_count": 0,
"provider": "",
"model": "",
"response_model": None,
"api_mode": "chat_completions",
})
try:
Expand All @@ -2876,6 +2877,7 @@ async def wrapped(*args, **kwargs):
"attempt_count": 0,
"provider": "",
"model": "",
"response_model": None,
"api_mode": "chat_completions",
})
try:
Expand All @@ -2899,6 +2901,7 @@ def _set_relay_auxiliary_route(
return
context["provider"] = str(provider or "auxiliary")
context["model"] = str(model or "unknown")
context["response_model"] = None
context["api_mode"] = str(api_mode or "chat_completions")


Expand Down Expand Up @@ -8070,6 +8073,7 @@ def _validate_llm_response(
except (AttributeError, TypeError, IndexError) as exc:
recovered = _recover_aux_response_message(response)
if recovered is not None:
_record_relay_auxiliary_response_model(response)
_complete_relay_auxiliary_call()
return recovered
response_type = type(response).__name__
Expand All @@ -8080,6 +8084,7 @@ def _validate_llm_response(
f"Expected object with .choices[0].message β€” check provider "
f"adapter or custom endpoint compatibility."
) from exc
_record_relay_auxiliary_response_model(response)
_complete_relay_auxiliary_call()
return response

Expand All @@ -8094,9 +8099,25 @@ def _complete_relay_auxiliary_call(*, outcome: str = "success") -> None:
relay_llm.complete_logical_call(
str(context.get("request_id") or ""),
outcome=outcome,
model_name=str(context.get("model") or "unknown"),
provider_name=str(context.get("provider") or "auxiliary"),
response_model_name=context.get("response_model"),
)


def _record_relay_auxiliary_response_model(response: Any) -> None:
"""Retain the provider-reported model for terminal route attribution."""
context = _RELAY_AUX_CALL_CONTEXT.get()
if context is None:
return
if isinstance(response, dict):
model = response.get("model")
else:
model = getattr(response, "model", None)
if isinstance(model, str) and model.strip():
context["response_model"] = model


def _fail_relay_auxiliary_call() -> None:
"""Close a terminally failed call without replacing its original error."""
try:
Expand Down
87 changes: 78 additions & 9 deletions agent/relay_llm.py
Original file line number Diff line number Diff line change
Expand Up @@ -376,6 +376,14 @@ def __init__(
self._callback_error: BaseException | None = None
self._logical: tuple[relay_runtime.RelayTurnContext, Any, str] | None = None
self._defer_logical_completion = defer_logical_completion
if str((metadata or {}).get("call_role") or "").startswith("auxiliary:"):
self._logical_model_name: str | None = model_name
self._logical_provider_name: str | None = name
self._logical_response_model_name: str | None = None
else:
self._logical_model_name = None
self._logical_provider_name = None
self._logical_response_model_name = None
self._on_chunk = on_chunk
self._chunk_adapter = chunk_adapter or _namespace
self._accept_chunk = accept_chunk
Expand Down Expand Up @@ -485,8 +493,12 @@ def relay_finalizer() -> Any:
return None
try:
if self.final_response is not None:
return _jsonable(self.final_response)
return _jsonable(run_callback(finalizer))
response = self.final_response
else:
response = run_callback(finalizer)
if self._logical_model_name is not None:
self._logical_response_model_name = _response_model_name(response)
return _jsonable(response)
except BaseException as exc:
self._callback_error = exc
raise
Expand Down Expand Up @@ -528,6 +540,9 @@ def relay_finalizer() -> Any:
_complete_logical(
self._logical,
outcome="cancelled" if _is_cancellation(exc) else "failed",
model_name=self._logical_model_name,
provider_name=self._logical_provider_name,
response_model_name=self._logical_response_model_name,
)
self._logical = None
loop.close()
Expand Down Expand Up @@ -560,7 +575,13 @@ async def next_chunk() -> Any:
if self._raw_chunks:
self.output_modified = True
if not self._defer_logical_completion:
_complete_logical(self._logical, outcome="success")
_complete_logical(
self._logical,
outcome="success",
model_name=self._logical_model_name,
provider_name=self._logical_provider_name,
response_model_name=self._logical_response_model_name,
)
self._logical = None
self._close(logical_outcome="cancelled")
raise StopIteration from None
Expand Down Expand Up @@ -633,7 +654,13 @@ async def close_stream() -> None:
)
loop.close()
if not self._defer_logical_completion:
_complete_logical(self._logical, outcome="success")
_complete_logical(
self._logical,
outcome="success",
model_name=self._logical_model_name,
provider_name=self._logical_provider_name,
response_model_name=self._logical_response_model_name,
)
self._logical = None

def _close(self, *, logical_outcome: str) -> None:
Expand Down Expand Up @@ -663,7 +690,13 @@ def _close(self, *, logical_outcome: str) -> None:
exc_info=True,
)
if not self._defer_logical_completion:
_complete_logical(self._logical, outcome=logical_outcome)
_complete_logical(
self._logical,
outcome=logical_outcome,
model_name=self._logical_model_name,
provider_name=self._logical_provider_name,
response_model_name=self._logical_response_model_name,
)
self._logical = None
return
close = getattr(self._stream, "aclose", None)
Expand All @@ -678,7 +711,13 @@ async def close_stream() -> None:
if self._close_error is None:
self._close_error = exc
if not self._defer_logical_completion:
_complete_logical(self._logical, outcome=logical_outcome)
_complete_logical(
self._logical,
outcome=logical_outcome,
model_name=self._logical_model_name,
provider_name=self._logical_provider_name,
response_model_name=self._logical_response_model_name,
)
self._logical = None
loop.close()

Expand Down Expand Up @@ -817,6 +856,9 @@ def _complete_logical(
logical: tuple[relay_runtime.RelayTurnContext, Any, str] | None,
*,
outcome: str,
model_name: str | None = None,
provider_name: str | None = None,
response_model_name: str | None = None,
) -> None:
if logical is None:
return
Expand All @@ -831,11 +873,16 @@ def _complete_logical(
if lease.session is None:
return
try:
output = {"outcome": outcome}
if model_name is not None and provider_name is not None:
output.update({"model": model_name, "provider": provider_name})
if response_model_name is not None:
output["response_model"] = response_model_name
lease.host.run_in_session(
lease.session,
lease.host.relay.scope.pop,
handle,
output={"outcome": outcome},
output=output,
metadata={
relay_runtime.RUNTIME_SCHEMA_KEY: relay_runtime.RUNTIME_SCHEMA_VERSION,
relay_runtime.RUNTIME_INSTANCE_KEY: lease.host.runtime_id,
Expand Down Expand Up @@ -885,15 +932,37 @@ def _is_cancellation(error: BaseException) -> bool:
)


def complete_logical_call(api_request_id: str, *, outcome: str) -> None:
def complete_logical_call(
api_request_id: str,
*,
outcome: str,
model_name: str | None = None,
provider_name: str | None = None,
response_model_name: str | None = None,
) -> None:
"""Complete the active turn's logical LLM call after caller validation."""
turn = relay_runtime.active_turn()
if turn is None or not api_request_id:
return
with turn.logical_llm_lock:
handle = turn.logical_llm_calls.get(api_request_id)
if handle is not None:
_complete_logical((turn, handle, api_request_id), outcome=outcome)
_complete_logical(
(turn, handle, api_request_id),
outcome=outcome,
model_name=model_name,
provider_name=provider_name,
response_model_name=response_model_name,
)


def _response_model_name(response: Any) -> str | None:
"""Return a provider-reported model name when one is available."""
if isinstance(response, dict):
value = response.get("model")
else:
value = getattr(response, "model", None)
return value if isinstance(value, str) and value.strip() else None


def _provider_request(
Expand Down
21 changes: 15 additions & 6 deletions docs/observability/relay-shared-metrics.md
Original file line number Diff line number Diff line change
Expand Up @@ -67,10 +67,16 @@ Hermes turn, API, and tool hooks

Hermes sends an empty `LLMRequest` into the metrics-owned lifecycle. This does
not describe the separate managed-execution call through the native runtime
documented above. The terminal metrics event contains only bounded model
family, provider family, locality, call role, and outcome values. Prompts,
responses, exact model IDs, endpoints, errors, session IDs, task IDs, and
request IDs are not included in the metrics event or package.
documented above. The terminal metrics event contains the model identifier and
provider route that Hermes used for the logical call, such as
`nvidia/nemotron-3-ultra` through `openrouter`. These identifiers are
lowercased and structurally bounded, but they are not normalized through a
checked-in model catalog. Pricing and model-family classification belong to
the metrics backend. Prompts, responses, endpoints, errors, session IDs, task
IDs, and request IDs are not included in the metrics event or package.
New calls use `hermes.model_route.count`. The previous
`hermes.model_call.count` contract remains readable only so pending local
counters created by older builds can be exported without losing data.

Each task run is a Relay `Function` scope named `hermes.task_run`, parented to
the owning Hermes session. The start counter contains only bounded execution
Expand All @@ -96,6 +102,9 @@ files are immutable delta documents that conform to a closed JSON schema and
are written with atomic replacement. Fully packaged aggregate rows and
successfully exported package rows and files are retained locally for 30 days.
Pending package rows and counters with unexported deltas are never pruned.
Package schema v1 remains unchanged for existing outbox files. New packages
use v2, which accepts both the retired model-call contract and the current
model-route contract so upgrades can drain pending counters safely.

Each package contains an `install_id` generated as a random UUID. Despite the
schema field name, its current scope is one `HERMES_HOME`, so it is more
Expand Down Expand Up @@ -123,5 +132,5 @@ The script uses the installed `nemo-relay` dependency by default. Pass
binding.

The smoke verifies the model request reached the local server, model and task
counters were stored, one package was exported, and prompt, response, and
exact-model canaries are absent from the package.
counters were stored with the expected model and provider, one package was
exported, and prompt and response canaries are absent from the package.
Loading
Loading