hermes-agent/tests/telemetry/test_rollup.py
emozilla 0ebdd48f9d refactor(telemetry): cut dead schema; tests assert what's actually written
Self-review after the #51714 feedback found the reviewer's dead-table finding
was not isolated — the schema advertised far more than the code populates, and
our own tests hid it by hand-feeding fields production never sends. Make the
surface honest by subtraction.

Schema (10 tel_* tables -> 5):
  - Delete tel_gateway_events, tel_cron_events, tel_skill_events,
    tel_memory_events, tel_feedback_events — declared, never written, never read.
  - Drop columns nothing populates: tel_runs.{profile_id,estimated_cost_usd,
    cost_status}; tel_model_calls.{ttft_ms,estimated_cost_usd,cost_status,
    cost_source,end_reason,retry_count}; tel_tool_calls.{backend,retry_count,
    approval}; tel_spans.attrs_json. Cost duplicated the existing sessions
    billing columns and was always NULL here.
  - events.py / emitter _TABLE_COLUMNS / OTLP _span_attrs / rollup / preview
    display all trimmed to match.

Correctness:
  - end_reason no longer hardcodes "completed". Production finalize callers pass
    `reason` (shutdown/session_expired/session_reset); _coarse_end_reason now
    reads it and maps accordingly.
  - Fix a latent bug the trim exposed: the model_call hook passed end_reason= to
    ModelCallEvent, which the @_safe wrapper was silently swallowing — so
    tel_model_calls dropped every row in real runs. Now writes correctly.

Tests:
  - Stop hand-feeding estimated_cost_usd / turn_exit_reason that no production
    call site sends. Finalize is now driven with the real `reason` kwarg, and
    assertions cover only fields that are actually populated. This is what let
    the model_call drop hide — the suite graded on a fictional contract.

Net: a smaller system that does what it says. Verified end-to-end over the real
dispatch path (runs + connected span tree + model/tool rows populate; dead
tables gone). 160 telemetry/state/insights tests green.
2026-06-27 01:21:41 -04:00

88 lines
3.5 KiB
Python

"""rollup tests: tel_* -> per-run summary events with REAL values (local only)."""
from __future__ import annotations
import sqlite3
import time
import hermes_state
from agent.telemetry import rollup
from agent.telemetry.emitter import TelemetryEmitter
from agent.telemetry.events import ModelCallEvent, RunEvent, ToolCallEvent
def _seed(tmp_path):
db = tmp_path / "state.db"
conn = sqlite3.connect(db)
conn.executescript(hermes_state.SCHEMA_SQL)
conn.close()
em = TelemetryEmitter(events_path=tmp_path / "tel" / "e.jsonl", db_path=db)
now = time.time_ns()
em.emit(RunEvent(run_id="r1", trace_id="t1", entrypoint="gateway",
platform="telegram", end_reason="completed",
start_ns=now - 90_000_000, end_ns=now,
model_call_count=2, tool_call_count=2))
em.emit(ModelCallEvent(span_id="m1", run_id="r1", provider="anthropic",
model="claude-opus-4", input_tokens=60000, output_tokens=8000))
em.emit(ModelCallEvent(span_id="m2", run_id="r1", provider="anthropic",
model="claude-opus-4", input_tokens=5000, output_tokens=500))
em.emit(ToolCallEvent(span_id="tc1", run_id="r1", tool_name="web_search",
result_class="ok"))
em.emit(ToolCallEvent(span_id="tc2", run_id="r1", tool_name="browser_navigate",
result_class="ok"))
# an in-progress run (no end_ns) must be excluded
em.emit(RunEvent(run_id="r2", trace_id="t2", entrypoint="cli", start_ns=now))
em.flush()
em.close()
return db
def test_builds_one_event_per_completed_run_with_real_values(tmp_path):
db = _seed(tmp_path)
events = rollup.build_aggregate_events(install_id="fixed-id", db_path=db,
include_heartbeat=False)
wf = [e for e in events if e["event_name"] == "workflow_completed"]
assert len(wf) == 1 # r2 (no end_ns) excluded
e = wf[0]
assert e["entrypoint"] == "gateway"
assert e["platform"] == "telegram"
# REAL model id + provider, not a bucket/class
models = {m["model"] for m in e["models_used"]}
assert models == {"claude-opus-4"}
assert e["models_used"][0]["provider"] == "anthropic"
assert sorted(e["tools_used"]) == ["browser_navigate", "web_search"]
# real token totals, not buckets
assert e["input_tokens"] == 65000
assert e["output_tokens"] == 8500
def test_real_model_and_tool_names_present(tmp_path):
db = _seed(tmp_path)
events = rollup.build_aggregate_events(install_id="fixed-id", db_path=db)
blob = " ".join(str(v) for e in events for v in e.values())
assert "claude-opus-4" in blob
assert "web_search" in blob
def test_heartbeat_included_by_default(tmp_path):
db = _seed(tmp_path)
events = rollup.build_aggregate_events(install_id="fixed-id", db_path=db)
assert any(e["event_name"] == "heartbeat" for e in events)
def test_summarize_counts_by_event_name(tmp_path):
db = _seed(tmp_path)
events = rollup.build_aggregate_events(install_id="fixed-id", db_path=db)
s = rollup.summarize(events)
assert s["total"] == len(events)
assert s["by_event_name"]["workflow_completed"] == 1
assert s["by_event_name"]["heartbeat"] == 1
def test_empty_db_yields_only_heartbeat(tmp_path):
db = tmp_path / "state.db"
conn = sqlite3.connect(db)
conn.executescript(hermes_state.SCHEMA_SQL)
conn.close()
events = rollup.build_aggregate_events(install_id="x", db_path=db)
assert [e["event_name"] for e in events] == ["heartbeat"]