diff --git a/docs/content/docs/metrics.mdx b/docs/content/docs/metrics.mdx index ea8c2577f..c07953d6a 100644 --- a/docs/content/docs/metrics.mdx +++ b/docs/content/docs/metrics.mdx @@ -113,7 +113,11 @@ HEADROOM_OTEL_RESOURCE_ATTRIBUTES=deployment.environment=prod | `HEADROOM_OTEL_SERVICE_NAME` | `headroom-proxy` | OTEL `service.name` | | `HEADROOM_OTEL_RESOURCE_ATTRIBUTES` | unset | Comma-separated resource attributes | -Exported counters include `headroom.proxy.requests`, `headroom.proxy.tokens.input`, `headroom.proxy.tokens.output`, `headroom.proxy.tokens.saved`, and `headroom.proxy.cache.read_tokens` / `write_tokens`. +Exported counters include `headroom.proxy.requests`, `headroom.proxy.tokens.input`, and +`headroom.proxy.tokens.output`. `headroom.proxy.tokens.saved` is the all-layer total: +message/compression savings plus tool-schema deferral savings. The component counter +`headroom.proxy.tokens.tool_schema_saved` exposes the deferral portion separately; +`headroom.compression.tokens.saved` remains the compression-pipeline component. Confirm the exporter is live with `curl -s http://localhost:8787/stats | jq .otel`. diff --git a/headroom/observability/metrics.py b/headroom/observability/metrics.py index a2c55ae7b..fd020cfa0 100644 --- a/headroom/observability/metrics.py +++ b/headroom/observability/metrics.py @@ -165,7 +165,14 @@ class HeadroomOtelMetrics: ) self._proxy_saved_tokens = self._meter.create_counter( "headroom.proxy.tokens.saved", - description="Input tokens saved by Headroom compression.", + description=( + "Input tokens saved across Headroom compression and tool-schema deferral." + ), + unit="1", + ) + self._proxy_tool_schema_saved_tokens = self._meter.create_counter( + "headroom.proxy.tokens.tool_schema_saved", + description="Input tokens saved by deferring tool schemas.", unit="1", ) self._proxy_cache_read_tokens = self._meter.create_counter( @@ -346,6 +353,7 @@ class HeadroomOtelMetrics: output_tokens: int, tokens_saved: int, latency_ms: float, + tool_search_saved: int = 0, cached: bool = False, overhead_ms: float = 0.0, ttfb_ms: float = 0.0, @@ -363,7 +371,11 @@ class HeadroomOtelMetrics: self._proxy_input_tokens.add(max(input_tokens, 0), attrs) self._proxy_output_tokens.add(max(output_tokens, 0), attrs) - self._proxy_saved_tokens.add(max(tokens_saved, 0), attrs) + compression_saved = max(tokens_saved, 0) + tool_schema_saved = max(tool_search_saved, 0) + self._proxy_saved_tokens.add(compression_saved + tool_schema_saved, attrs) + if tool_schema_saved > 0: + self._proxy_tool_schema_saved_tokens.add(tool_schema_saved, attrs) self._proxy_latency.record(max(latency_ms, 0.0) / _MILLISECONDS_TO_SECONDS, attrs) if overhead_ms > 0: diff --git a/headroom/proxy/prometheus_metrics.py b/headroom/proxy/prometheus_metrics.py index 39bef4754..994321100 100644 --- a/headroom/proxy/prometheus_metrics.py +++ b/headroom/proxy/prometheus_metrics.py @@ -951,6 +951,7 @@ class PrometheusMetrics: input_tokens=input_tokens, output_tokens=output_tokens, tokens_saved=tokens_saved, + tool_search_saved=tool_search_saved, latency_ms=latency_ms, cached=cached, overhead_ms=overhead_ms, diff --git a/tests/test_observability_metrics.py b/tests/test_observability_metrics.py index ae539bddb..181864b74 100644 --- a/tests/test_observability_metrics.py +++ b/tests/test_observability_metrics.py @@ -51,6 +51,7 @@ def test_headroom_otel_metrics_records_proxy_and_pipeline_metrics() -> None: input_tokens=120, output_tokens=30, tokens_saved=45, + tool_search_saved=15, latency_ms=18.5, cached=True, overhead_ms=4.0, @@ -84,6 +85,32 @@ def test_headroom_otel_metrics_records_proxy_and_pipeline_metrics() -> None: ) assert request_point.value == 1 + saved_tokens = metrics["headroom.proxy.tokens.saved"] + saved_point = _find_point( + saved_tokens, + provider="anthropic", + model="claude-opus-4-6", + cached=True, + ) + assert saved_point.value == 60 + + tool_schema_saved = metrics["headroom.proxy.tokens.tool_schema_saved"] + tool_schema_point = _find_point( + tool_schema_saved, + provider="anthropic", + model="claude-opus-4-6", + cached=True, + ) + assert tool_schema_point.value == 15 + + compression_saved = metrics["headroom.compression.tokens.saved"] + compression_saved_point = _find_point( + compression_saved, + provider="anthropic", + model="claude-opus-4-6", + ) + assert compression_saved_point.value == 45 + latency = metrics["headroom.proxy.request.duration"] latency_point = _find_point( latency, diff --git a/tests/test_savings_ledger_before_forwarded.py b/tests/test_savings_ledger_before_forwarded.py index 72d158377..c50f6b938 100644 --- a/tests/test_savings_ledger_before_forwarded.py +++ b/tests/test_savings_ledger_before_forwarded.py @@ -21,8 +21,11 @@ class _FakeSavingsTracker: class _FakeOtelMetrics: + def __init__(self) -> None: + self.calls: list[dict[str, Any]] = [] + def record_proxy_request(self, **kwargs: Any) -> None: - pass + self.calls.append(kwargs) @pytest.mark.asyncio @@ -65,6 +68,29 @@ async def test_record_savings_event_uses_original_input_as_before( ] +@pytest.mark.asyncio +async def test_record_request_forwards_tool_savings_to_otel() -> None: + otel = _FakeOtelMetrics() + metrics = prometheus_metrics.PrometheusMetrics( + savings_tracker=_FakeSavingsTracker(), + otel_metrics=otel, + ) + + await metrics.record_request( + provider="anthropic", + model="claude-opus-4-6", + input_tokens=600, + output_tokens=25, + tokens_saved=400, + tool_search_saved=13182, + latency_ms=10.0, + ) + + assert len(otel.calls) == 1 + assert otel.calls[0]["tokens_saved"] == 400 + assert otel.calls[0]["tool_search_saved"] == 13182 + + def _capture_ledger(monkeypatch: pytest.MonkeyPatch) -> list[dict[str, Any]]: calls: list[dict[str, Any]] = [] monkeypatch.setattr(