diff --git a/CHANGELOG.md b/CHANGELOG.md index cb41d6292..33e1817b1 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -81,6 +81,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ### Features +* **proxy:** report a new-content-relative input savings rate in `/stats`: `tokens.new_input_tokens` (provider-billed non-cache-read input: uncached + cache-write tokens, from response usage) and `tokens.new_input_savings_percent` (savings as a fraction of new input plus the tokens compression removed before they could be billed). The existing whole-request ratios recount the full transcript on every turn, so a 200-turn session counts its history 200x into the denominator and long-running cached sessions (especially 1M-context models, which never compact) dilute toward ~0% regardless of how well compression performs on content newly entering context. Purely additive; existing fields unchanged. Reports 0 when no cache usage data exists (e.g. providers without cache metrics) rather than dividing savings by themselves. * **transforms:** first-class C# support in `CodeAwareCompressor` via the tree-sitter `csharp` grammar already shipped in the pinned `tree-sitter-language-pack` — no new dependencies ([#1664](https://github.com/headroomlabs-ai/headroom/issues/1664)). Parity with Java/C++/Rust: signatures preserved verbatim, method/constructor/destructor/operator/local-function bodies compressed; block-scoped and file-scoped namespaces, records, structs, interfaces, and enums handled; C#-distinctive auto-detection. Preprocessor conditionals (`#if`…`#endif`) are preserved verbatim as opaque regions (blocks wrapping only `using` directives stay with the imports), `#region` markers no longer swallow the following line during class-member extraction, and top-of-file license banners / `#region License` headers stay on top instead of being relocated below the code. Real-repo runs: 16.1% tokens saved on Newtonsoft.Json (945 files), 37.8% on Polly (797 files), output syntax-valid for 1742/1742 files. * **proxy:** add provider-only HTTP proxy routing via `--http-proxy` and `HEADROOM_HTTP_PROXY`. Upstream LLM provider calls can now use an HTTP proxy without setting process-wide `HTTP_PROXY`/`HTTPS_PROXY` variables that are inherited by tool executions; proxied provider clients use HTTP/1.1 so HTTPS provider APIs can tunnel through CONNECT. * **proxy:** add output shaping for OpenAI Responses traffic on `/v1/responses` HTTP requests and Codex WebSocket `response.create` frames, with stable output-savings holdout keys and counted WS token strata for the experiment. diff --git a/headroom/proxy/server.py b/headroom/proxy/server.py index d69225862..e8d929bcd 100644 --- a/headroom/proxy/server.py +++ b/headroom/proxy/server.py @@ -3112,6 +3112,19 @@ def create_app(config: ProxyConfig | None = None) -> FastAPI: # schema size. So the savings rate is plain `saved / attempted` # — adding `saved` again would double-count. attempted_input_tokens = getattr(m, "attempted_input_tokens_total", 0) + # New-content denominator: what the provider actually billed as + # non-cache-read input (uncached + cache-write tokens, summed + # across providers from response usage). Unlike + # `proxy_total_before_compression`, this does NOT recount the + # full transcript on every turn — a long session's history is + # served from prefix cache, not re-billed, so it doesn't belong + # in a denominator that claims to measure what compression had + # any power over. Tokens Headroom removed never reached the + # provider at all, so they're added back to form the baseline. + _pc_totals = prefix_cache_stats.get("totals", {}) + new_input_tokens = int(_pc_totals.get("uncached_input_tokens", 0) or 0) + int( + _pc_totals.get("cache_write_tokens", 0) or 0 + ) # Build human-readable summary summary = _build_session_summary( @@ -3340,6 +3353,27 @@ def create_app(config: ProxyConfig | None = None) -> FastAPI: else 0, 2, ), + # New-content-relative rate: savings as a fraction of the + # input that would have newly entered context (provider- + # billed uncached + cache-write tokens, plus the tokens + # compression removed before they could be billed). The + # whole-request ratios above recount the FULL transcript + # every turn, so a 200-turn session counts its history + # 200x into the denominator and long-running sessions + # (1M-context models never compact) read as ~0% no + # matter how well compression performs on new content. + # Guarded on new_input_tokens > 0 (not the full sum): the + # cache accumulators only see requests with cache + # activity, so a deployment with no cache metrics (e.g. + # Bedrock) would otherwise divide savings by themselves + # and report ~100%. No usage data -> report 0, not a lie. + "new_input_tokens": new_input_tokens, + "new_input_savings_percent": round( + (proxy_compression_tokens / (new_input_tokens + proxy_compression_tokens) * 100) + if new_input_tokens > 0 + else 0, + 2, + ), "savings_percent": round( (all_layers_tokens_saved / total_tokens_before * 100) if total_tokens_before > 0 diff --git a/tests/test_stats_new_input_savings_rate.py b/tests/test_stats_new_input_savings_rate.py new file mode 100644 index 000000000..f566a6d4a --- /dev/null +++ b/tests/test_stats_new_input_savings_rate.py @@ -0,0 +1,81 @@ +"""New-content-relative savings rate in /stats (tokens.new_input_savings_percent). + +The whole-request ratios recount the full transcript on every turn, so long +cached sessions dilute toward 0% regardless of how well compression performs +on content that newly enters context. The new rate divides by provider-billed +non-cache-read input (uncached + cache-write) plus the tokens compression +removed before they could be billed. +""" + +from __future__ import annotations + +import asyncio + +from fastapi.testclient import TestClient + +from headroom.proxy.server import ProxyConfig, create_app + + +def _make_client(tmp_path, monkeypatch) -> TestClient: + monkeypatch.setenv("HEADROOM_SAVINGS_PATH", str(tmp_path / "proxy_savings.json")) + config = ProxyConfig( + cache_enabled=False, + rate_limit_enabled=False, + log_requests=False, + ) + return TestClient(create_app(config)) + + +def test_stats_reports_new_input_savings_rate(tmp_path, monkeypatch): + with _make_client(tmp_path, monkeypatch) as client: + proxy = client.app.state.proxy + # A late turn of a long cached session: the local transcript recount + # (input_tokens) dwarfs what the provider newly billed (uncached + + # cache_write = 50k), so the whole-request ratio dilutes to ~0.5% + # while the new-content rate reports the undiluted 9.09%. + asyncio.run( + proxy.metrics.record_request( + provider="anthropic", + model="claude-opus-4-6", + input_tokens=1_000_000, + output_tokens=200, + tokens_saved=5_000, + latency_ms=10.0, + cache_read_tokens=900_000, + cache_write_tokens=45_000, + uncached_input_tokens=5_000, + ) + ) + + stats = client.get("/stats") + assert stats.status_code == 200 + tokens = stats.json()["tokens"] + + assert tokens["new_input_tokens"] == 50_000 + # 5_000 saved / (50_000 billed-new + 5_000 saved) = 9.09% + assert tokens["new_input_savings_percent"] == 9.09 + # The transcript-diluted ratio stays as-is — the new rate sits alongside, + # it does not replace existing fields. + assert tokens["proxy_savings_percent"] == 0.5 + + +def test_stats_new_input_rate_is_zero_without_cache_usage_data(tmp_path, monkeypatch): + with _make_client(tmp_path, monkeypatch) as client: + proxy = client.app.state.proxy + # Savings recorded but no cache usage observed (provider without + # cache metrics): the rate must report 0, not savings/savings=100%. + asyncio.run( + proxy.metrics.record_request( + provider="bedrock", + model="claude-opus-4-6", + input_tokens=10_000, + output_tokens=200, + tokens_saved=2_000, + latency_ms=10.0, + ) + ) + + tokens = client.get("/stats").json()["tokens"] + + assert tokens["new_input_tokens"] == 0 + assert tokens["new_input_savings_percent"] == 0