From 2b5ee7cde809ca37f6998d9679b1eb2133ab50ca Mon Sep 17 00:00:00 2001 From: Abhay Singh Date: Wed, 12 Aug 2026 10:37:51 +0530 Subject: [PATCH] fix(proxy/anthropic): None-guard usage token counts on the direct buffered path (#2434) ## Description The direct (non-backend) Anthropic buffered `/v1/messages` path reads token counts from the response usage to record metrics and update the prefix tracker: ```python usage = resp_json.get("usage", {}) output_tokens = usage.get("output_tokens", 0) cr_tokens = usage.get("cache_read_input_tokens", 0) cw_tokens = usage.get("cache_creation_input_tokens", 0) ... uncached_input_tokens = usage.get("input_tokens", 0) ``` `.get(key, default)` only falls back when the key is **absent**. When a key is present with a **null** value, `.get` returns `None`. The direct Anthropic API always sends integer usage, but this same handler serves any Anthropic-compatible upstream reached through a custom `ANTHROPIC_TARGET_API_URL` gateway (the scenario `install apply` now supports), and such a gateway can emit null counts on a stopped or empty turn. Those `None`s then reach `max(0, expected_cached - cr_tokens)` in the cache-bust block and the int-typed `RequestOutcome` / metrics recorder, so a single such response raises an uncaught `TypeError` and 502s the request. This is the same class as the Gemini crash fixed in #2347 and the OpenAI chat path. ## Fix Coerce the four counts with `int(... or 0)` at the direct-path usage-extraction site, matching `_extract_anthropic_cache_ttl_metrics` (which already guards its TTL buckets this way) and the Gemini fix. A normal integer usage is unchanged; only a null (or absent) value now becomes 0. ## Type of Change - [x] Bug fix (non-breaking change that fixes an issue) - [ ] New feature (non-breaking change that adds functionality) - [ ] Breaking change (fix or feature that would cause existing functionality to change) - [ ] Documentation update - [ ] Performance improvement - [ ] Code refactoring (no functional changes) ## Changes Made - `headroom/proxy/handlers/anthropic.py`: `int(... or 0)`-guard `output_tokens` / `cache_read_input_tokens` / `cache_creation_input_tokens` / `input_tokens` at the direct buffered-path usage-extraction site. - `tests/test_proxy/test_anthropic_buffered_timeout.py`: regression driving a buffered `/v1/messages` request whose upstream usage reports null counts, asserting a 200 instead of a 502. ## Testing - [x] Unit tests pass (`pytest`) - [x] Linting passes (`ruff check .`) - [x] Type checking passes (`mypy headroom`) - [x] New tests added for new functionality - [ ] Manual testing performed ### Test Output ```text $ python -m pytest tests/test_proxy/test_anthropic_buffered_timeout.py -q # all pass # with the fix reverted, the new test fails (the null-usage response 502s): $ git stash push -- headroom/proxy/handlers/anthropic.py $ python -m pytest "tests/test_proxy/test_anthropic_buffered_timeout.py::test_anthropic_messages_buffered_survives_null_usage_counts" -q 1 failed (TypeError: unsupported operand type(s) for +: 'NoneType' and 'NoneType') $ uvx ruff@0.15.17 check headroom/proxy/handlers/anthropic.py tests/test_proxy/test_anthropic_buffered_timeout.py All checks passed! ``` ## Real Behavior Proof - Environment: Windows 11, Python 3.12, project venv (`uv sync --extra proxy`), `uvx ruff@0.15.17` / `uvx mypy@1.20.2`, pytest in the venv. - Exact command / steps: ran the new FastAPI `TestClient` regression, which drives the real direct buffered `/v1/messages` handler with `proxy._retry_request` returning a 200 whose `usage` has null `input_tokens` / `output_tokens` / `cache_read_input_tokens` / `cache_creation_input_tokens`; then reverted only `anthropic.py` and re-ran. - Observed result: with the fix the request returns 200; with the fix reverted the same request 502s with `TypeError: unsupported operand type(s) for +: 'NoneType' and 'NoneType'`. Ran against the actual handler via the app. - Not tested: a live third-party Anthropic-compatible gateway emitting null usage. ## Review Readiness - [x] I have performed a self-review - [x] This PR is ready for human review ## Checklist - [x] My code follows the project's style guidelines - [x] I have performed a self-review of my code - [x] I have commented my code, particularly in hard-to-understand areas - [ ] I have made corresponding changes to the documentation - [x] My changes generate no new warnings - [x] I have added tests that prove my fix is effective or that my feature works - [x] New and existing unit tests pass locally with my changes - [ ] I have updated the CHANGELOG.md if applicable Co-authored-by: JerrettDavis --- headroom/proxy/handlers/anthropic.py | 8 +-- .../test_anthropic_buffered_timeout.py | 53 +++++++++++++++++++ 2 files changed, 57 insertions(+), 4 deletions(-) diff --git a/headroom/proxy/handlers/anthropic.py b/headroom/proxy/handlers/anthropic.py index 8a17311a5..d19413ac0 100644 --- a/headroom/proxy/handlers/anthropic.py +++ b/headroom/proxy/handlers/anthropic.py @@ -3576,13 +3576,13 @@ class AnthropicHandlerMixin: uncached_input_tokens = 0 if resp_json: usage = resp_json.get("usage", {}) - output_tokens = usage.get("output_tokens", 0) - cr_tokens = usage.get("cache_read_input_tokens", 0) - cw_tokens = usage.get("cache_creation_input_tokens", 0) + output_tokens = int(usage.get("output_tokens", 0) or 0) + cr_tokens = int(usage.get("cache_read_input_tokens", 0) or 0) + cw_tokens = int(usage.get("cache_creation_input_tokens", 0) or 0) cw_5m_tokens, cw_1h_tokens = self._extract_anthropic_cache_ttl_metrics( usage ) - uncached_input_tokens = usage.get("input_tokens", 0) + uncached_input_tokens = int(usage.get("input_tokens", 0) or 0) # Track cache bust: tokens that lost their cache discount due to compression. # If we had X tokens cached last turn and only Y hit cache this turn, diff --git a/tests/test_proxy/test_anthropic_buffered_timeout.py b/tests/test_proxy/test_anthropic_buffered_timeout.py index 69e44e7f5..6e0c753bb 100644 --- a/tests/test_proxy/test_anthropic_buffered_timeout.py +++ b/tests/test_proxy/test_anthropic_buffered_timeout.py @@ -110,6 +110,59 @@ def _anthropic_list_response() -> httpx.Response: ) +def _anthropic_null_usage_response() -> dict[str, object]: + """An Anthropic Messages response whose usage counts are present but null. + + An Anthropic-compatible gateway (custom ANTHROPIC_TARGET_API_URL) can emit + this shape on a stopped or empty turn — the same class that crashed the + Gemini path in #2347. + """ + return { + "id": "msg_test_null", + "type": "message", + "role": "assistant", + "content": [{"type": "text", "text": "ok"}], + "usage": { + "input_tokens": None, + "output_tokens": None, + "cache_read_input_tokens": None, + "cache_creation_input_tokens": None, + }, + } + + +def test_anthropic_messages_buffered_survives_null_usage_counts(): + config = _make_config() + app = create_app(config) + with TestClient(app) as client: + proxy = client.app.state.proxy + # Use the real prefix tracker (its `_cached_token_count` is read on this + # path); only pin the session id so the tracker resolves deterministically. + proxy.session_tracker_store.compute_session_id = lambda request, model, messages: "s1" + + async def _fake_retry(method, url, headers, body, stream=False, **kwargs): # noqa: ANN001 + return httpx.Response(200, json=_anthropic_null_usage_response()) + + proxy._retry_request = _fake_retry # type: ignore[assignment] + + response = client.post( + "/v1/messages", + headers={ + "x-api-key": "test-key", + "anthropic-version": "2023-06-01", + "content-type": "application/json", + }, + json={ + "model": "claude-sonnet-4-6", + "max_tokens": 64, + "messages": [{"role": "user", "content": "hello"}], + }, + ) + + # Null usage counts must not crash outcome recording / cache-bust math. + assert response.status_code == 200, response.text + + def test_anthropic_messages_buffered_timeout_override_reaches_retry_request(): config = _make_config() app = create_app(config)