diff --git a/CHANGELOG.md b/CHANGELOG.md index 2c2fceb9d..0f59388e8 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -102,6 +102,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ### Bug Fixes * **ccr:** don't crash `parse_tool_call` on a CCR tool call whose arguments aren't an object. For the OpenAI/`openai_responses` shape the arguments are `json.loads`-decoded and only `JSONDecodeError` was caught, so a model that emitted `arguments='[]'`/`'"abc"'`/`'123'` (decoding to a list/str/number) — or a non-dict Anthropic `input` — reached `input_data.get("hash")` and raised `AttributeError`; a null `arguments` raised an uncaught `TypeError` from `json.loads(None)`. Both are now handled: the decode also catches `TypeError`, and a non-dict `input_data` returns `None` (not a valid CCR call) instead of crashing CCR response processing. +* **proxy/vertex:** route Vertex `publisher=google` (Gemini) requests to the region matching the request path. `vertex_generate_content`, `vertex_stream_generate_content`, and `vertex_count_tokens` discarded the path's `location` and forwarded to the single fixed host from `_api_target(proxy, "vertex")` (default `us-central1`), instead of the region-aware `_vertex_target_for_location` the sibling Anthropic `rawPredict` route already uses. So a request to `.../locations/europe-west1/publishers/google/...` was sent to a `us-central1` host, which Vertex rejects on the region/host mismatch. The three google routes now derive the host from the request's `location` (operator-pinned upstreams are still honored). * **proxy/anthropic:** give each Anthropic conversation its own session id. `SessionTrackerStore.compute_session_id` derived its fallback id from `model` + system text harvested only from `role:"system"` entries inside `messages` — but Anthropic carries the system prompt as a top-level `body["system"]` field, so genuine Anthropic requests (which never carry `x-headroom-session-id`) collapsed to `md5(model:[])` and every conversation on the same model shared one `PrefixCacheTracker`. That let session-sticky state cross-contaminate: conversation A's sticky `headroom_retrieve`/memory tools and `anthropic-beta` headers were injected into conversation B, and frozen-prefix/compression-cache state mixed across conversations. The Anthropic handler now folds the top-level `system` into the session-id inputs (prepending a synthetic `role:"system"` message used only to derive the id), giving distinct conversations distinct ids. * **cache/semantic:** key entries by the full-context hash, not the trailing query text. `SemanticCache.put` stored each response under `sha256(query)[:16]` where `query` is only the last user message, and the exact-match branch of `get` returned the slot without checking the stored entry's `messages_hash`. Two requests that share a trailing message ("continue", "yes", "run the tests") but differ in earlier context therefore collided on one slot — the second overwrote the first, and the first's hash then resolved to the second's cached response (wrong data served). Entries are now keyed by `messages_hash` when present, and `get` verifies `entry.messages_hash` before returning. * **proxy/openai:** stop overriding an explicit client `stream_options.include_usage` on the streaming chat path. To count tokens from the trailing usage chunk, the handler set `include_usage: True` unconditionally — including flipping an explicit client `false` to `true`. The upstream then appended a usage-only chunk (`choices: []`) the client never requested, and the common `chunk.choices[0].delta` loop raised `IndexError`. The option is now only filled in when the client left the choice open (no `stream_options`, or a dict without `include_usage`); an explicit `true`/`false` is respected. diff --git a/headroom/providers/proxy_routes.py b/headroom/providers/proxy_routes.py index 04a56613c..002c3661d 100644 --- a/headroom/providers/proxy_routes.py +++ b/headroom/providers/proxy_routes.py @@ -254,12 +254,12 @@ def register_provider_routes(app: FastAPI, proxy: Any) -> None: publisher: str, model: str, ): - del api_version, project, location + del api_version, project if is_vertex_google_publisher(publisher): return await proxy.handle_gemini_generate_content( request, model, - _api_target(proxy, "vertex"), + _vertex_target_for_location(proxy, location), VERTEX_GOOGLE_PROVIDER_NAME, ) return await vertex_publisher_passthrough(request, publisher, VERTEX_GENERATE_CONTENT.name) @@ -275,12 +275,12 @@ def register_provider_routes(app: FastAPI, proxy: Any) -> None: publisher: str, model: str, ): - del api_version, project, location + del api_version, project if is_vertex_google_publisher(publisher): return await proxy.handle_gemini_generate_content( request, model, - _api_target(proxy, "vertex"), + _vertex_target_for_location(proxy, location), VERTEX_GOOGLE_PROVIDER_NAME, ) return await vertex_publisher_passthrough( @@ -300,12 +300,12 @@ def register_provider_routes(app: FastAPI, proxy: Any) -> None: publisher: str, model: str, ): - del api_version, project, location + del api_version, project if is_vertex_google_publisher(publisher): return await proxy.handle_gemini_count_tokens( request, model, - _api_target(proxy, "vertex"), + _vertex_target_for_location(proxy, location), VERTEX_GOOGLE_PROVIDER_NAME, ) return await vertex_publisher_passthrough(request, publisher, VERTEX_COUNT_TOKENS.name) diff --git a/tests/test_vertex_claude_compression.py b/tests/test_vertex_claude_compression.py index a778d1353..6f31d38a5 100644 --- a/tests/test_vertex_claude_compression.py +++ b/tests/test_vertex_claude_compression.py @@ -142,6 +142,52 @@ def test_vertex_rawpredict_anthropic_runs_compression_handler(monkeypatch) -> No assert captured["model"] == "claude-sonnet-4-6" +def test_vertex_google_generate_content_uses_region_derived_host(monkeypatch) -> None: + """The google-publisher generateContent route must derive the upstream host + from the request's location (like the anthropic route), not send a + europe-west1 request to the fixed us-central1 host.""" + captured: dict[str, str] = {} + + # The route calls handle_gemini_generate_content(request, model, base_url, provider). + async def fake(self, request, model, base_url, provider, *rest): # type: ignore[no-untyped-def] + captured.update(base_url=str(base_url), provider=str(provider), model=str(model)) + return JSONResponse({"ok": True}) + + monkeypatch.setattr(HeadroomProxy, "handle_gemini_generate_content", fake) + + with TestClient(_default_vertex_app()) as client: + resp = client.post( + "/v1/projects/p/locations/europe-west1/publishers/google/models/" + "gemini-2.0-flash:generateContent", + json={"contents": []}, + ) + assert resp.status_code == 200 + assert captured["provider"] == "vertex:google" + assert captured["base_url"] == "https://europe-west1-aiplatform.googleapis.com" + assert captured["model"] == "gemini-2.0-flash" + + +def test_vertex_google_count_tokens_uses_region_derived_host(monkeypatch) -> None: + """The google-publisher countTokens route is region-aware too.""" + captured: dict[str, str] = {} + + async def fake(self, request, model, base_url, provider, *rest): # type: ignore[no-untyped-def] + captured.update(base_url=str(base_url), provider=str(provider)) + return JSONResponse({"ok": True}) + + monkeypatch.setattr(HeadroomProxy, "handle_gemini_count_tokens", fake) + + with TestClient(_default_vertex_app()) as client: + resp = client.post( + "/v1/projects/p/locations/europe-west1/publishers/google/models/" + "gemini-2.0-flash:countTokens", + json={"contents": []}, + ) + assert resp.status_code == 200 + assert captured["provider"] == "vertex:google" + assert captured["base_url"] == "https://europe-west1-aiplatform.googleapis.com" + + def test_vertex_rawpredict_versionless_anthropic_rewrites_to_v1(monkeypatch) -> None: captured: dict[str, Any] = {}