From 1448718fcadcc6cde474719105289924155dce7c Mon Sep 17 00:00:00 2001 From: GaryOcean <81794144+GaryOcean428@users.noreply.github.com> Date: Tue, 14 Jul 2026 02:01:48 +0800 Subject: [PATCH] fix: strip output-only fallback blocks from request messages (#1870) Anthropic's server-side refusal-fallback feature (`server-side-fallback-2026-06-01`) can emit an output-only `{ "type": "fallback", ... }` block inside an assistant response. That block is valid on the response path but rejected on the request path, so when a client replays the assistant turn the next request can 400 with an invalid `fallback` input tag. This strips output-only request blocks in the shared request-body readers before forwarding. `read_request_json_with_bytes` re-encodes raw bytes only when stripping occurs, so byte-faithful passthrough paths cannot leak the invalid block while clean requests keep their original bytes. If stripping empties an assistant content array, the helper backfills a benign text block so the request remains schema-valid. ## Type of Change - [x] Bug fix ## Changes Made - Add `strip_output_only_request_blocks` in `headroom/proxy/helpers.py`. - Strip fallback blocks in both `_read_request_json` and `read_request_json_with_bytes`. - Add regression tests for stripping, empty-turn backfill, raw-byte re-encoding, and byte-identical clean requests. - Update `CHANGELOG.md`. ## Testing Focused verification on the current PR head after merging `main`: ```text python -m pytest tests/test_output_only_request_blocks.py -q 4 passed uvx ruff==0.15.17 check headroom/proxy/helpers.py tests/test_output_only_request_blocks.py All checks passed! uvx ruff==0.15.17 format --check headroom/proxy/helpers.py tests/test_output_only_request_blocks.py 2 files already formatted ``` ## Notes No live Anthropic request was run; the regression is covered at the shared request-reader layer used by the proxy forwarding paths. Co-authored-by: Claude Opus 4.8 (1M context) Co-authored-by: JerrettDavis Co-authored-by: Tejas Chopra --- CHANGELOG.md | 1 + headroom/proxy/helpers.py | 81 ++++++++++++++++++ tests/test_output_only_request_blocks.py | 102 +++++++++++++++++++++++ 3 files changed, 184 insertions(+) create mode 100644 tests/test_output_only_request_blocks.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 65fd0c277..30ea55978 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -102,6 +102,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ### Bug Fixes +* **proxy:** strip output-only content blocks from request messages before forwarding. Anthropic's server-side refusal-fallback feature (`server-side-fallback-2026-06-01`) emits a `{"type":"fallback","from":{...},"to":{...}}` block inside the assistant response to signal that a refused request was re-served by the fallback model. That block is valid on the *response* path but rejected on the *request* path, so when a client replays the assistant turn the next request 400s (`invalid_request_error: messages.N.content.0: Input tag 'fallback' ...`) and the conversation gets permanently stuck through the proxy. `read_request_json_with_bytes` (Anthropic/OpenAI/Bedrock) and `_read_request_json` (Gemini) now drop such blocks — re-encoding the raw bytes so byte-faithful passthrough cannot leak the pre-strip body, backfilling a benign text block if a turn is emptied, and leaving requests without such blocks byte-identical (no cache churn). * **wrap/doctor:** make the Claude Remote Control gate warning accurate and stop it firing for users who never had the feature ([#1779](https://github.com/headroomlabs-ai/headroom/issues/1779)). Claude Code 2.1.196 added a client-side check that **deterministically** disables first-party Remote Control (`/remote-control` / `/rc`) whenever `ANTHROPIC_BASE_URL` points at a non-`api.anthropic.com` host — which Headroom always does. The old notice hedged ("may hide the Remote Control menu"); it now states the disable as fact, names the `/rc` command, and detects the installed Claude Code version so the wording is exact (`2.1.196` when known, `2.1.196+` when not). The gate is upstream and RC's control-plane talks to `claude.ai` (not the API host), so Headroom cannot restore it — the warning tells you to run Claude without Headroom for RC sessions. The warning is suppressed for auth modes that never had Remote Control (API-key/PAYG via `ANTHROPIC_API_KEY`/`ANTHROPIC_AUTH_TOKEN`, and Bedrock/Vertex/Foundry cloud IAM) and on Claude Code builds older than 2.1.196 where RC is unaffected by a custom base URL. Both the `headroom wrap claude` launch banner and `headroom doctor` co-report the sibling base-URL gates Headroom *does* restore — on-demand tool loading (#746, automatic) and the 1M context window (#1158, via `--1m`) — and the wrap-side co-report is session-accurate: it says "already restored via --1m" when the flag is in effect and reports tool deferral OFF (not falsely "kept on") when the user chose `--tool-search false`/`ENABLE_TOOL_SEARCH=false`; the `ENABLE_TOOL_SEARCH=...` banner line got the same accuracy fix. `is_custom_anthropic_base_url` now recognizes scheme-less values (`myproxy.local:8080`, `127.0.0.1:8787`) as custom hosts and degrades gracefully on malformed URLs instead of crashing `doctor`. `doctor` resolves the Claude Code version lazily, so runs with no custom base URL never pay the `claude --version` subprocess. No request bytes are touched (cache-safe); this is UX/notice-only. * **install:** default the docker image to `ghcr.io/headroomlabs-ai/headroom:latest` instead of the dead `ghcr.io/chopratejas/headroom:latest`. After the repo moved to the `headroomlabs-ai` org, GHCR did not redirect the old package, so `headroom install` / `headroom init` and the install scripts pulled a frozen `0.27.0` image while current releases publish to the new path ([#1867](https://github.com/headroomlabs-ai/headroom/issues/1867)). * **transforms/content-router:** stop a profile-derived `read_protection_window` kwarg from weakening an explicit `--protect-tool-results` guarantee. `ContentRouter.apply()` computes `read_protection_window` from `protect_recent_reads_fraction`, where `0.0` (the sentinel `--protect-tool-results` sets) means "protect all excluded-tool output regardless of conversation depth" per #1374's documented contract — but the method then unconditionally overwrote that window with a `read_protection_window` kwarg whenever one was present. `proxy_pipeline_kwargs()` supplies that kwarg on every request from the active `AgentSavingsProfile.protect_recent` (the default `coding` profile sets `protect_recent=2`), so in practice only the last 2 messages ever kept read-protection and older excluded-tool output silently fell through to lossy compression. The runtime kwarg may now only narrow the window when `protect_recent_reads_fraction > 0`; it can no longer shrink the "protect everything" guarantee set by `--protect-tool-results`. diff --git a/headroom/proxy/helpers.py b/headroom/proxy/helpers.py index 1eb41c124..eb9e26e13 100644 --- a/headroom/proxy/helpers.py +++ b/headroom/proxy/helpers.py @@ -2379,6 +2379,61 @@ async def _read_request_body_bytes(request: Request) -> bytes: return cast(bytes, raw) +# --------------------------------------------------------------------------- +# Output-only content blocks +# --------------------------------------------------------------------------- +# The Anthropic *response* schema can emit signaling blocks that the *request* +# schema (messages[].content[]) does not accept. The primary case is the +# server-side refusal fallback notification introduced with the +# ``server-side-fallback-2026-06-01`` beta:: +# +# {"type": "fallback", +# "from": {"model": "claude-fable-5"}, +# "to": {"model": "claude-opus-4-8"}} +# +# The API returns it inside an assistant turn to signal that a refused request +# was transparently re-served by the fallback model. When a client replays that +# assistant turn on the next call, the request validator rejects it:: +# +# 400 invalid_request_error: messages.N.content.0: Input tag 'fallback' +# found using 'type' does not match any of the expected tags +# +# These blocks are output-only and carry no state the model needs on input, so +# they are safe to drop before forwarding. +OUTPUT_ONLY_REQUEST_BLOCK_TYPES: frozenset[str] = frozenset({"fallback"}) + + +def strip_output_only_request_blocks(messages: Any) -> bool: + """Remove output-only content blocks from request ``messages`` in place. + + Returns ``True`` if any block was removed. If stripping empties a message's + ``content`` list it is backfilled with a single benign text block, because + the API also rejects an empty ``content`` array. Idempotent. + """ + if not isinstance(messages, list): + return False + changed = False + for msg in messages: + if not isinstance(msg, dict): + continue + content = msg.get("content") + if not isinstance(content, list): + continue + kept = [ + block + for block in content + if not ( + isinstance(block, dict) and block.get("type") in OUTPUT_ONLY_REQUEST_BLOCK_TYPES + ) + ] + if len(kept) != len(content): + changed = True + if not kept: + kept = [{"type": "text", "text": "(model fallback)"}] + msg["content"] = kept + return changed + + async def _read_request_json(request: Request) -> dict[str, Any]: """Read and parse JSON from a request, handling compressed bodies. @@ -2401,6 +2456,17 @@ async def _read_request_json(request: Request) -> dict[str, Any]: result = json.loads(text) if not isinstance(result, dict): raise ValueError("Request body must be a JSON object, not " + type(result).__name__) + + # Drop output-only blocks the request schema rejects (see + # ``strip_output_only_request_blocks``). Callers of this bytes-less reader + # (e.g. the Gemini path) re-serialize ``result`` themselves. + if strip_output_only_request_blocks(result.get("messages")): + logger.warning( + "removed output-only content block(s) (%s) from request messages " + "before forwarding (not valid on the request path)", + ",".join(sorted(OUTPUT_ONLY_REQUEST_BLOCK_TYPES)), + ) + return result @@ -2422,6 +2488,21 @@ async def read_request_json_with_bytes(request: Request) -> tuple[dict[str, Any] result = json.loads(text) if not isinstance(result, dict): raise ValueError("Request body must be a JSON object, not " + type(result).__name__) + + # Drop output-only blocks (see ``strip_output_only_request_blocks``) before + # any downstream deepcopy / compression / 400-retry path. This is the shared + # reader for the Anthropic, OpenAI, and Bedrock handlers, so one guard here + # covers every client that routes through the proxy. When a block is removed + # we re-encode ``raw`` so a byte-faithful passthrough forwarder cannot leak + # the pre-strip body; unchanged requests keep their exact original bytes. + if strip_output_only_request_blocks(result.get("messages")): + raw = json.dumps(result, ensure_ascii=False).encode("utf-8") + logger.warning( + "removed output-only content block(s) (%s) from request messages " + "before forwarding (not valid on the request path)", + ",".join(sorted(OUTPUT_ONLY_REQUEST_BLOCK_TYPES)), + ) + return result, raw diff --git a/tests/test_output_only_request_blocks.py b/tests/test_output_only_request_blocks.py new file mode 100644 index 000000000..2f624f893 --- /dev/null +++ b/tests/test_output_only_request_blocks.py @@ -0,0 +1,102 @@ +"""Output-only content blocks must be stripped from request messages. + +Anthropic's server-side refusal-fallback feature emits an output-only +``{"type": "fallback", ...}`` block inside an assistant response. It is valid on +the response path but rejected on the request path, so replaying that assistant +turn 400s the whole request. The shared body readers must drop it before +forwarding. See ``strip_output_only_request_blocks`` in ``headroom.proxy.helpers``. +""" + +import asyncio +import json + +from headroom.proxy.helpers import ( + read_request_json_with_bytes, + strip_output_only_request_blocks, +) + +_FALLBACK = { + "type": "fallback", + "from": {"model": "claude-fable-5"}, + "to": {"model": "claude-opus-4-8"}, +} + + +class _FakeHeaders: + def __init__(self, d=None): + self._d = {k.lower(): v for k, v in (d or {}).items()} + + def get(self, k, default=None): + return self._d.get(k.lower(), default) + + +class _FakeRequest: + def __init__(self, raw, headers=None): + self._raw = raw + self.headers = _FakeHeaders(headers) + + async def body(self): + return self._raw + + +def _has_fallback(messages): + for msg in messages: + content = msg.get("content") + if isinstance(content, list): + for block in content: + if isinstance(block, dict) and block.get("type") == "fallback": + return True + return False + + +def test_strip_removes_fallback_and_backfills_emptied_turn(): + messages = [ + {"role": "user", "content": "hi"}, + # assistant turn that is ONLY a fallback signal (the crash case) + {"role": "assistant", "content": [dict(_FALLBACK)]}, + # fallback prefix + real content + {"role": "assistant", "content": [dict(_FALLBACK), {"type": "text", "text": "A."}]}, + ] + assert strip_output_only_request_blocks(messages) is True + assert not _has_fallback(messages) + # emptied turn is backfilled with a single benign text block + assert messages[1]["content"] == [{"type": "text", "text": "(model fallback)"}] + # mixed turn keeps only the real content + assert [b["type"] for b in messages[2]["content"]] == ["text"] + # idempotent + assert strip_output_only_request_blocks(messages) is False + + +def test_strip_is_noop_on_clean_or_invalid_input(): + assert strip_output_only_request_blocks(None) is False + assert strip_output_only_request_blocks([{"role": "user", "content": "hi"}]) is False + assert ( + strip_output_only_request_blocks( + [{"role": "user", "content": [{"type": "text", "text": "x"}]}] + ) + is False + ) + + +def test_reader_strips_and_reencodes_raw_bytes(): + body = { + "model": "claude-fable-5", + "messages": [ + {"role": "user", "content": "hi"}, + {"role": "assistant", "content": [dict(_FALLBACK)]}, + ], + } + raw = json.dumps(body).encode("utf-8") + result, out_raw = asyncio.run(read_request_json_with_bytes(_FakeRequest(raw))) + assert not _has_fallback(result["messages"]) + # raw bytes re-encoded so byte-faithful passthrough cannot leak the pre-strip body + assert not _has_fallback(json.loads(out_raw)["messages"]) + assert json.loads(out_raw) == result + + +def test_reader_leaves_clean_requests_byte_identical(): + raw = json.dumps({"model": "x", "messages": [{"role": "user", "content": "hi"}]}).encode( + "utf-8" + ) + _, out_raw = asyncio.run(read_request_json_with_bytes(_FakeRequest(raw))) + assert out_raw == raw