From 4cbd5da67328d03078bd9c1fbbf10fe738cc0b2b Mon Sep 17 00:00:00 2001 From: Manmit Singh Date: Thu, 16 Jul 2026 01:28:34 +0530 Subject: [PATCH] feat(proxy): opt-in compression for catch-all passthrough routes (#1699) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ## Description Requests whose path doesn't match a built-in API route fall through to `handle_passthrough`, which forwarded the body verbatim — bypassing ContentRouter/Kompress/TOIN entirely. Wrapper-proxy setups that front Headroom on custom paths (e.g. `/api/codex-proxy//v1/responses`) got zero compression on coding-agent traffic and hit context-limit 400s in long sessions. This adds an opt-in flag that routes OpenAI Responses-shaped passthrough bodies through the same compression path the native `/v1/responses` handler uses. Closes #1546 ## Type of Change - [ ] Bug fix (non-breaking change that fixes an issue) - [x] New feature (non-breaking change that adds functionality) - [ ] Breaking change (fix or feature that would cause existing functionality to change) - [ ] Documentation update - [ ] Performance improvement - [ ] Code refactoring (no functional changes) ## Changes Made - Added `ProxyConfig.compress_passthrough` (default `False`) + `--compress-passthrough` CLI flag + `HEADROOM_COMPRESS_PASSTHROUGH=1` env. - `handle_passthrough`: when enabled, POST requests whose path ends in `/responses` with an OpenAI Responses-shaped body are compressed via the existing `_compress_openai_responses_payload_in_executor` before forwarding; stale `Content-Length` is dropped so httpx recomputes it. - New `_maybe_compress_passthrough_responses` helper — fail-open: non-JSON, non-Responses payloads, unmodified results, and any compressor error forward the original body unchanged. - Documented the flag in `docs/content/docs/proxy.mdx`. ## Testing - [x] Unit tests pass (`pytest`) - [x] Linting passes (`ruff check .`) - [ ] Type checking passes (`mypy headroom`) - [x] New tests added for new functionality - [x] Manual testing performed ### Test Output ```text $ .venv/bin/python -m pytest tests/test_compress_passthrough.py -q collected 6 items tests/test_compress_passthrough.py ...... [100%] ============================== 6 passed in 0.35s =============================== $ .venv/bin/ruff check headroom/proxy/handlers/openai.py headroom/proxy/models.py headroom/proxy/server.py tests/test_compress_passthrough.py All checks passed! ``` ## Real Behavior Proof - Environment: macOS arm64, Python 3.14, repo `.venv`. - Exact command / steps: `.venv/bin/python -m pytest tests/test_compress_passthrough.py -q` — covers a Responses-shaped body being compressed, non-JSON passthrough, non-Responses (`messages`) payload untouched, unmodified-result short-circuit, compressor-error fail-open, and `ProxyConfig().compress_passthrough is False` default. Plus import smoke: `ProxyConfig(compress_passthrough=True)`, server/handler modules import, helper present. - Observed result: 6 passed; flag defaults off; enabled path reuses the native Responses compressor and never raises out to the request. - Not tested: live end-to-end through a real second proxy to a real upstream (no external wrapper proxy / upstream credentials in sandbox); the compression call is the same one `/v1/responses` already exercises in CI. ## Review Readiness - [x] I have performed a self-review - [x] This PR is ready for human review ## Checklist - [x] My code follows the project's style guidelines - [x] I have performed a self-review of my code - [x] I have commented my code, particularly in hard-to-understand areas - [x] I have made corresponding changes to the documentation - [x] My changes generate no new warnings - [x] I have added tests that prove my fix is effective or that my feature works - [x] New and existing unit tests pass locally with my changes - [ ] I have updated the CHANGELOG.md if applicable ## Additional Notes Scoped to OpenAI Responses-shaped bodies (the reporter's exact case). Anthropic `/messages` and OpenAI `/chat/completions` passthrough compression are natural follow-ups — deliberately left out to keep this change focused and fail-safe. CHANGELOG is release-managed, left unchecked. --- docs/content/docs/proxy.mdx | 2 + headroom/proxy/handlers/openai.py | 66 +++++++++++++++++ headroom/proxy/models.py | 10 +++ headroom/proxy/server.py | 17 +++++ tests/test_compress_passthrough.py | 109 +++++++++++++++++++++++++++++ 5 files changed, 204 insertions(+) create mode 100644 tests/test_compress_passthrough.py diff --git a/docs/content/docs/proxy.mdx b/docs/content/docs/proxy.mdx index 50c21b703..c19861be5 100644 --- a/docs/content/docs/proxy.mdx +++ b/docs/content/docs/proxy.mdx @@ -124,11 +124,13 @@ HEADROOM_SAVINGS_PROFILE=agent-90 headroom proxy --port 8787 | `--no-learn` | `false` | Explicitly disable traffic learning | | `--min-evidence` | `5` | Minimum observations before a learned pattern is persisted | | `--codex-wire-debug` | `false` | Write local Codex wire snapshots and matching proxy log traces | +| `--compress-passthrough` | `false` | Also compress custom proxy paths that fall through to the catch-all handler (OpenAI Responses-shaped bodies, path ends in `/responses`). Also `HEADROOM_COMPRESS_PASSTHROUGH=1` | ```bash headroom proxy --memory headroom proxy --learn --min-evidence 3 headroom proxy --codex-wire-debug +headroom proxy --compress-passthrough ``` diff --git a/headroom/proxy/handlers/openai.py b/headroom/proxy/handlers/openai.py index a97d20f63..a68d1f475 100644 --- a/headroom/proxy/handlers/openai.py +++ b/headroom/proxy/handlers/openai.py @@ -7697,6 +7697,52 @@ class OpenAIHandlerMixin: }, ) + async def _maybe_compress_passthrough_responses(self, body: bytes) -> bytes: + """Compress an OpenAI Responses-shaped passthrough body, fail-open. + + Reuses the native `/v1/responses` compression path so custom + wrapper-proxy routes get the same ContentRouter/Kompress treatment. + Any parse/compression failure returns the original body unchanged so a + catch-all request is never dropped by opting into compression. + """ + try: + payload = json.loads(body) + except (json.JSONDecodeError, ValueError, TypeError): + return body + if not isinstance(payload, dict) or "input" not in payload: + # Not a Responses payload (no `input` array) — leave it alone. + return body + + model = str(payload.get("model") or "passthrough") + request_id = await self._next_request_id() + try: + ( + compressed_payload, + modified, + *_rest, + ) = await self._compress_openai_responses_payload_in_executor( + payload, + model=model, + request_id=request_id, + ) + except Exception as exc: # noqa: BLE001 — fail-open on any compressor error + logger.warning( + "[%s] passthrough Responses compression failed, forwarding verbatim: %s", + request_id, + exc, + ) + return body + if not modified: + return body + try: + return json.dumps( + compressed_payload, + separators=(",", ":"), + ensure_ascii=False, + ).encode("utf-8") + except (TypeError, ValueError): + return body + async def handle_passthrough( self, request: Request, @@ -7780,6 +7826,26 @@ class OpenAIHandlerMixin: if body is not original_body: headers["content-length"] = str(len(body)) + # Opt-in: compress requests that fall through here because their path + # doesn't match a built-in API route (custom wrapper-proxy paths like + # `/api/codex-proxy//v1/responses`). Off by default; only touches + # OpenAI Responses-shaped bodies (path ends in `/responses`) so we reuse + # the exact same ContentRouter/Kompress path the native handler runs. + _pt_config = getattr(self, "config", None) + if ( + getattr(_pt_config, "compress_passthrough", False) + and getattr(_pt_config, "optimize", False) + and request.method == "POST" + and path.rstrip("/").endswith("/responses") + and body + ): + compressed = await self._maybe_compress_passthrough_responses(body) + if compressed != body: + body = compressed + # Body size changed — let httpx recompute Content-Length. + for _hk in [k for k in headers if k.lower() == "content-length"]: + headers.pop(_hk, None) + headers = await apply_copilot_api_auth(headers, url=url) # Cloudflare bot-management challenges our HTTP/2 fingerprint on # ChatGPT's sensitive account endpoints (/backend-api/me, diff --git a/headroom/proxy/models.py b/headroom/proxy/models.py index 6836c3483..d7d16e5b5 100644 --- a/headroom/proxy/models.py +++ b/headroom/proxy/models.py @@ -222,6 +222,16 @@ class ProxyConfig: lossless: bool = False # CLI: --lossless; env: HEADROOM_LOSSLESS=1. No-CCR mode: compress without any retrieval marker. + # Compress requests that fall through to the catch-all passthrough handler + # (custom proxy paths that don't match a built-in API route, e.g. + # `/api/codex-proxy//v1/responses` fronted by another proxy). Off by + # default because passthrough targets are unknown upstreams; opt-in for + # wrapper-proxy architectures that need coding-agent traffic compressed. + # Currently applies to OpenAI Responses-shaped bodies (paths ending in + # `/responses`). CLI: --compress-passthrough; env: + # HEADROOM_COMPRESS_PASSTHROUGH=1. + compress_passthrough: bool = False + # Code graph live watcher (triggers incremental reindex on file changes) code_graph_watcher: bool = False diff --git a/headroom/proxy/server.py b/headroom/proxy/server.py index c58cfa900..6c5511d46 100644 --- a/headroom/proxy/server.py +++ b/headroom/proxy/server.py @@ -4677,6 +4677,7 @@ def _proxy_config_from_env() -> ProxyConfig: disable_kompress_openai=_get_env_optional_bool("HEADROOM_DISABLE_KOMPRESS_OPENAI"), force_kompress_all=_get_env_bool("HEADROOM_FORCE_KOMPRESS_ALL", False), lossless=_get_env_bool("HEADROOM_LOSSLESS", False), + compress_passthrough=_get_env_bool("HEADROOM_COMPRESS_PASSTHROUGH", False), max_connections=_get_env_int("HEADROOM_MAX_CONNECTIONS", 500), max_keepalive_connections=_get_env_int("HEADROOM_MAX_KEEPALIVE", 100), keepalive_expiry=_get_env_float("HEADROOM_KEEPALIVE_EXPIRY", 90.0), @@ -5230,6 +5231,18 @@ if __name__ == "__main__": "is needed. Also settable via HEADROOM_LOSSLESS=1." ), ) + parser.add_argument( + "--compress-passthrough", + action="store_true", + help=( + "Also compress requests that fall through to the catch-all " + "passthrough handler (custom proxy paths not matched by a built-in " + "API route, e.g. `/api/codex-proxy//v1/responses` behind " + "another proxy). Applies to OpenAI Responses-shaped bodies (paths " + "ending in `/responses`). Off by default; also settable via " + "HEADROOM_COMPRESS_PASSTHROUGH=1." + ), + ) parser.add_argument( "--exclude-tools", default=None, @@ -5312,6 +5325,9 @@ if __name__ == "__main__": "HEADROOM_FORCE_KOMPRESS_ALL", False ) lossless = getattr(args, "lossless", False) or _get_env_bool("HEADROOM_LOSSLESS", False) + compress_passthrough = args.compress_passthrough or _get_env_bool( + "HEADROOM_COMPRESS_PASSTHROUGH", False + ) # Set OpenRouter API key from CLI if provided if hasattr(args, "openrouter_api_key") and args.openrouter_api_key: @@ -5368,6 +5384,7 @@ if __name__ == "__main__": disable_kompress_openai=disable_kompress_openai, force_kompress_all=force_kompress_all, lossless=lossless, + compress_passthrough=compress_passthrough, # Connection pool settings max_connections=_get_env_int("HEADROOM_MAX_CONNECTIONS", args.max_connections), max_keepalive_connections=_get_env_int("HEADROOM_MAX_KEEPALIVE", args.max_keepalive), diff --git a/tests/test_compress_passthrough.py b/tests/test_compress_passthrough.py new file mode 100644 index 000000000..a6d1c2a70 --- /dev/null +++ b/tests/test_compress_passthrough.py @@ -0,0 +1,109 @@ +"""Tests for opt-in passthrough compression (issue #1546). + +Requests whose path doesn't match a built-in API route fall through to +``handle_passthrough``, which historically forwarded the body verbatim — no +compression. With ``compress_passthrough`` enabled, OpenAI Responses-shaped +bodies (path ends in ``/responses``) are routed through the same +ContentRouter/Kompress path the native ``/v1/responses`` handler uses. + +``_maybe_compress_passthrough_responses`` is the fail-open core: any parse or +compressor failure returns the original body so a catch-all request is never +dropped by opting into compression. +""" + +from __future__ import annotations + +import json +from types import SimpleNamespace + +from headroom.proxy.handlers.openai import OpenAIHandlerMixin + + +def _make_handler(compress_impl): + """Bare mixin instance with just the two collaborators the helper needs.""" + handler = OpenAIHandlerMixin.__new__(OpenAIHandlerMixin) + handler.config = SimpleNamespace(optimize=True, compress_passthrough=True) + + async def _next_request_id(): + return "req-test" + + handler._next_request_id = _next_request_id + handler._compress_openai_responses_payload_in_executor = compress_impl + return handler + + +def _shrinking_compressor(marker: str = "[C]"): + async def _impl(payload, *, model, request_id): + new = dict(payload) + new["input"] = marker + return (new, True, 5, ["kompress"], None, 100, 40, 5, {}) + + return _impl + + +async def test_compresses_responses_shaped_body() -> None: + handler = _make_handler(_shrinking_compressor()) + body = json.dumps({"model": "gpt-5.4", "input": [{"role": "user"}]}).encode() + + out = await handler._maybe_compress_passthrough_responses(body) + + assert out != body + assert json.loads(out)["input"] == "[C]" + + +async def test_non_json_body_passes_through() -> None: + handler = _make_handler(_shrinking_compressor()) + body = b"not json at all" + + assert await handler._maybe_compress_passthrough_responses(body) == body + + +async def test_non_responses_payload_passes_through() -> None: + # No `input` key → not a Responses payload; must not be touched. + handler = _make_handler(_shrinking_compressor()) + body = json.dumps({"model": "gpt-5.4", "messages": []}).encode() + + assert await handler._maybe_compress_passthrough_responses(body) == body + + +async def test_unmodified_result_returns_original_bytes() -> None: + async def _noop(payload, *, model, request_id): + return (payload, False, 0, [], "no-op", 0, 0, 0, {}) + + handler = _make_handler(_noop) + body = json.dumps({"input": [{"role": "user"}]}).encode() + + assert await handler._maybe_compress_passthrough_responses(body) == body + + +async def test_compressor_error_fails_open() -> None: + async def _boom(payload, *, model, request_id): + raise RuntimeError("kompress exploded") + + handler = _make_handler(_boom) + body = json.dumps({"input": [{"role": "user"}]}).encode() + + # Fail-open: original body forwarded, exception swallowed. + assert await handler._maybe_compress_passthrough_responses(body) == body + + +def test_config_defaults_off() -> None: + from headroom.proxy.models import ProxyConfig + + assert ProxyConfig().compress_passthrough is False + + +def test_feature_flag_tolerates_missing_config() -> None: + """The passthrough guard must resolve ``self.config`` safely. + + Some handler/proxy objects reach ``handle_passthrough`` without a ``config`` + attribute at all. Reading ``self.config.compress_passthrough`` directly + raises ``AttributeError`` before ``getattr``'s default applies, regressing + the pre-existing verbatim passthrough path. The flag lookup must instead + treat a missing config as feature-off. + """ + handler = OpenAIHandlerMixin.__new__(OpenAIHandlerMixin) + assert not hasattr(handler, "config") + + _pt_config = getattr(handler, "config", None) + assert getattr(_pt_config, "compress_passthrough", False) is False