2026-04-04 14:36:29 -05:00
|
|
|
"""Regression tests for OpenAI cache-mode stability in proxy mode."""
|
|
|
|
|
|
|
|
|
|
from __future__ import annotations
|
|
|
|
|
|
|
|
|
|
from types import SimpleNamespace
|
|
|
|
|
|
|
|
|
|
import httpx
|
|
|
|
|
import pytest
|
|
|
|
|
|
|
|
|
|
pytest.importorskip("fastapi")
|
|
|
|
|
|
|
|
|
|
from fastapi.testclient import TestClient
|
|
|
|
|
|
|
|
|
|
from headroom.proxy.server import ProxyConfig, create_app
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
class _FakePrefixTracker:
|
|
|
|
|
def __init__(self, frozen_count: int):
|
|
|
|
|
self._frozen_count = frozen_count
|
|
|
|
|
|
|
|
|
|
def get_frozen_message_count(self) -> int:
|
|
|
|
|
return self._frozen_count
|
|
|
|
|
|
fix(proxy): freeze must forward cached (compressed) prefix byte-identical — stop token-mode cache busting (#1850)
The freeze path (both providers) emits the agent's ORIGINAL bytes for a
frozen message, but the provider cached whatever we FORWARDED last turn
(the compressed form). Forwarding original then mismatches the cached
prefix and busts it from that point — re-creating the whole suffix.
Measured on a real SWE-bench run: 100% of attributed misses were
prefix_change, ~56% of ALL cache-writes were bust-induced (2.8M tokens),
driving cache_create +150% and cost +41% vs baseline.
Cache mode already avoided this via _extract_cache_stable_delta (replay
the previously-forwarded prefix, compress only the delta). Token mode
called apply(frozen_count) directly, which forwards original for the
frozen region.
Fix: add a shared, provider-agnostic overlay_cached_prefix() that
replays the previously-forwarded (cached, compressed) prefix
byte-identical, append-only guarded and idempotent, and apply it in BOTH
the Anthropic and OpenAI handlers right before forwarding. This makes
freezing byte-identical in every mode, so the only remaining difference
between "token" and "cache" mode is how large a mutable
(still-compressible) tail each leaves — not whether the frozen prefix
busts the cache.
Tests:
- test_cache_prefix_overlay.py: the helper (replay, append-only guard,
idempotence).
- test_cross_turn_cache_safety.py: the invariant that was missing —
drive the REAL tracker + freeze + overlay over multiple append-only
turns against a simulated provider prefix cache and assert the forwarded
prefix stays byte-identical turn-over-turn. Load-bearing: it fails
(detects the bust) without the overlay.
## Description
<!-- Briefly explain the change and why it is needed. -->
Closes #
## Type of Change
- [ ] Bug fix (non-breaking change that fixes an issue)
- [ ] New feature (non-breaking change that adds functionality)
- [ ] Breaking change (fix or feature that would cause existing
functionality to change)
- [ ] Documentation update
- [ ] Performance improvement
- [ ] Code refactoring (no functional changes)
## Changes Made
-
## Testing
<!-- Check what you actually ran, then paste the real command output
below. -->
- [ ] Unit tests pass (`pytest`)
- [ ] Linting passes (`ruff check .`)
- [ ] Type checking passes (`mypy headroom`)
- [ ] New tests added for new functionality
- [ ] Manual testing performed
### Test Output
```text
# Paste relevant command output or artifact links here
```
## Real Behavior Proof
- Environment:
- Exact command / steps:
- Observed result:
- Not tested:
## Review Readiness
- [ ] I have performed a self-review
- [ ] This PR is ready for human review
## Checklist
- [ ] My code follows the project's style guidelines
- [ ] I have performed a self-review of my code
- [ ] I have commented my code, particularly in hard-to-understand areas
- [ ] I have made corresponding changes to the documentation
- [ ] My changes generate no new warnings
- [ ] I have added tests that prove my fix is effective or that my
feature works
- [ ] New and existing unit tests pass locally with my changes
- [ ] I have updated the CHANGELOG.md if applicable
## Screenshots (if applicable)
Add screenshots to help explain your changes.
## Additional Notes
<!-- Mention any N/A checklist items, tradeoffs, follow-ups, or
maintainer context. -->
2026-07-06 14:54:39 -07:00
|
|
|
# Empty history → overlay_cached_prefix() is a no-op here, so these tests
|
|
|
|
|
# keep asserting the cache-freeze behavior they always have. The cross-turn
|
|
|
|
|
# overlay itself is exercised in test_cross_turn_cache_safety.py against the
|
|
|
|
|
# real tracker; these stubs just satisfy the handler's overlay call.
|
|
|
|
|
def get_last_original_messages(self): # noqa: ANN201
|
|
|
|
|
return []
|
|
|
|
|
|
|
|
|
|
def get_last_forwarded_messages(self): # noqa: ANN201
|
|
|
|
|
return []
|
|
|
|
|
|
2026-04-04 14:36:29 -05:00
|
|
|
def update_from_response(self, **kwargs): # noqa: ANN003
|
|
|
|
|
return None
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _make_proxy_client() -> TestClient:
|
|
|
|
|
config = ProxyConfig(
|
|
|
|
|
optimize=False,
|
|
|
|
|
cache_enabled=False,
|
|
|
|
|
rate_limit_enabled=False,
|
|
|
|
|
cost_tracking_enabled=False,
|
|
|
|
|
log_requests=False,
|
|
|
|
|
ccr_inject_tool=False,
|
|
|
|
|
ccr_handle_responses=False,
|
|
|
|
|
ccr_context_tracking=False,
|
|
|
|
|
image_optimize=False,
|
|
|
|
|
)
|
|
|
|
|
app = create_app(config)
|
|
|
|
|
return TestClient(app)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_openai_cache_mode_freezes_previous_turns() -> None:
|
|
|
|
|
captured = {}
|
|
|
|
|
with _make_proxy_client() as client:
|
|
|
|
|
proxy = client.app.state.proxy
|
|
|
|
|
proxy.config.optimize = True
|
|
|
|
|
proxy.config.mode = "cache"
|
|
|
|
|
|
|
|
|
|
fake_tracker = _FakePrefixTracker(frozen_count=0)
|
2026-04-04 22:33:44 -05:00
|
|
|
proxy.session_tracker_store.compute_session_id = lambda request, model, messages: (
|
|
|
|
|
"stable-session"
|
2026-04-04 14:42:32 -05:00
|
|
|
)
|
2026-04-04 14:36:29 -05:00
|
|
|
proxy.session_tracker_store.get_or_create = lambda session_id, provider: fake_tracker
|
|
|
|
|
|
|
|
|
|
def _fake_apply(**kwargs):
|
|
|
|
|
captured["frozen_message_count"] = kwargs.get("frozen_message_count")
|
|
|
|
|
return SimpleNamespace(
|
|
|
|
|
messages=kwargs["messages"],
|
|
|
|
|
transforms_applied=[],
|
|
|
|
|
timing={},
|
|
|
|
|
tokens_before=60,
|
|
|
|
|
tokens_after=60,
|
|
|
|
|
waste_signals=None,
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
proxy.openai_pipeline.apply = _fake_apply
|
|
|
|
|
|
fix: A3 — byte-faithful Python forwarders; serialize canonical only when mutated
Eliminates P0-2 universally. Every Python forwarder (server.py
`_retry_request`, handlers/streaming.py `_stream_response`,
handlers/openai.py `_ws_http_fallback`, handlers/batch.py `_batch_passthrough`
+ batch-create + Google batch passthrough, handlers/anthropic.py CCR
continuation + batch endpoint) now switches from `httpx ... json=body` to
`httpx ... content=raw_bytes`. The default httpx JSON encoder was
re-serializing every request with `, `/`: ` separators and `\\uXXXX` ASCII
escapes — collapsing Anthropic prompt-cache hit-rate.
Forwarder strategy:
- unmutated body → forward `await request.body()` verbatim;
- mutated body → re-serialize once via the new
`serialize_body_canonical(body) -> bytes` helper (compact separators,
`ensure_ascii=False`, dict insertion order preserved).
`HEADROOM_PROXY_PYTHON_FORWARDER_MODE` env var configures the mode:
- `byte_faithful` (default) — the new behavior;
- `legacy_json_kwarg` — explicit operator opt-in for emergency rollback.
Documented in `docs/content/docs/configuration.mdx`. NOT a fallback —
unknown values raise loudly per build constraint #4.
`BodyMutationTracker` accompanies each request through the handler so
transform sites mark the tracker (`memory_injection`,
`image_compression`, `compression_*`, `batch_compression`,
`ccr_continuation`, etc.). At forwarder dispatch we additionally compare
the final body dict against the parsed original bytes as a structural
safety net — any silent mutation we missed still triggers canonical
re-serialization.
A2 follow-up: `handlers/openai.py:534-540` (Chat Completions memory
injection) was prepending a system message; replaced with
`append_text_to_latest_user_chat_message`, the OpenAI Chat Completions
analog of `_append_context_to_latest_non_frozen_user_turn`. The cache
hot zone (system messages) is now sacrosanct on /v1/chat/completions
too. Honors `HEADROOM_MEMORY_INJECTION_MODE=disabled`.
Structured logging: every forwarder emits an `event=outbound_request`
log line with `forwarder`, `path`, `body_bytes`, `body_mutated`,
`mutation_reasons`, `source` (passthrough|canonical|legacy),
`request_id`. Never logs Authorization or full body.
`_read_request_json` factored to share `_read_request_body_bytes` with
new `read_request_json_with_bytes` so the anthropic handler can capture
both the parsed dict and the original (decompressed) bytes.
Tests:
- `tests/test_proxy_byte_faithful_forwarding.py` (28 tests):
SHA-256 byte-equality on /v1/messages and streaming, unicode
preservation, numeric precision, mutation-tracker invariants,
canonical-serializer properties, legacy-mode rollback, OpenAI
Chat memory routing.
- Existing test mocks updated to accept the new `**kwargs` on
`_retry_request` (no behavior change).
- `tests/test_proxy_handlers_batch.py` updated to read the captured
`content=` bytes (formerly `json=`).
- One A2 test corrected (`test_anthropic_tool_sort_and_context_append_helpers`)
to match the live-zone-tail semantics introduced by A2.
Constraints satisfied: configurable env var; no new regex / hardcodes;
no silent fallback (`legacy_json_kwarg` is operator opt-in);
performant (`prepare_outbound_body_bytes` is O(1) for passthrough);
elegant single-responsibility helpers; structured tracing logs.
2026-05-02 09:02:10 -07:00
|
|
|
async def _fake_retry(method, url, headers, body, stream=False, **kwargs): # noqa: ANN001
|
2026-04-04 14:36:29 -05:00
|
|
|
return httpx.Response(
|
|
|
|
|
200,
|
|
|
|
|
json={
|
|
|
|
|
"id": "chatcmpl_1",
|
|
|
|
|
"choices": [
|
2026-04-04 14:42:32 -05:00
|
|
|
{
|
|
|
|
|
"index": 0,
|
|
|
|
|
"message": {"role": "assistant", "content": "ok"},
|
|
|
|
|
"finish_reason": "stop",
|
|
|
|
|
}
|
2026-04-04 14:36:29 -05:00
|
|
|
],
|
|
|
|
|
"usage": {"prompt_tokens": 60, "completion_tokens": 3, "total_tokens": 63},
|
|
|
|
|
},
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
proxy._retry_request = _fake_retry
|
|
|
|
|
|
|
|
|
|
response = client.post(
|
|
|
|
|
"/v1/chat/completions",
|
|
|
|
|
headers={"authorization": "Bearer test-key"},
|
|
|
|
|
json={
|
|
|
|
|
"model": "gpt-4o-mini",
|
|
|
|
|
"messages": [
|
|
|
|
|
{"role": "user", "content": "turn1"},
|
|
|
|
|
{"role": "assistant", "content": "turn1-assistant"},
|
|
|
|
|
{"role": "user", "content": "current turn"},
|
|
|
|
|
],
|
|
|
|
|
},
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
assert response.status_code == 200
|
|
|
|
|
assert captured["frozen_message_count"] == 2
|
|
|
|
|
|
|
|
|
|
|
fix(proxy): keep OpenAI tool observations mutable in cache mode (#1884)
## Description
Diagnoses and fixes the low-savings OpenAI-compatible cache-mode path
reported in #1696.
OpenAI-compatible tool-calling clients can end a turn with `role:
"tool"` (or legacy `role: "function"`) rather than `role: "user"`. The
OpenAI chat handler's cache-mode freeze boundary treated those tails as
non-mutable, and because `HeadroomProxy` resolves
`_strict_previous_turn_frozen_count` from the Anthropic mixin first, the
OpenAI-specific helper was not used in production. That froze the entire
conversation before `ContentRouter` ran, leaving no live tool
observation to compress and producing near-pass-through savings on long
coding sessions.
This PR keeps final OpenAI tool/function observations mutable in cache
mode, explicitly calls the OpenAI helper to avoid the mixin-name
collision, and clamps negative token-savings artifacts at the
metrics/cost aggregation boundary so stats cannot under-report actual
forwarded savings.
Closes #1696
## Type of Change
- [x] Bug fix (non-breaking change that fixes an issue)
- [ ] New feature (non-breaking change that adds functionality)
- [ ] Breaking change (fix or feature that would cause existing
functionality to change)
- [ ] Documentation update
- [ ] Performance improvement
- [ ] Code refactoring (no functional changes)
## Changes Made
- Treat final OpenAI `user`, `tool`, and `function` messages as the
mutable cache-mode live zone.
- Route OpenAI cache-boundary calls through
`OpenAIHandlerMixin._strict_previous_turn_frozen_count` explicitly so
the Anthropic mixin method cannot shadow it in `HeadroomProxy`'s MRO.
- Preserve cache-mode live-tail boundaries even when compression-cache
state would otherwise freeze the whole request.
- Clamp negative `tokens_saved` artifacts in `CostTracker.record_tokens`
and `PrometheusMetrics.record_request`.
- Add regression coverage for OpenAI final `tool`/`function` tails,
over-frozen tracker state, and non-negative savings aggregation.
## Testing
- [ ] Unit tests pass (`pytest`)
- [x] Linting passes (`ruff check .`)
- [x] Type checking passes (`mypy headroom`)
- [x] New tests added for new functionality
- [x] Manual testing performed
### Test Output
```text
$ maturin build --profile ci --out dist --interpreter python
Built wheel for abi3 Python >= 3.10 to dist\headroom_ai-0.29.0-cp310-abi3-win_amd64.whl
$ python -m pytest tests\test_proxy_handler_helpers.py tests\test_proxy_openai_cache_stability.py tests\test_observability_metrics.py tests\test_cost_tracker_counterfactual.py
49 passed in 10.27s
$ python -m ruff check .
All checks passed!
$ python -m mypy headroom
Success: no issues found in 407 source files
$ python -m pytest
53 failed, 7703 passed, 488 skipped, 5893 warnings, 131 errors in 595.18s (0:09:55)
```
Full-suite note: the full local `pytest` run was attempted on
Windows/Python 3.13 after building `headroom._core`. It did not complete
green due to broad pre-existing/local-environment failures outside this
change area, dominated by SQLite/memory persistence permission/path
errors plus unrelated adapter/cache/tool tests. The focused regression
suite for this PR passes, and repo-level lint/type gates pass.
## Real Behavior Proof
- Environment: Windows, Python 3.13.13, Rust/Cargo available, local
`headroom._core` wheel built with `maturin build --profile ci`.
- Exact command / steps: ran the OpenAI cache-stability tests with final
`role: "tool"` and `role: "function"` chat tails.
- Observed result:
`test_openai_cache_mode_keeps_final_tool_observation_mutable[tool]` and
`[function]` pass, proving the pipeline receives `frozen_message_count
== 2` for a 3-message request instead of freezing all 3 messages.
- Not tested: live Lemonade/KiloCode upstream session; no local Lemonade
Server was available.
## Review Readiness
- [x] I have performed a self-review
- [x] This PR is ready for human review
## Checklist
- [x] My code follows the project's style guidelines
- [x] I have performed a self-review of my code
- [x] I have commented my code, particularly in hard-to-understand areas
- [ ] I have made corresponding changes to the documentation
- [x] My changes generate no new warnings
- [x] I have added tests that prove my fix is effective or that my
feature works
- [ ] New and existing unit tests pass locally with my changes
- [ ] I have updated the CHANGELOG.md if applicable
## Screenshots (if applicable)
N/A
## Additional Notes
Docs and CHANGELOG are N/A for this narrow proxy bug fix. The broad
local `pytest` checkbox is intentionally left unchecked because the full
suite had unrelated local-environment failures; see the test output
above. Focused regression tests, `ruff check .`, and `mypy headroom` are
green.
2026-07-09 14:51:01 +00:00
|
|
|
@pytest.mark.parametrize("tail_role", ["tool", "function"])
|
|
|
|
|
def test_openai_cache_mode_keeps_final_tool_observation_mutable(tail_role: str) -> None:
|
|
|
|
|
captured = {}
|
|
|
|
|
with _make_proxy_client() as client:
|
|
|
|
|
proxy = client.app.state.proxy
|
|
|
|
|
proxy.config.optimize = True
|
|
|
|
|
proxy.config.mode = "cache"
|
|
|
|
|
|
|
|
|
|
fake_tracker = _FakePrefixTracker(frozen_count=0)
|
|
|
|
|
proxy.session_tracker_store.compute_session_id = lambda request, model, messages: (
|
|
|
|
|
"stable-session"
|
|
|
|
|
)
|
|
|
|
|
proxy.session_tracker_store.get_or_create = lambda session_id, provider: fake_tracker
|
|
|
|
|
|
|
|
|
|
def _fake_apply(**kwargs):
|
|
|
|
|
captured.setdefault("calls", []).append(
|
|
|
|
|
{
|
|
|
|
|
"frozen_message_count": kwargs.get("frozen_message_count"),
|
|
|
|
|
"roles": [msg.get("role") for msg in kwargs["messages"]],
|
|
|
|
|
"mode": proxy.config.mode,
|
|
|
|
|
}
|
|
|
|
|
)
|
|
|
|
|
return SimpleNamespace(
|
|
|
|
|
messages=kwargs["messages"],
|
|
|
|
|
transforms_applied=["test:compress-tail"],
|
|
|
|
|
timing={},
|
|
|
|
|
tokens_before=120,
|
|
|
|
|
tokens_after=80,
|
|
|
|
|
waste_signals=None,
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
proxy.openai_pipeline.apply = _fake_apply
|
|
|
|
|
|
|
|
|
|
async def _fake_retry(method, url, headers, body, stream=False, **kwargs): # noqa: ANN001
|
|
|
|
|
return httpx.Response(
|
|
|
|
|
200,
|
|
|
|
|
json={
|
|
|
|
|
"id": "chatcmpl_tool_tail",
|
|
|
|
|
"choices": [
|
|
|
|
|
{
|
|
|
|
|
"index": 0,
|
|
|
|
|
"message": {"role": "assistant", "content": "ok"},
|
|
|
|
|
"finish_reason": "stop",
|
|
|
|
|
}
|
|
|
|
|
],
|
|
|
|
|
"usage": {"prompt_tokens": 80, "completion_tokens": 3, "total_tokens": 83},
|
|
|
|
|
},
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
proxy._retry_request = _fake_retry
|
|
|
|
|
|
|
|
|
|
tail = {
|
|
|
|
|
"role": tail_role,
|
|
|
|
|
"content": "large command observation " * 200,
|
|
|
|
|
}
|
|
|
|
|
if tail_role == "tool":
|
|
|
|
|
tail["tool_call_id"] = "call_1"
|
|
|
|
|
else:
|
|
|
|
|
tail["name"] = "bash"
|
|
|
|
|
|
|
|
|
|
response = client.post(
|
|
|
|
|
"/v1/chat/completions",
|
|
|
|
|
headers={"authorization": "Bearer test-key"},
|
|
|
|
|
json={
|
|
|
|
|
"model": "gpt-4o-mini",
|
|
|
|
|
"messages": [
|
|
|
|
|
{"role": "user", "content": "turn1"},
|
|
|
|
|
{"role": "assistant", "content": "run command"},
|
|
|
|
|
tail,
|
|
|
|
|
],
|
|
|
|
|
},
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
assert response.status_code == 200
|
|
|
|
|
assert any(call["frozen_message_count"] == 2 for call in captured["calls"]), captured[
|
|
|
|
|
"calls"
|
|
|
|
|
]
|
|
|
|
|
|
|
|
|
|
|
2026-04-04 14:36:29 -05:00
|
|
|
def test_openai_cache_mode_restores_mutated_frozen_prefix() -> None:
|
|
|
|
|
captured = {}
|
|
|
|
|
with _make_proxy_client() as client:
|
|
|
|
|
proxy = client.app.state.proxy
|
|
|
|
|
proxy.config.optimize = True
|
|
|
|
|
proxy.config.mode = "cache"
|
|
|
|
|
|
|
|
|
|
fake_tracker = _FakePrefixTracker(frozen_count=0)
|
2026-04-04 22:33:44 -05:00
|
|
|
proxy.session_tracker_store.compute_session_id = lambda request, model, messages: (
|
|
|
|
|
"stable-session"
|
2026-04-04 14:42:32 -05:00
|
|
|
)
|
2026-04-04 14:36:29 -05:00
|
|
|
proxy.session_tracker_store.get_or_create = lambda session_id, provider: fake_tracker
|
|
|
|
|
|
|
|
|
|
original_messages = [
|
|
|
|
|
{"role": "user", "content": "turn1"},
|
|
|
|
|
{"role": "assistant", "content": "turn1-assistant"},
|
|
|
|
|
{"role": "user", "content": "current turn"},
|
|
|
|
|
]
|
|
|
|
|
|
|
|
|
|
def _fake_apply(**kwargs):
|
|
|
|
|
mutated = list(kwargs["messages"])
|
|
|
|
|
mutated[0] = {**mutated[0], "content": "MUTATED_PREFIX"}
|
|
|
|
|
return SimpleNamespace(
|
|
|
|
|
messages=mutated,
|
|
|
|
|
transforms_applied=["fake:mutated"],
|
|
|
|
|
timing={},
|
|
|
|
|
tokens_before=70,
|
|
|
|
|
tokens_after=65,
|
|
|
|
|
waste_signals=None,
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
proxy.openai_pipeline.apply = _fake_apply
|
|
|
|
|
|
fix: A3 — byte-faithful Python forwarders; serialize canonical only when mutated
Eliminates P0-2 universally. Every Python forwarder (server.py
`_retry_request`, handlers/streaming.py `_stream_response`,
handlers/openai.py `_ws_http_fallback`, handlers/batch.py `_batch_passthrough`
+ batch-create + Google batch passthrough, handlers/anthropic.py CCR
continuation + batch endpoint) now switches from `httpx ... json=body` to
`httpx ... content=raw_bytes`. The default httpx JSON encoder was
re-serializing every request with `, `/`: ` separators and `\\uXXXX` ASCII
escapes — collapsing Anthropic prompt-cache hit-rate.
Forwarder strategy:
- unmutated body → forward `await request.body()` verbatim;
- mutated body → re-serialize once via the new
`serialize_body_canonical(body) -> bytes` helper (compact separators,
`ensure_ascii=False`, dict insertion order preserved).
`HEADROOM_PROXY_PYTHON_FORWARDER_MODE` env var configures the mode:
- `byte_faithful` (default) — the new behavior;
- `legacy_json_kwarg` — explicit operator opt-in for emergency rollback.
Documented in `docs/content/docs/configuration.mdx`. NOT a fallback —
unknown values raise loudly per build constraint #4.
`BodyMutationTracker` accompanies each request through the handler so
transform sites mark the tracker (`memory_injection`,
`image_compression`, `compression_*`, `batch_compression`,
`ccr_continuation`, etc.). At forwarder dispatch we additionally compare
the final body dict against the parsed original bytes as a structural
safety net — any silent mutation we missed still triggers canonical
re-serialization.
A2 follow-up: `handlers/openai.py:534-540` (Chat Completions memory
injection) was prepending a system message; replaced with
`append_text_to_latest_user_chat_message`, the OpenAI Chat Completions
analog of `_append_context_to_latest_non_frozen_user_turn`. The cache
hot zone (system messages) is now sacrosanct on /v1/chat/completions
too. Honors `HEADROOM_MEMORY_INJECTION_MODE=disabled`.
Structured logging: every forwarder emits an `event=outbound_request`
log line with `forwarder`, `path`, `body_bytes`, `body_mutated`,
`mutation_reasons`, `source` (passthrough|canonical|legacy),
`request_id`. Never logs Authorization or full body.
`_read_request_json` factored to share `_read_request_body_bytes` with
new `read_request_json_with_bytes` so the anthropic handler can capture
both the parsed dict and the original (decompressed) bytes.
Tests:
- `tests/test_proxy_byte_faithful_forwarding.py` (28 tests):
SHA-256 byte-equality on /v1/messages and streaming, unicode
preservation, numeric precision, mutation-tracker invariants,
canonical-serializer properties, legacy-mode rollback, OpenAI
Chat memory routing.
- Existing test mocks updated to accept the new `**kwargs` on
`_retry_request` (no behavior change).
- `tests/test_proxy_handlers_batch.py` updated to read the captured
`content=` bytes (formerly `json=`).
- One A2 test corrected (`test_anthropic_tool_sort_and_context_append_helpers`)
to match the live-zone-tail semantics introduced by A2.
Constraints satisfied: configurable env var; no new regex / hardcodes;
no silent fallback (`legacy_json_kwarg` is operator opt-in);
performant (`prepare_outbound_body_bytes` is O(1) for passthrough);
elegant single-responsibility helpers; structured tracing logs.
2026-05-02 09:02:10 -07:00
|
|
|
async def _fake_retry(method, url, headers, body, stream=False, **kwargs): # noqa: ANN001
|
2026-04-04 14:36:29 -05:00
|
|
|
captured["body"] = body
|
|
|
|
|
return httpx.Response(
|
|
|
|
|
200,
|
|
|
|
|
json={
|
|
|
|
|
"id": "chatcmpl_2",
|
|
|
|
|
"choices": [
|
2026-04-04 14:42:32 -05:00
|
|
|
{
|
|
|
|
|
"index": 0,
|
|
|
|
|
"message": {"role": "assistant", "content": "ok"},
|
|
|
|
|
"finish_reason": "stop",
|
|
|
|
|
}
|
2026-04-04 14:36:29 -05:00
|
|
|
],
|
|
|
|
|
"usage": {"prompt_tokens": 65, "completion_tokens": 3, "total_tokens": 68},
|
|
|
|
|
},
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
proxy._retry_request = _fake_retry
|
|
|
|
|
|
|
|
|
|
response = client.post(
|
|
|
|
|
"/v1/chat/completions",
|
|
|
|
|
headers={"authorization": "Bearer test-key"},
|
|
|
|
|
json={
|
|
|
|
|
"model": "gpt-4o-mini",
|
|
|
|
|
"messages": original_messages,
|
|
|
|
|
},
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
assert response.status_code == 200
|
|
|
|
|
sent_messages = captured["body"]["messages"]
|
|
|
|
|
assert sent_messages[0] == original_messages[0]
|
|
|
|
|
assert sent_messages[1] == original_messages[1]
|
fix(proxy): remove content-keyed TTL walker that conflated content with positional cache (#327)
The Anthropic token-mode handler walked past prefix_tracker.frozen_message_count
whenever an upcoming tool_result's content-hash matched comp_cache._stable_hashes
or should_defer_compression returned True. That conflated content equality with
positional cache membership.
Anthropic's prefix cache is POSITIONAL: bytes 0..K cached, anything past K is
fresh. _stable_hashes is content-keyed and grows unbounded. In long Claude Code
sessions where tool_result content rhymes across turns (repeated system prompts,
repeated file reads, repeated tool descriptions), the walker advanced
frozen_message_count to len(messages) on every turn and the pipeline produced
transforms_applied=[] on 73% of requests in user SvenMeyer's reported session
(headroom-stats-2026-05-01.json: 74 of 101 eligible requests "prefix_frozen") —
even after the prior fix in 44944fb. The 15 requests that did compress averaged
21%, proving compression itself works when reached.
Fix: delete the walker. The freeze boundary is now
frozen_message_count = min(
prefix_tracker.frozen_message_count, # positional ground truth
comp_cache.compute_frozen_count(messages), # local cache lower bound
)
compute_frozen_count's use of _stable_hashes can only LOWER the freeze via the
min clamp, never raise it past prefix_tracker's value. For any position in the
gap [compute_frozen_count, prefix_tracker.frozen_count], recompressing produces
byte-stable output (compression is deterministic on input content), so
Anthropic's prefix cache stays valid.
Cross-handler verification:
* OpenAI handler (proxy/handlers/openai.py:358-382) does not have this walker
— uses only compute_frozen_count. Codex routes through OpenAI handler. Both
unaffected.
* Streaming and non-streaming both invoke anthropic_pipeline.apply() before the
upstream call. One fix covers both paths.
* Cache mode (is_cache_mode) takes the _extract_cache_stable_delta path and is
independent of the walker. Unaffected.
Tests: six new regression tests lock down the post-fix invariants — clamp to
min(prefix_tracker, compute_frozen_count); fresh tool_result whose hash matches
old _stable_hashes entry is not frozen; frozen prefix byte-stable across the
pipeline; 10-turn session produces non-empty compression suffix every turn;
streaming and non-streaming compute identical frozen_message_count; OpenAI
handler never calls the walker functions. Plus scripts/smoke_issue_327.py
(gated by RUN_LIVE_API=1) drives a 10-turn conversation against
api.anthropic.com in both shapes (string + list-of-blocks) and both modes
(streaming + non-streaming).
ci-precheck clean. 191 tests pass.
Follow-ups (separate PRs):
* Fix _cache perpetually empty (anthropic.py result.messages != working_messages
comparison rarely fires in token mode).
* Cap _stable_hashes with bounded LRU + 1h TTL — hygiene only after the freeze
gate is removed.
* List-shape tool_result content gates at content_router.py:1975 and
intelligent_context.py:657 (cluster A from the audit).
2026-05-01 12:04:28 -07:00
|
|
|
|
|
|
|
|
|
|
|
|
|
# ─── Issue #327 cross-handler regression ────────────────────────────────
|
|
|
|
|
#
|
|
|
|
|
# The OpenAI handler was never affected by issue #327's content-keyed walker
|
|
|
|
|
# bug — it has only ever used `compute_frozen_count` (positional). This test
|
|
|
|
|
# locks that property by spying on the OpenAI traffic path and asserting that
|
|
|
|
|
# the buggy walker functions (`should_defer_compression`, `mark_stable`) are
|
|
|
|
|
# never called from the production handler. If a future refactor accidentally
|
|
|
|
|
# adds the same walker to OpenAI, this test fails immediately.
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_issue_327_openai_handler_does_not_call_walker_functions() -> None:
|
|
|
|
|
calls: list[tuple[str, tuple, dict]] = []
|
|
|
|
|
|
|
|
|
|
class _SpyCompCache:
|
|
|
|
|
def apply_cached(self, messages): # noqa: ANN001
|
|
|
|
|
calls.append(("apply_cached", (), {}))
|
|
|
|
|
return list(messages)
|
|
|
|
|
|
|
|
|
|
def compute_frozen_count(self, messages): # noqa: ANN001
|
|
|
|
|
calls.append(("compute_frozen_count", (), {}))
|
|
|
|
|
return 0
|
|
|
|
|
|
|
|
|
|
def update_from_result(self, originals, compressed): # noqa: ANN001
|
|
|
|
|
calls.append(("update_from_result", (), {}))
|
|
|
|
|
|
|
|
|
|
def mark_stable_from_messages(self, messages, up_to): # noqa: ANN001
|
|
|
|
|
calls.append(("mark_stable_from_messages", (up_to,), {}))
|
|
|
|
|
|
|
|
|
|
# Methods below MUST NOT be called from OpenAI handler.
|
|
|
|
|
def should_defer_compression(self, *args, **kwargs): # noqa: ANN001, ANN002, ANN003
|
|
|
|
|
calls.append(("should_defer_compression", args, kwargs))
|
|
|
|
|
return False
|
|
|
|
|
|
|
|
|
|
def mark_stable(self, content_hash): # noqa: ANN001
|
|
|
|
|
calls.append(("mark_stable", (content_hash,), {}))
|
|
|
|
|
|
|
|
|
|
@staticmethod
|
|
|
|
|
def content_hash(content): # noqa: ANN001
|
|
|
|
|
return f"H({content[:40] if isinstance(content, str) else 'list'})"
|
|
|
|
|
|
|
|
|
|
with _make_proxy_client() as client:
|
|
|
|
|
proxy = client.app.state.proxy
|
|
|
|
|
proxy.config.optimize = True
|
|
|
|
|
proxy.config.mode = "token" # token mode is where Anthropic had the bug
|
|
|
|
|
|
|
|
|
|
fake_tracker = _FakePrefixTracker(frozen_count=0)
|
|
|
|
|
proxy.session_tracker_store.compute_session_id = lambda request, model, messages: (
|
|
|
|
|
"openai-spy-session"
|
|
|
|
|
)
|
|
|
|
|
proxy.session_tracker_store.get_or_create = lambda s, p: fake_tracker
|
|
|
|
|
proxy._get_compression_cache = lambda s: _SpyCompCache()
|
|
|
|
|
|
|
|
|
|
def _fake_apply(**kwargs): # noqa: ANN003
|
|
|
|
|
return SimpleNamespace(
|
|
|
|
|
messages=list(kwargs["messages"]),
|
|
|
|
|
transforms_applied=[],
|
|
|
|
|
timing={},
|
|
|
|
|
tokens_before=60,
|
|
|
|
|
tokens_after=60,
|
|
|
|
|
waste_signals=None,
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
proxy.openai_pipeline.apply = _fake_apply
|
|
|
|
|
|
fix: A3 — byte-faithful Python forwarders; serialize canonical only when mutated
Eliminates P0-2 universally. Every Python forwarder (server.py
`_retry_request`, handlers/streaming.py `_stream_response`,
handlers/openai.py `_ws_http_fallback`, handlers/batch.py `_batch_passthrough`
+ batch-create + Google batch passthrough, handlers/anthropic.py CCR
continuation + batch endpoint) now switches from `httpx ... json=body` to
`httpx ... content=raw_bytes`. The default httpx JSON encoder was
re-serializing every request with `, `/`: ` separators and `\\uXXXX` ASCII
escapes — collapsing Anthropic prompt-cache hit-rate.
Forwarder strategy:
- unmutated body → forward `await request.body()` verbatim;
- mutated body → re-serialize once via the new
`serialize_body_canonical(body) -> bytes` helper (compact separators,
`ensure_ascii=False`, dict insertion order preserved).
`HEADROOM_PROXY_PYTHON_FORWARDER_MODE` env var configures the mode:
- `byte_faithful` (default) — the new behavior;
- `legacy_json_kwarg` — explicit operator opt-in for emergency rollback.
Documented in `docs/content/docs/configuration.mdx`. NOT a fallback —
unknown values raise loudly per build constraint #4.
`BodyMutationTracker` accompanies each request through the handler so
transform sites mark the tracker (`memory_injection`,
`image_compression`, `compression_*`, `batch_compression`,
`ccr_continuation`, etc.). At forwarder dispatch we additionally compare
the final body dict against the parsed original bytes as a structural
safety net — any silent mutation we missed still triggers canonical
re-serialization.
A2 follow-up: `handlers/openai.py:534-540` (Chat Completions memory
injection) was prepending a system message; replaced with
`append_text_to_latest_user_chat_message`, the OpenAI Chat Completions
analog of `_append_context_to_latest_non_frozen_user_turn`. The cache
hot zone (system messages) is now sacrosanct on /v1/chat/completions
too. Honors `HEADROOM_MEMORY_INJECTION_MODE=disabled`.
Structured logging: every forwarder emits an `event=outbound_request`
log line with `forwarder`, `path`, `body_bytes`, `body_mutated`,
`mutation_reasons`, `source` (passthrough|canonical|legacy),
`request_id`. Never logs Authorization or full body.
`_read_request_json` factored to share `_read_request_body_bytes` with
new `read_request_json_with_bytes` so the anthropic handler can capture
both the parsed dict and the original (decompressed) bytes.
Tests:
- `tests/test_proxy_byte_faithful_forwarding.py` (28 tests):
SHA-256 byte-equality on /v1/messages and streaming, unicode
preservation, numeric precision, mutation-tracker invariants,
canonical-serializer properties, legacy-mode rollback, OpenAI
Chat memory routing.
- Existing test mocks updated to accept the new `**kwargs` on
`_retry_request` (no behavior change).
- `tests/test_proxy_handlers_batch.py` updated to read the captured
`content=` bytes (formerly `json=`).
- One A2 test corrected (`test_anthropic_tool_sort_and_context_append_helpers`)
to match the live-zone-tail semantics introduced by A2.
Constraints satisfied: configurable env var; no new regex / hardcodes;
no silent fallback (`legacy_json_kwarg` is operator opt-in);
performant (`prepare_outbound_body_bytes` is O(1) for passthrough);
elegant single-responsibility helpers; structured tracing logs.
2026-05-02 09:02:10 -07:00
|
|
|
async def _fake_retry(method, url, headers, body, stream=False, **kwargs): # noqa: ANN001
|
fix(proxy): remove content-keyed TTL walker that conflated content with positional cache (#327)
The Anthropic token-mode handler walked past prefix_tracker.frozen_message_count
whenever an upcoming tool_result's content-hash matched comp_cache._stable_hashes
or should_defer_compression returned True. That conflated content equality with
positional cache membership.
Anthropic's prefix cache is POSITIONAL: bytes 0..K cached, anything past K is
fresh. _stable_hashes is content-keyed and grows unbounded. In long Claude Code
sessions where tool_result content rhymes across turns (repeated system prompts,
repeated file reads, repeated tool descriptions), the walker advanced
frozen_message_count to len(messages) on every turn and the pipeline produced
transforms_applied=[] on 73% of requests in user SvenMeyer's reported session
(headroom-stats-2026-05-01.json: 74 of 101 eligible requests "prefix_frozen") —
even after the prior fix in 44944fb. The 15 requests that did compress averaged
21%, proving compression itself works when reached.
Fix: delete the walker. The freeze boundary is now
frozen_message_count = min(
prefix_tracker.frozen_message_count, # positional ground truth
comp_cache.compute_frozen_count(messages), # local cache lower bound
)
compute_frozen_count's use of _stable_hashes can only LOWER the freeze via the
min clamp, never raise it past prefix_tracker's value. For any position in the
gap [compute_frozen_count, prefix_tracker.frozen_count], recompressing produces
byte-stable output (compression is deterministic on input content), so
Anthropic's prefix cache stays valid.
Cross-handler verification:
* OpenAI handler (proxy/handlers/openai.py:358-382) does not have this walker
— uses only compute_frozen_count. Codex routes through OpenAI handler. Both
unaffected.
* Streaming and non-streaming both invoke anthropic_pipeline.apply() before the
upstream call. One fix covers both paths.
* Cache mode (is_cache_mode) takes the _extract_cache_stable_delta path and is
independent of the walker. Unaffected.
Tests: six new regression tests lock down the post-fix invariants — clamp to
min(prefix_tracker, compute_frozen_count); fresh tool_result whose hash matches
old _stable_hashes entry is not frozen; frozen prefix byte-stable across the
pipeline; 10-turn session produces non-empty compression suffix every turn;
streaming and non-streaming compute identical frozen_message_count; OpenAI
handler never calls the walker functions. Plus scripts/smoke_issue_327.py
(gated by RUN_LIVE_API=1) drives a 10-turn conversation against
api.anthropic.com in both shapes (string + list-of-blocks) and both modes
(streaming + non-streaming).
ci-precheck clean. 191 tests pass.
Follow-ups (separate PRs):
* Fix _cache perpetually empty (anthropic.py result.messages != working_messages
comparison rarely fires in token mode).
* Cap _stable_hashes with bounded LRU + 1h TTL — hygiene only after the freeze
gate is removed.
* List-shape tool_result content gates at content_router.py:1975 and
intelligent_context.py:657 (cluster A from the audit).
2026-05-01 12:04:28 -07:00
|
|
|
return httpx.Response(
|
|
|
|
|
200,
|
|
|
|
|
json={
|
|
|
|
|
"id": "cmpl",
|
|
|
|
|
"choices": [
|
|
|
|
|
{
|
|
|
|
|
"index": 0,
|
|
|
|
|
"message": {"role": "assistant", "content": "ok"},
|
|
|
|
|
"finish_reason": "stop",
|
|
|
|
|
}
|
|
|
|
|
],
|
|
|
|
|
"usage": {"prompt_tokens": 60, "completion_tokens": 3, "total_tokens": 63},
|
|
|
|
|
},
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
proxy._retry_request = _fake_retry
|
|
|
|
|
|
|
|
|
|
# Drive 5 turns so any walker bug would have time to fire repeatedly.
|
|
|
|
|
for turn in range(5):
|
|
|
|
|
r = client.post(
|
|
|
|
|
"/v1/chat/completions",
|
|
|
|
|
headers={"authorization": "Bearer test-key"},
|
|
|
|
|
json={
|
|
|
|
|
"model": "gpt-4o-mini",
|
|
|
|
|
"messages": [
|
|
|
|
|
{"role": "user", "content": f"turn-{turn}-q"},
|
|
|
|
|
{"role": "assistant", "content": f"turn-{turn}-a"},
|
|
|
|
|
{"role": "tool", "tool_call_id": "t1", "content": "x" * 600},
|
|
|
|
|
{"role": "user", "content": f"continue-{turn}"},
|
|
|
|
|
],
|
|
|
|
|
},
|
|
|
|
|
)
|
|
|
|
|
assert r.status_code == 200
|
|
|
|
|
|
|
|
|
|
method_names = [c[0] for c in calls]
|
|
|
|
|
assert "should_defer_compression" not in method_names, (
|
|
|
|
|
f"OpenAI handler unexpectedly called should_defer_compression. "
|
|
|
|
|
f"Calls observed: {method_names}"
|
|
|
|
|
)
|
|
|
|
|
assert "mark_stable" not in method_names, (
|
|
|
|
|
f"OpenAI handler unexpectedly called mark_stable (the walker side-effect). "
|
|
|
|
|
f"Calls observed: {method_names}"
|
|
|
|
|
)
|
|
|
|
|
# Sanity: the safe positional methods DID fire.
|
|
|
|
|
assert "compute_frozen_count" in method_names
|
|
|
|
|
assert "apply_cached" in method_names
|
fix(proxy): compress OpenCode tool schemas and embedded JSON (#1535)
## Description
Fixes two remaining OpenCode/OpenAI Chat compression gaps after `main`
incorporated the original savings-profile threading and user
content-block work from this PR.
OpenCode requests can still report very low savings when most input
tokens live in verbose `tools` schemas rather than messages. They can
also route poorly when a short instruction wraps a valid JSON block but
does not satisfy the existing long-prose heuristic.
Closes #1534
## Type of Change
- [x] Bug fix (non-breaking change that fixes an issue)
- [ ] New feature (non-breaking change that adds functionality)
- [ ] Breaking change (fix or feature that causes existing functionality
to change)
- [ ] Documentation update
- [ ] Performance improvement
- [ ] Code refactoring (no functional changes)
## Changes Made
- Compact OpenAI Chat Completions `tools` schemas whenever request
compression is active, reusing the existing OpenAI Responses schema
compactor. The outbound tool invocation shape is preserved while
non-semantic annotations such as `$schema`, `title`, and `examples` are
removed.
- Include the tool-schema token delta in Headroom's savings accounting
and expose `openai:chat:tool_schema_compaction` in the applied
transforms.
- Detect valid JSON blocks surrounded by prose or log text as mixed
content, so short OpenCode instructions route through mixed/SmartCrusher
handling instead of falling through or producing a no-op.
- Adapt the mixed-content change to the new
`headroom.transforms.mixed_content` module introduced on `main` by
#1939.
## Why the Focus Changed
The original headline fix—threading savings-profile kwargs into
`/v1/chat/completions`—is now already present on `main`, as is the user
content-block opt-in behavior. Those duplicate changes were removed
during the merge.
The branch also no longer changes developer/system role protection or
forced-Kompress semantics. It follows `main` for both, so the earlier
instruction-role safety concern is outside the current diff.
The resulting PR is limited to two OpenCode-specific compression gaps
that remain reproducible on current `main`.
## Testing
- [x] Unit tests pass (`pytest`)
- [x] Linting passes (`ruff check` on changed files)
- [x] Type checking passes (`mypy headroom`)
- [x] New tests added for the fixed behavior
- [ ] Manual live-upstream testing performed after the latest rebase
### Test Output
```text
59 passed, 1 warning in 83.53s
All checks passed! # ruff check
4 files already formatted # ruff format --check
python -m py_compile: passed
git diff --check: passed
```
Focused test coverage includes:
- OpenAI Chat tool-schema compaction, transform reporting, outbound
schema shape, and positive token savings.
- Embedded JSON mixed-content detection, SmartCrusher routing, positive
savings, and preservation of a critical sentinel value.
- Current `main` regressions for savings-profile threading, user content
blocks, turn hooks, and forced-Kompress behavior.
## Real Behavior Proof
- Environment: Linux ARM64, Python 3.13.12, current `main` at `9bacf481`
merged into the branch.
- Exact command / steps: focused pytest run across the OpenAI
cache-stability, content-router, mixed-content, savings-profile,
user-block, turn-hook, and forced-Kompress suites.
- Observed result: 59 tests passed; the chat request test forwarded
compacted tools and reported positive savings, while the embedded-JSON
fixture used mixed routing and preserved `CRITICAL_NEEDLE_42`.
- Not tested: full repository suite and a live external OpenCode request
after the latest merge; those remain for CI/live follow-up.
## Review Readiness
- [x] I have performed a self-review
- [x] This PR is ready for human review
## Checklist
- [x] My code follows the project's style guidelines
- [x] I have performed a self-review of my code
- [x] I have commented the non-obvious behavior
- [ ] I have made corresponding documentation changes — N/A; internal
routing behavior only
- [x] My changes generate no new warnings
- [x] I have added tests that prove the fixes are effective
- [x] New and existing focused tests pass locally
- [ ] I have updated the changelog — N/A; release automation handles fix
entries
## Screenshots
N/A — proxy/transform behavior only.
## Additional Notes
- Current diff versus `main`: 4 files, 172 insertions, no role-policy or
forced-Kompress changes.
- The mixed-content conflict was resolved by extending the new isolated
parser module rather than reintroducing parsing code into
`ContentRouter`.
2026-07-15 16:58:42 -03:00
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_openai_chat_completions_compacts_tools_when_profile_enabled() -> None:
|
|
|
|
|
captured = {}
|
|
|
|
|
with _make_proxy_client() as client:
|
|
|
|
|
proxy = client.app.state.proxy
|
|
|
|
|
proxy.config.optimize = True
|
|
|
|
|
proxy.config.mode = "token"
|
|
|
|
|
proxy.config.savings_profile = "agent-90"
|
|
|
|
|
|
|
|
|
|
def _fake_apply(**kwargs): # noqa: ANN003
|
|
|
|
|
return SimpleNamespace(
|
|
|
|
|
messages=kwargs["messages"],
|
|
|
|
|
transforms_applied=[],
|
|
|
|
|
timing={},
|
|
|
|
|
tokens_before=10,
|
|
|
|
|
tokens_after=10,
|
|
|
|
|
waste_signals=None,
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
proxy.openai_pipeline.apply = _fake_apply
|
|
|
|
|
|
|
|
|
|
async def _fake_retry(method, url, headers, body, stream=False, **kwargs): # noqa: ANN001
|
|
|
|
|
captured["body"] = body
|
|
|
|
|
return httpx.Response(
|
|
|
|
|
200,
|
|
|
|
|
json={
|
|
|
|
|
"id": "chatcmpl_tools",
|
|
|
|
|
"choices": [
|
|
|
|
|
{
|
|
|
|
|
"index": 0,
|
|
|
|
|
"message": {"role": "assistant", "content": "ok"},
|
|
|
|
|
"finish_reason": "stop",
|
|
|
|
|
}
|
|
|
|
|
],
|
|
|
|
|
"usage": {"prompt_tokens": 500, "completion_tokens": 3, "total_tokens": 503},
|
|
|
|
|
},
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
proxy._retry_request = _fake_retry
|
|
|
|
|
verbose_schema_note = "schema annotation repeated for opencode tool definitions " * 50
|
|
|
|
|
tools = [
|
|
|
|
|
{
|
|
|
|
|
"type": "function",
|
|
|
|
|
"function": {
|
|
|
|
|
"name": "read_file",
|
|
|
|
|
"description": "read file helper",
|
|
|
|
|
"parameters": {
|
|
|
|
|
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
|
|
|
"title": "ReadFileParameters",
|
|
|
|
|
"type": "object",
|
|
|
|
|
"properties": {
|
|
|
|
|
"path": {
|
|
|
|
|
"type": "string",
|
|
|
|
|
"title": "Path",
|
|
|
|
|
"description": verbose_schema_note,
|
|
|
|
|
"examples": [verbose_schema_note],
|
|
|
|
|
}
|
|
|
|
|
},
|
|
|
|
|
"required": ["path"],
|
|
|
|
|
},
|
|
|
|
|
},
|
|
|
|
|
}
|
|
|
|
|
]
|
|
|
|
|
|
|
|
|
|
response = client.post(
|
|
|
|
|
"/v1/chat/completions",
|
|
|
|
|
headers={"authorization": "Bearer test-key"},
|
|
|
|
|
json={
|
|
|
|
|
"model": "gpt-4o-mini",
|
|
|
|
|
"messages": [{"role": "user", "content": "inspect this file"}],
|
|
|
|
|
"tools": tools,
|
|
|
|
|
},
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
assert response.status_code == 200
|
|
|
|
|
assert "openai:chat:tool_schema_compaction" in response.headers["x-headroom-transforms"]
|
|
|
|
|
assert int(response.headers["x-headroom-tokens-saved"]) > 0
|
|
|
|
|
sent_params = captured["body"]["tools"][0]["function"]["parameters"]
|
|
|
|
|
assert "$schema" not in sent_params
|
|
|
|
|
assert "title" not in sent_params
|
|
|
|
|
assert "title" not in sent_params["properties"]["path"]
|
|
|
|
|
assert "examples" not in sent_params["properties"]["path"]
|