2026-01-06 23:16:58 -08:00
|
|
|
"""Transform benchmarks for Headroom SDK.
|
|
|
|
|
|
|
|
|
|
This module contains performance benchmarks for Headroom transforms:
|
|
|
|
|
- SmartCrusher: Statistical tool output compression
|
|
|
|
|
- CacheAligner: Cache-aligned prefix optimization
|
|
|
|
|
|
|
|
|
|
Performance Targets:
|
|
|
|
|
SmartCrusher:
|
|
|
|
|
- 100 items: < 2ms
|
|
|
|
|
- 1000 items: < 10ms
|
|
|
|
|
- 10000 items: < 100ms
|
|
|
|
|
|
|
|
|
|
CacheAligner:
|
|
|
|
|
- Date extraction: < 1ms
|
|
|
|
|
- Hash computation: < 0.5ms
|
|
|
|
|
|
|
|
|
|
Run with:
|
|
|
|
|
pytest benchmarks/bench_transforms.py --benchmark-only -v
|
|
|
|
|
"""
|
|
|
|
|
|
|
|
|
|
from __future__ import annotations
|
|
|
|
|
|
|
|
|
|
import json
|
|
|
|
|
|
|
|
|
|
import pytest
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
class TestSmartCrusherBenchmarks:
|
|
|
|
|
"""Benchmarks for SmartCrusher statistical compression.
|
|
|
|
|
|
|
|
|
|
SmartCrusher performs:
|
|
|
|
|
- Array analysis (field statistics, pattern detection)
|
|
|
|
|
- Change point detection for numeric fields
|
|
|
|
|
- Relevance scoring against query context
|
|
|
|
|
- Strategic sampling (first K, last K, errors, anomalies)
|
|
|
|
|
|
|
|
|
|
Expected performance:
|
|
|
|
|
- O(n) for array analysis
|
|
|
|
|
- O(n) for relevance scoring (BM25)
|
|
|
|
|
- Total: < 10ms for 1000 items
|
|
|
|
|
"""
|
|
|
|
|
|
|
|
|
|
@pytest.fixture
|
|
|
|
|
def crusher(self, smart_crusher_config):
|
|
|
|
|
"""Create SmartCrusher instance."""
|
|
|
|
|
from headroom.transforms.smart_crusher import SmartCrusher
|
|
|
|
|
|
|
|
|
|
return SmartCrusher(config=smart_crusher_config)
|
|
|
|
|
|
|
|
|
|
def test_compress_100_items(
|
|
|
|
|
self,
|
|
|
|
|
benchmark,
|
|
|
|
|
crusher,
|
|
|
|
|
mock_tokenizer,
|
|
|
|
|
items_100,
|
|
|
|
|
):
|
|
|
|
|
"""Benchmark crushing 100 search results.
|
|
|
|
|
|
|
|
|
|
Target: < 2ms
|
|
|
|
|
This is the typical size for API responses.
|
|
|
|
|
"""
|
|
|
|
|
messages = [
|
|
|
|
|
{"role": "system", "content": "You are a helpful assistant."},
|
|
|
|
|
{"role": "user", "content": "Search for users"},
|
|
|
|
|
{
|
|
|
|
|
"role": "tool",
|
|
|
|
|
"tool_call_id": "call_1",
|
|
|
|
|
"content": json.dumps(items_100),
|
|
|
|
|
},
|
|
|
|
|
]
|
|
|
|
|
|
|
|
|
|
result = benchmark(crusher.apply, messages, mock_tokenizer)
|
|
|
|
|
|
|
|
|
|
# Verify compression occurred
|
|
|
|
|
assert result.tokens_after < result.tokens_before
|
|
|
|
|
assert len(result.transforms_applied) > 0
|
|
|
|
|
|
|
|
|
|
def test_compress_1000_items(
|
|
|
|
|
self,
|
|
|
|
|
benchmark,
|
|
|
|
|
crusher,
|
|
|
|
|
mock_tokenizer,
|
|
|
|
|
items_1000,
|
|
|
|
|
):
|
|
|
|
|
"""Benchmark crushing 1000 search results.
|
|
|
|
|
|
|
|
|
|
Target: < 10ms
|
|
|
|
|
This tests larger tool outputs from extensive searches.
|
|
|
|
|
"""
|
|
|
|
|
messages = [
|
|
|
|
|
{"role": "system", "content": "You are a helpful assistant."},
|
|
|
|
|
{"role": "user", "content": "Search for all users"},
|
|
|
|
|
{
|
|
|
|
|
"role": "tool",
|
|
|
|
|
"tool_call_id": "call_1",
|
|
|
|
|
"content": json.dumps(items_1000),
|
|
|
|
|
},
|
|
|
|
|
]
|
|
|
|
|
|
|
|
|
|
result = benchmark(crusher.apply, messages, mock_tokenizer)
|
|
|
|
|
|
|
|
|
|
assert result.tokens_after < result.tokens_before
|
|
|
|
|
|
|
|
|
|
def test_compress_10000_items(
|
|
|
|
|
self,
|
|
|
|
|
benchmark,
|
|
|
|
|
crusher,
|
|
|
|
|
mock_tokenizer,
|
|
|
|
|
items_10000,
|
|
|
|
|
):
|
|
|
|
|
"""Benchmark crushing 10000 search results.
|
|
|
|
|
|
|
|
|
|
Target: < 100ms
|
|
|
|
|
Stress test for very large tool outputs.
|
|
|
|
|
"""
|
|
|
|
|
messages = [
|
|
|
|
|
{"role": "system", "content": "You are a helpful assistant."},
|
|
|
|
|
{"role": "user", "content": "Export all data"},
|
|
|
|
|
{
|
|
|
|
|
"role": "tool",
|
|
|
|
|
"tool_call_id": "call_1",
|
|
|
|
|
"content": json.dumps(items_10000),
|
|
|
|
|
},
|
|
|
|
|
]
|
|
|
|
|
|
|
|
|
|
result = benchmark(crusher.apply, messages, mock_tokenizer)
|
|
|
|
|
|
|
|
|
|
assert result.tokens_after < result.tokens_before
|
|
|
|
|
|
|
|
|
|
def test_analyze_log_entries(
|
|
|
|
|
self,
|
|
|
|
|
benchmark,
|
|
|
|
|
crusher,
|
|
|
|
|
mock_tokenizer,
|
|
|
|
|
log_entries_1000,
|
|
|
|
|
):
|
|
|
|
|
"""Benchmark crushing log entries (cluster detection).
|
|
|
|
|
|
|
|
|
|
Target: < 15ms
|
|
|
|
|
Tests cluster sampling strategy for repetitive logs.
|
|
|
|
|
"""
|
|
|
|
|
messages = [
|
|
|
|
|
{"role": "system", "content": "You are a helpful assistant."},
|
|
|
|
|
{"role": "user", "content": "Show recent logs"},
|
|
|
|
|
{
|
|
|
|
|
"role": "tool",
|
|
|
|
|
"tool_call_id": "call_1",
|
|
|
|
|
"content": json.dumps(log_entries_1000),
|
|
|
|
|
},
|
|
|
|
|
]
|
|
|
|
|
|
|
|
|
|
result = benchmark(crusher.apply, messages, mock_tokenizer)
|
|
|
|
|
|
|
|
|
|
assert result.tokens_after < result.tokens_before
|
|
|
|
|
|
|
|
|
|
def test_analyze_metrics_with_anomalies(
|
|
|
|
|
self,
|
|
|
|
|
benchmark,
|
|
|
|
|
crusher,
|
|
|
|
|
mock_tokenizer,
|
|
|
|
|
database_rows_1000,
|
|
|
|
|
):
|
|
|
|
|
"""Benchmark crushing metrics data (anomaly detection).
|
|
|
|
|
|
|
|
|
|
Target: < 15ms
|
|
|
|
|
Tests change point detection and anomaly preservation.
|
|
|
|
|
"""
|
|
|
|
|
messages = [
|
|
|
|
|
{"role": "system", "content": "You are a helpful assistant."},
|
|
|
|
|
{"role": "user", "content": "Get CPU metrics"},
|
|
|
|
|
{
|
|
|
|
|
"role": "tool",
|
|
|
|
|
"tool_call_id": "call_1",
|
|
|
|
|
"content": json.dumps(database_rows_1000),
|
|
|
|
|
},
|
|
|
|
|
]
|
|
|
|
|
|
|
|
|
|
result = benchmark(crusher.apply, messages, mock_tokenizer)
|
|
|
|
|
|
|
|
|
|
assert result.tokens_after < result.tokens_before
|
|
|
|
|
|
|
|
|
|
def test_multiple_tool_outputs(
|
|
|
|
|
self,
|
|
|
|
|
benchmark,
|
|
|
|
|
crusher,
|
|
|
|
|
mock_tokenizer,
|
|
|
|
|
items_100,
|
|
|
|
|
log_entries_100,
|
|
|
|
|
):
|
|
|
|
|
"""Benchmark crushing multiple tool outputs in one pass.
|
|
|
|
|
|
|
|
|
|
Target: < 5ms
|
|
|
|
|
Tests realistic scenario with multiple tool calls.
|
|
|
|
|
"""
|
|
|
|
|
messages = [
|
|
|
|
|
{"role": "system", "content": "You are a helpful assistant."},
|
|
|
|
|
{"role": "user", "content": "Search users and get logs"},
|
|
|
|
|
{
|
|
|
|
|
"role": "assistant",
|
|
|
|
|
"content": None,
|
|
|
|
|
"tool_calls": [
|
2026-01-10 15:33:44 -08:00
|
|
|
{
|
|
|
|
|
"id": "call_1",
|
|
|
|
|
"type": "function",
|
|
|
|
|
"function": {"name": "search", "arguments": "{}"},
|
|
|
|
|
},
|
|
|
|
|
{
|
|
|
|
|
"id": "call_2",
|
|
|
|
|
"type": "function",
|
|
|
|
|
"function": {"name": "logs", "arguments": "{}"},
|
|
|
|
|
},
|
2026-01-06 23:16:58 -08:00
|
|
|
],
|
|
|
|
|
},
|
|
|
|
|
{"role": "tool", "tool_call_id": "call_1", "content": json.dumps(items_100)},
|
|
|
|
|
{"role": "tool", "tool_call_id": "call_2", "content": json.dumps(log_entries_100)},
|
|
|
|
|
]
|
|
|
|
|
|
|
|
|
|
result = benchmark(crusher.apply, messages, mock_tokenizer)
|
|
|
|
|
|
|
|
|
|
assert result.tokens_after < result.tokens_before
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
class TestCacheAlignerBenchmarks:
|
|
|
|
|
"""Benchmarks for CacheAligner prefix optimization.
|
|
|
|
|
|
|
|
|
|
CacheAligner performs:
|
|
|
|
|
- Date pattern detection and extraction
|
|
|
|
|
- Whitespace normalization
|
|
|
|
|
- Stable prefix hash computation
|
|
|
|
|
|
|
|
|
|
Expected performance:
|
|
|
|
|
- Date extraction: < 1ms (regex matching)
|
|
|
|
|
- Hash computation: < 0.5ms (MD5)
|
|
|
|
|
- Total: < 2ms for typical system prompts
|
|
|
|
|
"""
|
|
|
|
|
|
|
|
|
|
@pytest.fixture
|
|
|
|
|
def aligner(self, cache_aligner_config):
|
|
|
|
|
"""Create CacheAligner instance."""
|
|
|
|
|
from headroom.transforms.cache_aligner import CacheAligner
|
|
|
|
|
|
|
|
|
|
return CacheAligner(config=cache_aligner_config)
|
|
|
|
|
|
|
|
|
|
def test_date_extraction(
|
|
|
|
|
self,
|
|
|
|
|
benchmark,
|
|
|
|
|
aligner,
|
|
|
|
|
mock_tokenizer,
|
|
|
|
|
messages_with_system_date,
|
|
|
|
|
):
|
|
|
|
|
"""Benchmark date extraction from system prompt.
|
|
|
|
|
|
|
|
|
|
Target: < 1ms
|
|
|
|
|
Tests regex-based date pattern matching.
|
|
|
|
|
"""
|
|
|
|
|
result = benchmark(aligner.apply, messages_with_system_date, mock_tokenizer)
|
|
|
|
|
|
|
|
|
|
# Verify date was extracted
|
|
|
|
|
assert "cache_align" in str(result.transforms_applied)
|
|
|
|
|
|
|
|
|
|
def test_hash_computation(
|
|
|
|
|
self,
|
|
|
|
|
benchmark,
|
|
|
|
|
aligner,
|
|
|
|
|
mock_tokenizer,
|
|
|
|
|
system_prompt_long,
|
|
|
|
|
):
|
|
|
|
|
"""Benchmark stable prefix hash computation.
|
|
|
|
|
|
|
|
|
|
Target: < 0.5ms
|
|
|
|
|
Tests hash stability for cache hit prediction.
|
|
|
|
|
"""
|
|
|
|
|
messages = [
|
|
|
|
|
{"role": "system", "content": system_prompt_long},
|
|
|
|
|
{"role": "user", "content": "Hello"},
|
|
|
|
|
]
|
|
|
|
|
|
|
|
|
|
result = benchmark(aligner.apply, messages, mock_tokenizer)
|
|
|
|
|
|
|
|
|
|
# Verify hash was computed
|
|
|
|
|
assert result.cache_metrics is not None
|
|
|
|
|
assert result.cache_metrics.stable_prefix_hash
|
|
|
|
|
|
|
|
|
|
def test_whitespace_normalization(
|
|
|
|
|
self,
|
|
|
|
|
benchmark,
|
|
|
|
|
aligner,
|
|
|
|
|
mock_tokenizer,
|
|
|
|
|
):
|
|
|
|
|
"""Benchmark whitespace normalization.
|
|
|
|
|
|
|
|
|
|
Target: < 0.5ms
|
|
|
|
|
Tests string processing for consistent formatting.
|
|
|
|
|
"""
|
|
|
|
|
messy_content = """You are a helpful assistant.
|
|
|
|
|
|
|
|
|
|
Current date: 2025-01-06
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
This has excessive whitespace.
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
And multiple blank lines."""
|
|
|
|
|
|
|
|
|
|
messages = [
|
|
|
|
|
{"role": "system", "content": messy_content},
|
|
|
|
|
{"role": "user", "content": "Hi"},
|
|
|
|
|
]
|
|
|
|
|
|
|
|
|
|
result = benchmark(aligner.apply, messages, mock_tokenizer)
|
|
|
|
|
|
|
|
|
|
assert result.messages[0]["content"] != messy_content # Was normalized
|
|
|
|
|
|
|
|
|
|
def test_long_system_prompt(
|
|
|
|
|
self,
|
|
|
|
|
benchmark,
|
|
|
|
|
aligner,
|
|
|
|
|
mock_tokenizer,
|
|
|
|
|
system_prompt_long,
|
|
|
|
|
):
|
|
|
|
|
"""Benchmark processing long system prompts.
|
|
|
|
|
|
|
|
|
|
Target: < 2ms
|
|
|
|
|
Tests performance with larger instruction sets.
|
|
|
|
|
"""
|
|
|
|
|
# Add date to trigger alignment
|
|
|
|
|
content_with_date = system_prompt_long + "\n\nCurrent date: 2025-01-06"
|
|
|
|
|
|
|
|
|
|
messages = [
|
|
|
|
|
{"role": "system", "content": content_with_date},
|
|
|
|
|
{"role": "user", "content": "Help me with code"},
|
|
|
|
|
]
|
|
|
|
|
|
|
|
|
|
result = benchmark(aligner.apply, messages, mock_tokenizer)
|
|
|
|
|
|
|
|
|
|
assert result.cache_metrics is not None
|
|
|
|
|
|
|
|
|
|
def test_multiple_system_messages(
|
|
|
|
|
self,
|
|
|
|
|
benchmark,
|
|
|
|
|
aligner,
|
|
|
|
|
mock_tokenizer,
|
|
|
|
|
):
|
|
|
|
|
"""Benchmark with multiple system messages.
|
|
|
|
|
|
|
|
|
|
Target: < 3ms
|
|
|
|
|
Tests edge case of multiple system prompts.
|
|
|
|
|
"""
|
|
|
|
|
messages = [
|
2026-01-10 15:33:44 -08:00
|
|
|
{
|
|
|
|
|
"role": "system",
|
|
|
|
|
"content": "You are a helpful assistant.\n\nCurrent date: 2025-01-06",
|
|
|
|
|
},
|
2026-01-06 23:16:58 -08:00
|
|
|
{"role": "system", "content": "Additional context: Technical support mode."},
|
|
|
|
|
{"role": "user", "content": "Hello"},
|
|
|
|
|
]
|
|
|
|
|
|
|
|
|
|
benchmark(aligner.apply, messages, mock_tokenizer)
|
|
|
|
|
|
|
|
|
|
|
fix: B2 — live-zone block dispatcher skeleton
Phase B step 2 of the live-zone-only realignment. Replaces PR-A1's
unconditional "passthrough" stub with a real dispatcher that
inspects the Anthropic /v1/messages body, identifies the live zone
(latest user message at index >= frozen_message_count), and routes
each block to a per-type compressor. PR-B2 wires every per-type
compressor to a no-op, so the dispatcher returns
LiveZoneOutcome::NoChange on every call — bytes-in == bytes-out.
PR-B3+ replaces the no-ops with SmartCrusher, Log, Search, Diff,
and Code compressors.
Adds:
- crates/headroom-core/src/transforms/live_zone.rs — public API:
- `compress_live_zone(body, frozen_message_count, AuthMode)`
- `LiveZoneOutcome::{NoChange, Modified}`
- `CompressionManifest` with per-block outcomes (message_index,
block_index, block_type, BlockAction).
- `BlockAction::{NoOpSkeleton, Excluded { reason }}`. The
HOT_ZONE_BLOCK_TYPES list (`tool_use`, `thinking`,
`redacted_thinking`, `compaction`) excludes blocks even when
they appear in the latest user message.
- `AuthMode::{Payg, OAuth, Subscription}` — accepted but unused
in B2; PR-F2 wires the auth-mode gate.
- 12 unit tests pin: empty messages, no messages field, invalid
JSON, latest user message selection, frozen_count respect,
hot-zone block exclusion, string-shaped content, no user msg
in live zone, AuthMode no-op, NoChange contract, manifest
counters, frozen-count clamping.
- crates/headroom-proxy/src/compression/live_zone_anthropic.rs —
new entry point. `compress_anthropic_request` parses the body,
resolves frozen_count via `resolve_frozen_count` (PR-A4 helper),
dispatches via `compress_live_zone`, and returns
`Outcome::NoCompression` on PR-B2 success / `Outcome::Passthrough
{ reason: NotJson | NoMessages | ModeOff }` on body-shape /
policy issues. Six unit tests pin: mode_off short-circuit, no
messages field, invalid JSON, valid body NoCompression,
empty body, cache_control disabled.
Modifies:
- compression/mod.rs — re-exports `compress_anthropic_request` from
`live_zone_anthropic` instead of `anthropic`. The old anthropic
module is reduced to the `resolve_frozen_count` helper only
(not deleted, because its CacheControlAutoFrozen-policy gate is
reused).
- proxy.rs — passes `state.config.cache_control_auto_frozen` into
the dispatcher. Drops the obsolete "live_zone reserved for
Phase B" warning that PR-A1 emitted on every request.
- compression/anthropic.rs — pruned to the resolve_frozen_count
helper plus its tests. The PR-A1 passthrough stub
`compress_anthropic_request` is gone (live_zone_anthropic owns
the name now).
- config.rs — `compression_mode` doc updated to reflect the wired
dispatcher (no longer "reserved for Phase B").
- tests/integration_compression.rs — `compression_decision_logged`
pins the new log contract (`decision="no_change"`,
`reason="no_op_skeleton_pr_b2"`, plus manifest fields
`frozen_message_count`, `messages_total`, `live_zone_blocks`).
Asserts the obsolete Phase A warning is NOT emitted.
- proxy.rs no longer imports CompressionMode (only used inside the
retired warning).
Benchmark cleanup (B1 leftovers that surfaced now):
- benchmarks/proxy_mode_benchmark.py + claude_session_mode_benchmark.py:
drop `intelligent_context=False` arg from ProxyConfig (the field
was retired in B1; tests/test_proxy_mode_benchmark.py and
tests/test_claude_session_mode_benchmark.py imported these
factories and started failing).
- benchmarks/bench_transforms.py: delete TestRollingWindowBenchmarks
class; rewire TestTransformPipelineBenchmarks fixture without
RollingWindow.
- benchmarks/conftest.py: drop rolling_window_config fixture.
- benchmarks/run_benchmarks.py: drop the `window` suite + table
rows referencing RollingWindow.
Cache-safety invariant:
- PR-B2 dispatcher never mutates body bytes (no-op skeleton). The
proxy forwards the original buffered bytes byte-equal. Phase A's
SHA-256 fixtures pin this.
- `passthrough_mode_live_zone_currently_passthrough_byte_equal_sha256`
retitled comment to reflect the dispatcher being live but
no-op.
Acceptance:
- cargo build --workspace + clippy + fmt: green.
- cargo test --workspace --exclude headroom-py: all green
(777 + 12 new live_zone + 6 new live_zone_anthropic tests).
- pytest: 4678 passed, 240 skipped, 0 failed.
- Anthropic decision log includes manifest fields per the
observability contract documented in
REALIGNMENT/02-architecture.md.
Per-PR-B2 plan: REALIGNMENT/04-phase-B-live-zone.md.
2026-05-02 12:45:43 -07:00
|
|
|
# RollingWindow benchmarks were retired in PR-B1 along with the
|
|
|
|
|
# RollingWindow transform itself. Live-zone-only compression
|
|
|
|
|
# (PR-B2..B7) does not drop messages, so message-count-based
|
|
|
|
|
# benchmarks no longer have a baseline to measure. Phase B's own
|
|
|
|
|
# performance suite lives alongside the live-zone dispatcher.
|
2026-01-06 23:16:58 -08:00
|
|
|
|
|
|
|
|
|
|
|
|
|
class TestTransformPipelineBenchmarks:
|
|
|
|
|
"""Benchmarks for full transform pipeline.
|
|
|
|
|
|
|
|
|
|
Tests the complete flow:
|
docs: sync README + benchmarks with code (drop retired IntelligentContext/RollingWindow) (#1545)
## Description
Sync the docs with the code after the live-zone realignment. The
`IntelligentContextManager` (ICM), `RollingWindow`, and scoring modules
were deleted in PR #350 (May 2026), but the README and benchmark
docstrings still advertised them as live, and an example still imported
the deleted module (broken on run). This fixes the README + benchmarks
and removes the dead example.
I validated the README against the code with three parallel
static-analysis sub-agents (features/architecture,
CLI/extras/wrap-matrix, public API/integrations). Most of the README
checked out accurate; only the items below were stale/wrong.
Closes #
## Type of Change
- [ ] Bug fix (non-breaking change that fixes an issue)
- [ ] New feature (non-breaking change that adds functionality)
- [ ] Breaking change (fix or feature that would cause existing
functionality to change)
- [x] Documentation update
- [ ] Performance improvement
- [ ] Code refactoring (no functional changes)
## Changes Made
- README: removed the `IntelligentContext` bullet and
`IntelligentContext / RollingWindow` from the transforms list (both
deleted in PR #350).
- README: standardized `Kompress-base` -> `Kompress-v2-base` to match
the HF model id `chopratejas/kompress-v2-base` and the existing badges
(diagram re-aligned).
- README: corrected the CodeCompressor language list to match the
`CodeLanguage` enum (added TS, C, Perl).
- README: softened the unanchored "6 algorithms" tagline to
"content-aware compressors".
- README: Cortex Code is library-mode only — there is no `headroom wrap
cortex`, so the compatibility-matrix row no longer shows a wrap
checkmark.
- Deleted `examples/test_intelligent_context_toin_ccr.py` — it imported
the deleted `IntelligentContextManager` (ImportError on run) and is
unreferenced.
- Removed stale `RollingWindow` mentions from benchmark
docstrings/comments (`benchmarks/__init__.py`, `bench_transforms.py`,
`bench_latency.py`, `scenarios/conversations.py`); the accurate PR-B1
retirement comment is kept.
## Testing
- [ ] Unit tests pass (`pytest`) — N/A, docs/docstring + example
deletion only
- [x] Linting passes — `ruff check` clean on all changed benchmark files
- [ ] Type checking passes — N/A (no type-relevant changes)
- [ ] New tests added — N/A
- [x] Manual testing performed — see Real Behavior Proof
### Test Output
```text
$ ruff check benchmarks/__init__.py benchmarks/bench_transforms.py benchmarks/bench_latency.py benchmarks/scenarios/conversations.py
All checks passed!
# stale refs remaining in README/benchmarks (excluding accurate retirement notes):
$ grep -rn "IntelligentContext|RollingWindow|Kompress-base" README.md benchmarks/ | grep -v retire
(only benchmarks/bench_transforms.py:362 — the accurate PR-B1 retirement comment)
# deleted example is unreferenced anywhere:
$ grep -rn "test_intelligent_context_toin_ccr" --include=*.md --include=*.yml --include=*.py .
(no hits)
```
## Real Behavior Proof
- Environment: macOS (darwin, arm64), Python 3.12 `.venv`, ruff 0.14.x,
repo at branch `docs/sync-readme-with-code` off latest `main`.
- Exact command / steps: (1) three parallel sub-agents
grep/Read-validated README claims vs `headroom/`, `pyproject.toml`,
`sdk/typescript/`; (2) directly verified each flagged mismatch
(`CodeLanguage` enum, `HF_MODEL_ID`, absence of
`IntelligentContext`/`RollingWindow` classes); (3) confirmed the example
imports a deleted module and is unreferenced; (4) `ruff check` on
changed benchmark files; (5) re-grepped README + benchmarks for any
remaining stale refs.
- Observed result: README and benchmark docstrings now match the code;
the only surviving `RollingWindow` string is the accurate retirement
comment; the broken example is removed; ruff passes; the ASCII
architecture diagram still aligns after the `Kompress-v2-base` rename.
- Not tested: rendering of the README on GitHub/PyPI (text-only change);
the separate `docs/content/` and `wiki/` doc sets (see Additional Notes
— out of scope for this PR).
## Review Readiness
- [x] I have performed a self-review
- [x] This PR is ready for human review
## Checklist
- [x] My code follows the project's style guidelines
- [x] I have performed a self-review of my code
- [x] I have commented my code, particularly in hard-to-understand areas
- [x] I have made corresponding changes to the documentation
- [x] My changes generate no new warnings
- [ ] I have added tests that prove my fix is effective — N/A
(docs/example cleanup)
- [x] New and existing unit tests pass locally with my changes
- [ ] I have updated the CHANGELOG.md — N/A (Release Please
auto-generates from the conventional commit)
## Additional Notes
**Larger related finding (NOT in this PR):** the published docs site
(`docs/content/docs/*.mdx`) and the `wiki/*.md` set still document
`IntelligentContextManager`, `RollingWindow`, `RollingWindowConfig`,
`IntelligentContextConfig`, and `ScoringWeights` as live API — with
`from headroom import RollingWindow` / `from headroom.transforms import
IntelligentContextManager` code examples that would `ImportError`. It is
half-migrated (a couple of `.mdx` files already note "removed in 0.9.x"
while neighbors still teach it as current). This is ~15 files and the
fixes require rewriting examples to the live-zone model, not just
deletions — recommended as a focused follow-up PR rather than bundling
it here.
2026-06-28 22:36:41 -07:00
|
|
|
CacheAligner -> SmartCrusher
|
2026-01-06 23:16:58 -08:00
|
|
|
|
|
|
|
|
Expected performance:
|
|
|
|
|
- Simple conversation: < 5ms
|
|
|
|
|
- Agentic with tools: < 30ms
|
|
|
|
|
- Large RAG context: < 50ms
|
|
|
|
|
"""
|
|
|
|
|
|
|
|
|
|
@pytest.fixture
|
|
|
|
|
def mock_provider(self, mock_token_counter):
|
|
|
|
|
"""Create mock provider for pipeline."""
|
|
|
|
|
from unittest.mock import Mock
|
|
|
|
|
|
|
|
|
|
provider = Mock()
|
|
|
|
|
provider.get_token_counter.return_value = mock_token_counter
|
|
|
|
|
return provider
|
|
|
|
|
|
|
|
|
|
@pytest.fixture
|
fix: B2 — live-zone block dispatcher skeleton
Phase B step 2 of the live-zone-only realignment. Replaces PR-A1's
unconditional "passthrough" stub with a real dispatcher that
inspects the Anthropic /v1/messages body, identifies the live zone
(latest user message at index >= frozen_message_count), and routes
each block to a per-type compressor. PR-B2 wires every per-type
compressor to a no-op, so the dispatcher returns
LiveZoneOutcome::NoChange on every call — bytes-in == bytes-out.
PR-B3+ replaces the no-ops with SmartCrusher, Log, Search, Diff,
and Code compressors.
Adds:
- crates/headroom-core/src/transforms/live_zone.rs — public API:
- `compress_live_zone(body, frozen_message_count, AuthMode)`
- `LiveZoneOutcome::{NoChange, Modified}`
- `CompressionManifest` with per-block outcomes (message_index,
block_index, block_type, BlockAction).
- `BlockAction::{NoOpSkeleton, Excluded { reason }}`. The
HOT_ZONE_BLOCK_TYPES list (`tool_use`, `thinking`,
`redacted_thinking`, `compaction`) excludes blocks even when
they appear in the latest user message.
- `AuthMode::{Payg, OAuth, Subscription}` — accepted but unused
in B2; PR-F2 wires the auth-mode gate.
- 12 unit tests pin: empty messages, no messages field, invalid
JSON, latest user message selection, frozen_count respect,
hot-zone block exclusion, string-shaped content, no user msg
in live zone, AuthMode no-op, NoChange contract, manifest
counters, frozen-count clamping.
- crates/headroom-proxy/src/compression/live_zone_anthropic.rs —
new entry point. `compress_anthropic_request` parses the body,
resolves frozen_count via `resolve_frozen_count` (PR-A4 helper),
dispatches via `compress_live_zone`, and returns
`Outcome::NoCompression` on PR-B2 success / `Outcome::Passthrough
{ reason: NotJson | NoMessages | ModeOff }` on body-shape /
policy issues. Six unit tests pin: mode_off short-circuit, no
messages field, invalid JSON, valid body NoCompression,
empty body, cache_control disabled.
Modifies:
- compression/mod.rs — re-exports `compress_anthropic_request` from
`live_zone_anthropic` instead of `anthropic`. The old anthropic
module is reduced to the `resolve_frozen_count` helper only
(not deleted, because its CacheControlAutoFrozen-policy gate is
reused).
- proxy.rs — passes `state.config.cache_control_auto_frozen` into
the dispatcher. Drops the obsolete "live_zone reserved for
Phase B" warning that PR-A1 emitted on every request.
- compression/anthropic.rs — pruned to the resolve_frozen_count
helper plus its tests. The PR-A1 passthrough stub
`compress_anthropic_request` is gone (live_zone_anthropic owns
the name now).
- config.rs — `compression_mode` doc updated to reflect the wired
dispatcher (no longer "reserved for Phase B").
- tests/integration_compression.rs — `compression_decision_logged`
pins the new log contract (`decision="no_change"`,
`reason="no_op_skeleton_pr_b2"`, plus manifest fields
`frozen_message_count`, `messages_total`, `live_zone_blocks`).
Asserts the obsolete Phase A warning is NOT emitted.
- proxy.rs no longer imports CompressionMode (only used inside the
retired warning).
Benchmark cleanup (B1 leftovers that surfaced now):
- benchmarks/proxy_mode_benchmark.py + claude_session_mode_benchmark.py:
drop `intelligent_context=False` arg from ProxyConfig (the field
was retired in B1; tests/test_proxy_mode_benchmark.py and
tests/test_claude_session_mode_benchmark.py imported these
factories and started failing).
- benchmarks/bench_transforms.py: delete TestRollingWindowBenchmarks
class; rewire TestTransformPipelineBenchmarks fixture without
RollingWindow.
- benchmarks/conftest.py: drop rolling_window_config fixture.
- benchmarks/run_benchmarks.py: drop the `window` suite + table
rows referencing RollingWindow.
Cache-safety invariant:
- PR-B2 dispatcher never mutates body bytes (no-op skeleton). The
proxy forwards the original buffered bytes byte-equal. Phase A's
SHA-256 fixtures pin this.
- `passthrough_mode_live_zone_currently_passthrough_byte_equal_sha256`
retitled comment to reflect the dispatcher being live but
no-op.
Acceptance:
- cargo build --workspace + clippy + fmt: green.
- cargo test --workspace --exclude headroom-py: all green
(777 + 12 new live_zone + 6 new live_zone_anthropic tests).
- pytest: 4678 passed, 240 skipped, 0 failed.
- Anthropic decision log includes manifest fields per the
observability contract documented in
REALIGNMENT/02-architecture.md.
Per-PR-B2 plan: REALIGNMENT/04-phase-B-live-zone.md.
2026-05-02 12:45:43 -07:00
|
|
|
def pipeline(self, smart_crusher_config, cache_aligner_config, mock_provider):
|
|
|
|
|
"""Create transform pipeline.
|
|
|
|
|
|
|
|
|
|
PR-B1 retired RollingWindow; the live-zone-only architecture
|
|
|
|
|
runs CacheAligner → SmartCrusher (followed by ContentRouter
|
|
|
|
|
in production, omitted here to keep the fixture pure-stage).
|
|
|
|
|
"""
|
2026-01-06 23:16:58 -08:00
|
|
|
from headroom.transforms.cache_aligner import CacheAligner
|
2026-01-10 15:33:44 -08:00
|
|
|
from headroom.transforms.pipeline import TransformPipeline
|
|
|
|
|
from headroom.transforms.smart_crusher import SmartCrusher
|
2026-01-06 23:16:58 -08:00
|
|
|
|
|
|
|
|
return TransformPipeline(
|
|
|
|
|
transforms=[
|
|
|
|
|
CacheAligner(cache_aligner_config),
|
|
|
|
|
SmartCrusher(smart_crusher_config),
|
|
|
|
|
],
|
|
|
|
|
provider=mock_provider,
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
def test_pipeline_simple(
|
|
|
|
|
self,
|
|
|
|
|
benchmark,
|
|
|
|
|
pipeline,
|
|
|
|
|
messages_with_system_date,
|
|
|
|
|
):
|
|
|
|
|
"""Benchmark pipeline on simple conversation.
|
|
|
|
|
|
|
|
|
|
Target: < 5ms
|
|
|
|
|
Tests minimal overhead scenario.
|
|
|
|
|
"""
|
|
|
|
|
benchmark(
|
|
|
|
|
pipeline.apply,
|
|
|
|
|
messages_with_system_date,
|
|
|
|
|
"benchmark-model",
|
|
|
|
|
model_limit=100000,
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
def test_pipeline_agentic(
|
|
|
|
|
self,
|
|
|
|
|
benchmark,
|
|
|
|
|
pipeline,
|
|
|
|
|
conversation_50_turns,
|
|
|
|
|
):
|
|
|
|
|
"""Benchmark pipeline on agentic conversation.
|
|
|
|
|
|
|
|
|
|
Target: < 30ms
|
|
|
|
|
Tests realistic agentic workload.
|
|
|
|
|
"""
|
|
|
|
|
result = benchmark(
|
|
|
|
|
pipeline.apply,
|
|
|
|
|
conversation_50_turns,
|
|
|
|
|
"benchmark-model",
|
|
|
|
|
model_limit=50000,
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
assert result.tokens_after < result.tokens_before
|
|
|
|
|
|
|
|
|
|
def test_pipeline_rag(
|
|
|
|
|
self,
|
|
|
|
|
benchmark,
|
|
|
|
|
pipeline,
|
|
|
|
|
rag_conversation_20k,
|
|
|
|
|
):
|
|
|
|
|
"""Benchmark pipeline on RAG conversation.
|
|
|
|
|
|
|
|
|
|
Target: < 50ms
|
|
|
|
|
Tests large context handling.
|
|
|
|
|
|
|
|
|
|
Note: CacheAligner may add small markers (e.g., "[Dynamic Context]"),
|
|
|
|
|
so we allow up to 1% token increase.
|
|
|
|
|
"""
|
|
|
|
|
result = benchmark(
|
|
|
|
|
pipeline.apply,
|
|
|
|
|
rag_conversation_20k,
|
|
|
|
|
"benchmark-model",
|
|
|
|
|
model_limit=30000,
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
# Allow for small overhead from cache alignment markers
|
|
|
|
|
assert result.tokens_after <= result.tokens_before * 1.01
|