headroom/benchmarks/bench_transforms.py
chopratejas e4a41faa33 Fix all ruff lint and format errors for CI
- Fix E402: Move module-level imports to top of file
- Fix F401: Add noqa for availability check imports
- Fix F402: Rename loop variables shadowing imports
- Fix E722: Replace bare except with except Exception
- Fix B904: Add exception chaining (from e)
- Fix F811: Remove duplicate imports
- Fix B027: Add noqa for empty close() method
- Fix E741: Rename ambiguous variable l -> label
- Fix I001: Import sorting issues
- Apply ruff format to all 106 files

All 902 tests pass.
2026-01-10 15:33:44 -08:00

602 lines
16 KiB
Python

"""Transform benchmarks for Headroom SDK.
This module contains performance benchmarks for Headroom transforms:
- SmartCrusher: Statistical tool output compression
- CacheAligner: Cache-aligned prefix optimization
- RollingWindow: Token budget management
Performance Targets:
SmartCrusher:
- 100 items: < 2ms
- 1000 items: < 10ms
- 10000 items: < 100ms
CacheAligner:
- Date extraction: < 1ms
- Hash computation: < 0.5ms
RollingWindow:
- 50 turns: < 5ms
- 200 turns: < 20ms
Run with:
pytest benchmarks/bench_transforms.py --benchmark-only -v
"""
from __future__ import annotations
import json
import pytest
class TestSmartCrusherBenchmarks:
"""Benchmarks for SmartCrusher statistical compression.
SmartCrusher performs:
- Array analysis (field statistics, pattern detection)
- Change point detection for numeric fields
- Relevance scoring against query context
- Strategic sampling (first K, last K, errors, anomalies)
Expected performance:
- O(n) for array analysis
- O(n) for relevance scoring (BM25)
- Total: < 10ms for 1000 items
"""
@pytest.fixture
def crusher(self, smart_crusher_config):
"""Create SmartCrusher instance."""
from headroom.transforms.smart_crusher import SmartCrusher
return SmartCrusher(config=smart_crusher_config)
def test_compress_100_items(
self,
benchmark,
crusher,
mock_tokenizer,
items_100,
):
"""Benchmark crushing 100 search results.
Target: < 2ms
This is the typical size for API responses.
"""
messages = [
{"role": "system", "content": "You are a helpful assistant."},
{"role": "user", "content": "Search for users"},
{
"role": "tool",
"tool_call_id": "call_1",
"content": json.dumps(items_100),
},
]
result = benchmark(crusher.apply, messages, mock_tokenizer)
# Verify compression occurred
assert result.tokens_after < result.tokens_before
assert len(result.transforms_applied) > 0
def test_compress_1000_items(
self,
benchmark,
crusher,
mock_tokenizer,
items_1000,
):
"""Benchmark crushing 1000 search results.
Target: < 10ms
This tests larger tool outputs from extensive searches.
"""
messages = [
{"role": "system", "content": "You are a helpful assistant."},
{"role": "user", "content": "Search for all users"},
{
"role": "tool",
"tool_call_id": "call_1",
"content": json.dumps(items_1000),
},
]
result = benchmark(crusher.apply, messages, mock_tokenizer)
assert result.tokens_after < result.tokens_before
def test_compress_10000_items(
self,
benchmark,
crusher,
mock_tokenizer,
items_10000,
):
"""Benchmark crushing 10000 search results.
Target: < 100ms
Stress test for very large tool outputs.
"""
messages = [
{"role": "system", "content": "You are a helpful assistant."},
{"role": "user", "content": "Export all data"},
{
"role": "tool",
"tool_call_id": "call_1",
"content": json.dumps(items_10000),
},
]
result = benchmark(crusher.apply, messages, mock_tokenizer)
assert result.tokens_after < result.tokens_before
def test_analyze_log_entries(
self,
benchmark,
crusher,
mock_tokenizer,
log_entries_1000,
):
"""Benchmark crushing log entries (cluster detection).
Target: < 15ms
Tests cluster sampling strategy for repetitive logs.
"""
messages = [
{"role": "system", "content": "You are a helpful assistant."},
{"role": "user", "content": "Show recent logs"},
{
"role": "tool",
"tool_call_id": "call_1",
"content": json.dumps(log_entries_1000),
},
]
result = benchmark(crusher.apply, messages, mock_tokenizer)
assert result.tokens_after < result.tokens_before
def test_analyze_metrics_with_anomalies(
self,
benchmark,
crusher,
mock_tokenizer,
database_rows_1000,
):
"""Benchmark crushing metrics data (anomaly detection).
Target: < 15ms
Tests change point detection and anomaly preservation.
"""
messages = [
{"role": "system", "content": "You are a helpful assistant."},
{"role": "user", "content": "Get CPU metrics"},
{
"role": "tool",
"tool_call_id": "call_1",
"content": json.dumps(database_rows_1000),
},
]
result = benchmark(crusher.apply, messages, mock_tokenizer)
assert result.tokens_after < result.tokens_before
def test_multiple_tool_outputs(
self,
benchmark,
crusher,
mock_tokenizer,
items_100,
log_entries_100,
):
"""Benchmark crushing multiple tool outputs in one pass.
Target: < 5ms
Tests realistic scenario with multiple tool calls.
"""
messages = [
{"role": "system", "content": "You are a helpful assistant."},
{"role": "user", "content": "Search users and get logs"},
{
"role": "assistant",
"content": None,
"tool_calls": [
{
"id": "call_1",
"type": "function",
"function": {"name": "search", "arguments": "{}"},
},
{
"id": "call_2",
"type": "function",
"function": {"name": "logs", "arguments": "{}"},
},
],
},
{"role": "tool", "tool_call_id": "call_1", "content": json.dumps(items_100)},
{"role": "tool", "tool_call_id": "call_2", "content": json.dumps(log_entries_100)},
]
result = benchmark(crusher.apply, messages, mock_tokenizer)
assert result.tokens_after < result.tokens_before
class TestCacheAlignerBenchmarks:
"""Benchmarks for CacheAligner prefix optimization.
CacheAligner performs:
- Date pattern detection and extraction
- Whitespace normalization
- Stable prefix hash computation
Expected performance:
- Date extraction: < 1ms (regex matching)
- Hash computation: < 0.5ms (MD5)
- Total: < 2ms for typical system prompts
"""
@pytest.fixture
def aligner(self, cache_aligner_config):
"""Create CacheAligner instance."""
from headroom.transforms.cache_aligner import CacheAligner
return CacheAligner(config=cache_aligner_config)
def test_date_extraction(
self,
benchmark,
aligner,
mock_tokenizer,
messages_with_system_date,
):
"""Benchmark date extraction from system prompt.
Target: < 1ms
Tests regex-based date pattern matching.
"""
result = benchmark(aligner.apply, messages_with_system_date, mock_tokenizer)
# Verify date was extracted
assert "cache_align" in str(result.transforms_applied)
def test_hash_computation(
self,
benchmark,
aligner,
mock_tokenizer,
system_prompt_long,
):
"""Benchmark stable prefix hash computation.
Target: < 0.5ms
Tests hash stability for cache hit prediction.
"""
messages = [
{"role": "system", "content": system_prompt_long},
{"role": "user", "content": "Hello"},
]
result = benchmark(aligner.apply, messages, mock_tokenizer)
# Verify hash was computed
assert result.cache_metrics is not None
assert result.cache_metrics.stable_prefix_hash
def test_whitespace_normalization(
self,
benchmark,
aligner,
mock_tokenizer,
):
"""Benchmark whitespace normalization.
Target: < 0.5ms
Tests string processing for consistent formatting.
"""
messy_content = """You are a helpful assistant.
Current date: 2025-01-06
This has excessive whitespace.
And multiple blank lines."""
messages = [
{"role": "system", "content": messy_content},
{"role": "user", "content": "Hi"},
]
result = benchmark(aligner.apply, messages, mock_tokenizer)
assert result.messages[0]["content"] != messy_content # Was normalized
def test_long_system_prompt(
self,
benchmark,
aligner,
mock_tokenizer,
system_prompt_long,
):
"""Benchmark processing long system prompts.
Target: < 2ms
Tests performance with larger instruction sets.
"""
# Add date to trigger alignment
content_with_date = system_prompt_long + "\n\nCurrent date: 2025-01-06"
messages = [
{"role": "system", "content": content_with_date},
{"role": "user", "content": "Help me with code"},
]
result = benchmark(aligner.apply, messages, mock_tokenizer)
assert result.cache_metrics is not None
def test_multiple_system_messages(
self,
benchmark,
aligner,
mock_tokenizer,
):
"""Benchmark with multiple system messages.
Target: < 3ms
Tests edge case of multiple system prompts.
"""
messages = [
{
"role": "system",
"content": "You are a helpful assistant.\n\nCurrent date: 2025-01-06",
},
{"role": "system", "content": "Additional context: Technical support mode."},
{"role": "user", "content": "Hello"},
]
benchmark(aligner.apply, messages, mock_tokenizer)
class TestRollingWindowBenchmarks:
"""Benchmarks for RollingWindow token budget management.
RollingWindow performs:
- Token counting across all messages
- Tool unit identification (atomic drops)
- Protected index calculation
- Strategic message removal
Expected performance:
- 50 turns: < 5ms
- 200 turns: < 20ms
"""
@pytest.fixture
def window(self, rolling_window_config):
"""Create RollingWindow instance."""
from headroom.transforms.rolling_window import RollingWindow
return RollingWindow(config=rolling_window_config)
def test_window_50_turns(
self,
benchmark,
window,
mock_tokenizer,
conversation_50_turns,
):
"""Benchmark windowing 50-turn conversation.
Target: < 5ms
Tests typical long conversation scenario.
"""
# Set low limit to force dropping
result = benchmark(
window.apply,
conversation_50_turns,
mock_tokenizer,
model_limit=10000,
output_buffer=2000,
)
# Some messages should be dropped
assert len(result.messages) < len(conversation_50_turns)
def test_window_200_turns(
self,
benchmark,
window,
mock_tokenizer,
conversation_200_turns,
):
"""Benchmark windowing 200-turn conversation.
Target: < 20ms
Stress test for very long agentic sessions.
"""
result = benchmark(
window.apply,
conversation_200_turns,
mock_tokenizer,
model_limit=20000,
output_buffer=4000,
)
assert len(result.messages) < len(conversation_200_turns)
def test_window_no_drop_needed(
self,
benchmark,
window,
mock_tokenizer,
conversation_10_turns,
):
"""Benchmark when no dropping needed.
Target: < 1ms
Tests early-exit optimization.
"""
result = benchmark(
window.apply,
conversation_10_turns,
mock_tokenizer,
model_limit=1000000, # High limit, no dropping
output_buffer=4000,
)
assert len(result.messages) == len(conversation_10_turns)
def test_window_aggressive_drop(
self,
benchmark,
window,
mock_tokenizer,
conversation_50_turns,
):
"""Benchmark aggressive dropping (very low limit).
Target: < 5ms
Tests worst-case dropping scenario.
"""
result = benchmark(
window.apply,
conversation_50_turns,
mock_tokenizer,
model_limit=2000, # Very low
output_buffer=500,
)
# Should have dropped significantly
assert len(result.messages) < len(conversation_50_turns) // 2
def test_window_rag_context(
self,
benchmark,
window,
mock_tokenizer,
rag_conversation_20k,
):
"""Benchmark windowing RAG conversation.
Target: < 5ms
Tests handling of large context blocks.
"""
result = benchmark(
window.apply,
rag_conversation_20k,
mock_tokenizer,
model_limit=15000,
output_buffer=4000,
)
# RAG context preserved, later turns may be dropped
assert result.messages[0]["role"] == "system"
class TestTransformPipelineBenchmarks:
"""Benchmarks for full transform pipeline.
Tests the complete flow:
CacheAligner -> SmartCrusher -> RollingWindow
Expected performance:
- Simple conversation: < 5ms
- Agentic with tools: < 30ms
- Large RAG context: < 50ms
"""
@pytest.fixture
def mock_provider(self, mock_token_counter):
"""Create mock provider for pipeline."""
from unittest.mock import Mock
provider = Mock()
provider.get_token_counter.return_value = mock_token_counter
return provider
@pytest.fixture
def pipeline(
self, smart_crusher_config, cache_aligner_config, rolling_window_config, mock_provider
):
"""Create transform pipeline."""
from headroom.transforms.cache_aligner import CacheAligner
from headroom.transforms.pipeline import TransformPipeline
from headroom.transforms.rolling_window import RollingWindow
from headroom.transforms.smart_crusher import SmartCrusher
return TransformPipeline(
transforms=[
CacheAligner(cache_aligner_config),
SmartCrusher(smart_crusher_config),
RollingWindow(rolling_window_config),
],
provider=mock_provider,
)
def test_pipeline_simple(
self,
benchmark,
pipeline,
messages_with_system_date,
):
"""Benchmark pipeline on simple conversation.
Target: < 5ms
Tests minimal overhead scenario.
"""
benchmark(
pipeline.apply,
messages_with_system_date,
"benchmark-model",
model_limit=100000,
)
def test_pipeline_agentic(
self,
benchmark,
pipeline,
conversation_50_turns,
):
"""Benchmark pipeline on agentic conversation.
Target: < 30ms
Tests realistic agentic workload.
"""
result = benchmark(
pipeline.apply,
conversation_50_turns,
"benchmark-model",
model_limit=50000,
)
assert result.tokens_after < result.tokens_before
def test_pipeline_rag(
self,
benchmark,
pipeline,
rag_conversation_20k,
):
"""Benchmark pipeline on RAG conversation.
Target: < 50ms
Tests large context handling.
Note: CacheAligner may add small markers (e.g., "[Dynamic Context]"),
so we allow up to 1% token increase.
"""
result = benchmark(
pipeline.apply,
rag_conversation_20k,
"benchmark-model",
model_limit=30000,
)
# Allow for small overhead from cache alignment markers
assert result.tokens_after <= result.tokens_before * 1.01