mirror of
https://github.com/headroomlabs-ai/headroom.git
synced 2026-08-27 14:17:10 -04:00
- Fix E402: Move module-level imports to top of file - Fix F401: Add noqa for availability check imports - Fix F402: Rename loop variables shadowing imports - Fix E722: Replace bare except with except Exception - Fix B904: Add exception chaining (from e) - Fix F811: Remove duplicate imports - Fix B027: Add noqa for empty close() method - Fix E741: Rename ambiguous variable l -> label - Fix I001: Import sorting issues - Apply ruff format to all 106 files All 902 tests pass.
602 lines
16 KiB
Python
602 lines
16 KiB
Python
"""Transform benchmarks for Headroom SDK.
|
|
|
|
This module contains performance benchmarks for Headroom transforms:
|
|
- SmartCrusher: Statistical tool output compression
|
|
- CacheAligner: Cache-aligned prefix optimization
|
|
- RollingWindow: Token budget management
|
|
|
|
Performance Targets:
|
|
SmartCrusher:
|
|
- 100 items: < 2ms
|
|
- 1000 items: < 10ms
|
|
- 10000 items: < 100ms
|
|
|
|
CacheAligner:
|
|
- Date extraction: < 1ms
|
|
- Hash computation: < 0.5ms
|
|
|
|
RollingWindow:
|
|
- 50 turns: < 5ms
|
|
- 200 turns: < 20ms
|
|
|
|
Run with:
|
|
pytest benchmarks/bench_transforms.py --benchmark-only -v
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
|
|
import pytest
|
|
|
|
|
|
class TestSmartCrusherBenchmarks:
|
|
"""Benchmarks for SmartCrusher statistical compression.
|
|
|
|
SmartCrusher performs:
|
|
- Array analysis (field statistics, pattern detection)
|
|
- Change point detection for numeric fields
|
|
- Relevance scoring against query context
|
|
- Strategic sampling (first K, last K, errors, anomalies)
|
|
|
|
Expected performance:
|
|
- O(n) for array analysis
|
|
- O(n) for relevance scoring (BM25)
|
|
- Total: < 10ms for 1000 items
|
|
"""
|
|
|
|
@pytest.fixture
|
|
def crusher(self, smart_crusher_config):
|
|
"""Create SmartCrusher instance."""
|
|
from headroom.transforms.smart_crusher import SmartCrusher
|
|
|
|
return SmartCrusher(config=smart_crusher_config)
|
|
|
|
def test_compress_100_items(
|
|
self,
|
|
benchmark,
|
|
crusher,
|
|
mock_tokenizer,
|
|
items_100,
|
|
):
|
|
"""Benchmark crushing 100 search results.
|
|
|
|
Target: < 2ms
|
|
This is the typical size for API responses.
|
|
"""
|
|
messages = [
|
|
{"role": "system", "content": "You are a helpful assistant."},
|
|
{"role": "user", "content": "Search for users"},
|
|
{
|
|
"role": "tool",
|
|
"tool_call_id": "call_1",
|
|
"content": json.dumps(items_100),
|
|
},
|
|
]
|
|
|
|
result = benchmark(crusher.apply, messages, mock_tokenizer)
|
|
|
|
# Verify compression occurred
|
|
assert result.tokens_after < result.tokens_before
|
|
assert len(result.transforms_applied) > 0
|
|
|
|
def test_compress_1000_items(
|
|
self,
|
|
benchmark,
|
|
crusher,
|
|
mock_tokenizer,
|
|
items_1000,
|
|
):
|
|
"""Benchmark crushing 1000 search results.
|
|
|
|
Target: < 10ms
|
|
This tests larger tool outputs from extensive searches.
|
|
"""
|
|
messages = [
|
|
{"role": "system", "content": "You are a helpful assistant."},
|
|
{"role": "user", "content": "Search for all users"},
|
|
{
|
|
"role": "tool",
|
|
"tool_call_id": "call_1",
|
|
"content": json.dumps(items_1000),
|
|
},
|
|
]
|
|
|
|
result = benchmark(crusher.apply, messages, mock_tokenizer)
|
|
|
|
assert result.tokens_after < result.tokens_before
|
|
|
|
def test_compress_10000_items(
|
|
self,
|
|
benchmark,
|
|
crusher,
|
|
mock_tokenizer,
|
|
items_10000,
|
|
):
|
|
"""Benchmark crushing 10000 search results.
|
|
|
|
Target: < 100ms
|
|
Stress test for very large tool outputs.
|
|
"""
|
|
messages = [
|
|
{"role": "system", "content": "You are a helpful assistant."},
|
|
{"role": "user", "content": "Export all data"},
|
|
{
|
|
"role": "tool",
|
|
"tool_call_id": "call_1",
|
|
"content": json.dumps(items_10000),
|
|
},
|
|
]
|
|
|
|
result = benchmark(crusher.apply, messages, mock_tokenizer)
|
|
|
|
assert result.tokens_after < result.tokens_before
|
|
|
|
def test_analyze_log_entries(
|
|
self,
|
|
benchmark,
|
|
crusher,
|
|
mock_tokenizer,
|
|
log_entries_1000,
|
|
):
|
|
"""Benchmark crushing log entries (cluster detection).
|
|
|
|
Target: < 15ms
|
|
Tests cluster sampling strategy for repetitive logs.
|
|
"""
|
|
messages = [
|
|
{"role": "system", "content": "You are a helpful assistant."},
|
|
{"role": "user", "content": "Show recent logs"},
|
|
{
|
|
"role": "tool",
|
|
"tool_call_id": "call_1",
|
|
"content": json.dumps(log_entries_1000),
|
|
},
|
|
]
|
|
|
|
result = benchmark(crusher.apply, messages, mock_tokenizer)
|
|
|
|
assert result.tokens_after < result.tokens_before
|
|
|
|
def test_analyze_metrics_with_anomalies(
|
|
self,
|
|
benchmark,
|
|
crusher,
|
|
mock_tokenizer,
|
|
database_rows_1000,
|
|
):
|
|
"""Benchmark crushing metrics data (anomaly detection).
|
|
|
|
Target: < 15ms
|
|
Tests change point detection and anomaly preservation.
|
|
"""
|
|
messages = [
|
|
{"role": "system", "content": "You are a helpful assistant."},
|
|
{"role": "user", "content": "Get CPU metrics"},
|
|
{
|
|
"role": "tool",
|
|
"tool_call_id": "call_1",
|
|
"content": json.dumps(database_rows_1000),
|
|
},
|
|
]
|
|
|
|
result = benchmark(crusher.apply, messages, mock_tokenizer)
|
|
|
|
assert result.tokens_after < result.tokens_before
|
|
|
|
def test_multiple_tool_outputs(
|
|
self,
|
|
benchmark,
|
|
crusher,
|
|
mock_tokenizer,
|
|
items_100,
|
|
log_entries_100,
|
|
):
|
|
"""Benchmark crushing multiple tool outputs in one pass.
|
|
|
|
Target: < 5ms
|
|
Tests realistic scenario with multiple tool calls.
|
|
"""
|
|
messages = [
|
|
{"role": "system", "content": "You are a helpful assistant."},
|
|
{"role": "user", "content": "Search users and get logs"},
|
|
{
|
|
"role": "assistant",
|
|
"content": None,
|
|
"tool_calls": [
|
|
{
|
|
"id": "call_1",
|
|
"type": "function",
|
|
"function": {"name": "search", "arguments": "{}"},
|
|
},
|
|
{
|
|
"id": "call_2",
|
|
"type": "function",
|
|
"function": {"name": "logs", "arguments": "{}"},
|
|
},
|
|
],
|
|
},
|
|
{"role": "tool", "tool_call_id": "call_1", "content": json.dumps(items_100)},
|
|
{"role": "tool", "tool_call_id": "call_2", "content": json.dumps(log_entries_100)},
|
|
]
|
|
|
|
result = benchmark(crusher.apply, messages, mock_tokenizer)
|
|
|
|
assert result.tokens_after < result.tokens_before
|
|
|
|
|
|
class TestCacheAlignerBenchmarks:
|
|
"""Benchmarks for CacheAligner prefix optimization.
|
|
|
|
CacheAligner performs:
|
|
- Date pattern detection and extraction
|
|
- Whitespace normalization
|
|
- Stable prefix hash computation
|
|
|
|
Expected performance:
|
|
- Date extraction: < 1ms (regex matching)
|
|
- Hash computation: < 0.5ms (MD5)
|
|
- Total: < 2ms for typical system prompts
|
|
"""
|
|
|
|
@pytest.fixture
|
|
def aligner(self, cache_aligner_config):
|
|
"""Create CacheAligner instance."""
|
|
from headroom.transforms.cache_aligner import CacheAligner
|
|
|
|
return CacheAligner(config=cache_aligner_config)
|
|
|
|
def test_date_extraction(
|
|
self,
|
|
benchmark,
|
|
aligner,
|
|
mock_tokenizer,
|
|
messages_with_system_date,
|
|
):
|
|
"""Benchmark date extraction from system prompt.
|
|
|
|
Target: < 1ms
|
|
Tests regex-based date pattern matching.
|
|
"""
|
|
result = benchmark(aligner.apply, messages_with_system_date, mock_tokenizer)
|
|
|
|
# Verify date was extracted
|
|
assert "cache_align" in str(result.transforms_applied)
|
|
|
|
def test_hash_computation(
|
|
self,
|
|
benchmark,
|
|
aligner,
|
|
mock_tokenizer,
|
|
system_prompt_long,
|
|
):
|
|
"""Benchmark stable prefix hash computation.
|
|
|
|
Target: < 0.5ms
|
|
Tests hash stability for cache hit prediction.
|
|
"""
|
|
messages = [
|
|
{"role": "system", "content": system_prompt_long},
|
|
{"role": "user", "content": "Hello"},
|
|
]
|
|
|
|
result = benchmark(aligner.apply, messages, mock_tokenizer)
|
|
|
|
# Verify hash was computed
|
|
assert result.cache_metrics is not None
|
|
assert result.cache_metrics.stable_prefix_hash
|
|
|
|
def test_whitespace_normalization(
|
|
self,
|
|
benchmark,
|
|
aligner,
|
|
mock_tokenizer,
|
|
):
|
|
"""Benchmark whitespace normalization.
|
|
|
|
Target: < 0.5ms
|
|
Tests string processing for consistent formatting.
|
|
"""
|
|
messy_content = """You are a helpful assistant.
|
|
|
|
Current date: 2025-01-06
|
|
|
|
|
|
This has excessive whitespace.
|
|
|
|
|
|
And multiple blank lines."""
|
|
|
|
messages = [
|
|
{"role": "system", "content": messy_content},
|
|
{"role": "user", "content": "Hi"},
|
|
]
|
|
|
|
result = benchmark(aligner.apply, messages, mock_tokenizer)
|
|
|
|
assert result.messages[0]["content"] != messy_content # Was normalized
|
|
|
|
def test_long_system_prompt(
|
|
self,
|
|
benchmark,
|
|
aligner,
|
|
mock_tokenizer,
|
|
system_prompt_long,
|
|
):
|
|
"""Benchmark processing long system prompts.
|
|
|
|
Target: < 2ms
|
|
Tests performance with larger instruction sets.
|
|
"""
|
|
# Add date to trigger alignment
|
|
content_with_date = system_prompt_long + "\n\nCurrent date: 2025-01-06"
|
|
|
|
messages = [
|
|
{"role": "system", "content": content_with_date},
|
|
{"role": "user", "content": "Help me with code"},
|
|
]
|
|
|
|
result = benchmark(aligner.apply, messages, mock_tokenizer)
|
|
|
|
assert result.cache_metrics is not None
|
|
|
|
def test_multiple_system_messages(
|
|
self,
|
|
benchmark,
|
|
aligner,
|
|
mock_tokenizer,
|
|
):
|
|
"""Benchmark with multiple system messages.
|
|
|
|
Target: < 3ms
|
|
Tests edge case of multiple system prompts.
|
|
"""
|
|
messages = [
|
|
{
|
|
"role": "system",
|
|
"content": "You are a helpful assistant.\n\nCurrent date: 2025-01-06",
|
|
},
|
|
{"role": "system", "content": "Additional context: Technical support mode."},
|
|
{"role": "user", "content": "Hello"},
|
|
]
|
|
|
|
benchmark(aligner.apply, messages, mock_tokenizer)
|
|
|
|
|
|
class TestRollingWindowBenchmarks:
|
|
"""Benchmarks for RollingWindow token budget management.
|
|
|
|
RollingWindow performs:
|
|
- Token counting across all messages
|
|
- Tool unit identification (atomic drops)
|
|
- Protected index calculation
|
|
- Strategic message removal
|
|
|
|
Expected performance:
|
|
- 50 turns: < 5ms
|
|
- 200 turns: < 20ms
|
|
"""
|
|
|
|
@pytest.fixture
|
|
def window(self, rolling_window_config):
|
|
"""Create RollingWindow instance."""
|
|
from headroom.transforms.rolling_window import RollingWindow
|
|
|
|
return RollingWindow(config=rolling_window_config)
|
|
|
|
def test_window_50_turns(
|
|
self,
|
|
benchmark,
|
|
window,
|
|
mock_tokenizer,
|
|
conversation_50_turns,
|
|
):
|
|
"""Benchmark windowing 50-turn conversation.
|
|
|
|
Target: < 5ms
|
|
Tests typical long conversation scenario.
|
|
"""
|
|
# Set low limit to force dropping
|
|
result = benchmark(
|
|
window.apply,
|
|
conversation_50_turns,
|
|
mock_tokenizer,
|
|
model_limit=10000,
|
|
output_buffer=2000,
|
|
)
|
|
|
|
# Some messages should be dropped
|
|
assert len(result.messages) < len(conversation_50_turns)
|
|
|
|
def test_window_200_turns(
|
|
self,
|
|
benchmark,
|
|
window,
|
|
mock_tokenizer,
|
|
conversation_200_turns,
|
|
):
|
|
"""Benchmark windowing 200-turn conversation.
|
|
|
|
Target: < 20ms
|
|
Stress test for very long agentic sessions.
|
|
"""
|
|
result = benchmark(
|
|
window.apply,
|
|
conversation_200_turns,
|
|
mock_tokenizer,
|
|
model_limit=20000,
|
|
output_buffer=4000,
|
|
)
|
|
|
|
assert len(result.messages) < len(conversation_200_turns)
|
|
|
|
def test_window_no_drop_needed(
|
|
self,
|
|
benchmark,
|
|
window,
|
|
mock_tokenizer,
|
|
conversation_10_turns,
|
|
):
|
|
"""Benchmark when no dropping needed.
|
|
|
|
Target: < 1ms
|
|
Tests early-exit optimization.
|
|
"""
|
|
result = benchmark(
|
|
window.apply,
|
|
conversation_10_turns,
|
|
mock_tokenizer,
|
|
model_limit=1000000, # High limit, no dropping
|
|
output_buffer=4000,
|
|
)
|
|
|
|
assert len(result.messages) == len(conversation_10_turns)
|
|
|
|
def test_window_aggressive_drop(
|
|
self,
|
|
benchmark,
|
|
window,
|
|
mock_tokenizer,
|
|
conversation_50_turns,
|
|
):
|
|
"""Benchmark aggressive dropping (very low limit).
|
|
|
|
Target: < 5ms
|
|
Tests worst-case dropping scenario.
|
|
"""
|
|
result = benchmark(
|
|
window.apply,
|
|
conversation_50_turns,
|
|
mock_tokenizer,
|
|
model_limit=2000, # Very low
|
|
output_buffer=500,
|
|
)
|
|
|
|
# Should have dropped significantly
|
|
assert len(result.messages) < len(conversation_50_turns) // 2
|
|
|
|
def test_window_rag_context(
|
|
self,
|
|
benchmark,
|
|
window,
|
|
mock_tokenizer,
|
|
rag_conversation_20k,
|
|
):
|
|
"""Benchmark windowing RAG conversation.
|
|
|
|
Target: < 5ms
|
|
Tests handling of large context blocks.
|
|
"""
|
|
result = benchmark(
|
|
window.apply,
|
|
rag_conversation_20k,
|
|
mock_tokenizer,
|
|
model_limit=15000,
|
|
output_buffer=4000,
|
|
)
|
|
|
|
# RAG context preserved, later turns may be dropped
|
|
assert result.messages[0]["role"] == "system"
|
|
|
|
|
|
class TestTransformPipelineBenchmarks:
|
|
"""Benchmarks for full transform pipeline.
|
|
|
|
Tests the complete flow:
|
|
CacheAligner -> SmartCrusher -> RollingWindow
|
|
|
|
Expected performance:
|
|
- Simple conversation: < 5ms
|
|
- Agentic with tools: < 30ms
|
|
- Large RAG context: < 50ms
|
|
"""
|
|
|
|
@pytest.fixture
|
|
def mock_provider(self, mock_token_counter):
|
|
"""Create mock provider for pipeline."""
|
|
from unittest.mock import Mock
|
|
|
|
provider = Mock()
|
|
provider.get_token_counter.return_value = mock_token_counter
|
|
return provider
|
|
|
|
@pytest.fixture
|
|
def pipeline(
|
|
self, smart_crusher_config, cache_aligner_config, rolling_window_config, mock_provider
|
|
):
|
|
"""Create transform pipeline."""
|
|
from headroom.transforms.cache_aligner import CacheAligner
|
|
from headroom.transforms.pipeline import TransformPipeline
|
|
from headroom.transforms.rolling_window import RollingWindow
|
|
from headroom.transforms.smart_crusher import SmartCrusher
|
|
|
|
return TransformPipeline(
|
|
transforms=[
|
|
CacheAligner(cache_aligner_config),
|
|
SmartCrusher(smart_crusher_config),
|
|
RollingWindow(rolling_window_config),
|
|
],
|
|
provider=mock_provider,
|
|
)
|
|
|
|
def test_pipeline_simple(
|
|
self,
|
|
benchmark,
|
|
pipeline,
|
|
messages_with_system_date,
|
|
):
|
|
"""Benchmark pipeline on simple conversation.
|
|
|
|
Target: < 5ms
|
|
Tests minimal overhead scenario.
|
|
"""
|
|
benchmark(
|
|
pipeline.apply,
|
|
messages_with_system_date,
|
|
"benchmark-model",
|
|
model_limit=100000,
|
|
)
|
|
|
|
def test_pipeline_agentic(
|
|
self,
|
|
benchmark,
|
|
pipeline,
|
|
conversation_50_turns,
|
|
):
|
|
"""Benchmark pipeline on agentic conversation.
|
|
|
|
Target: < 30ms
|
|
Tests realistic agentic workload.
|
|
"""
|
|
result = benchmark(
|
|
pipeline.apply,
|
|
conversation_50_turns,
|
|
"benchmark-model",
|
|
model_limit=50000,
|
|
)
|
|
|
|
assert result.tokens_after < result.tokens_before
|
|
|
|
def test_pipeline_rag(
|
|
self,
|
|
benchmark,
|
|
pipeline,
|
|
rag_conversation_20k,
|
|
):
|
|
"""Benchmark pipeline on RAG conversation.
|
|
|
|
Target: < 50ms
|
|
Tests large context handling.
|
|
|
|
Note: CacheAligner may add small markers (e.g., "[Dynamic Context]"),
|
|
so we allow up to 1% token increase.
|
|
"""
|
|
result = benchmark(
|
|
pipeline.apply,
|
|
rag_conversation_20k,
|
|
"benchmark-model",
|
|
model_limit=30000,
|
|
)
|
|
|
|
# Allow for small overhead from cache alignment markers
|
|
assert result.tokens_after <= result.tokens_before * 1.01
|