mirror of
https://github.com/headroomlabs-ai/headroom.git
synced 2026-08-10 14:27:00 -04:00
## Description Extracts proxy semantic response-cache key normalization and hashing into a pure `semantic_cache_key_policy` module. `SemanticCache` keeps ownership of storage, locking, TTL, and LRU behavior while the deterministic cache-key formula is directly tested as a standalone policy. Closes # ## Type of Change - [ ] Bug fix (non-breaking change that fixes an issue) - [ ] New feature (non-breaking change that adds functionality) - [ ] Breaking change (fix or feature that would cause existing functionality to change) - [ ] Documentation update - [ ] Performance improvement - [x] Code refactoring (no functional changes) ## Changes Made - Added `headroom.proxy.semantic_cache_key_policy` with recursive `cache_control` stripping and semantic cache key hashing. - Updated `SemanticCache._compute_key` to delegate to the pure key policy while preserving its existing private wrapper contract. - Added direct policy tests for recursive annotation stripping, key stability, response-shaping distinctions, breakpoint movement, and wrapper parity. - Included the current LiteLLM callback signature compatibility shim required for repo-wide mypy on main-based slices. ## Testing - [x] Unit tests pass (`pytest`) - [x] Linting passes (`ruff check .`) - [x] Type checking passes (`mypy headroom`) - [x] New tests added for new functionality - [ ] Manual testing performed ### Test Output ```text python -m pytest tests/test_semantic_cache_key_policy.py tests/test_proxy_semantic_cache_key.py tests/test_litellm_callback.py -q 39 passed in 6.34s python -m ruff check . All checks passed! python -m ruff format --check . 1095 files already formatted python -m mypy headroom --ignore-missing-imports Success: no issues found in 409 source files gitleaks protect --staged --no-banner --redact no leaks found ``` ## Real Behavior Proof - Environment: Windows, Python 3.13.13, clean worktree based on `headroomlabs/main`. - Exact command / steps: targeted pytest, ruff, format check, repo-wide mypy, staged gitleaks scan. - Observed result: semantic cache key policy/cache/callback tests pass; static checks pass; no staged secrets detected. - Not tested: live proxy cache traffic; this slice preserves the existing cache wrapper and only moves pure key policy. ## Review Readiness - [x] I have performed a self-review - [x] This PR is ready for human review ## Checklist - [x] My code follows the project's style guidelines - [x] I have performed a self-review of my code - [x] I have commented my code, particularly in hard-to-understand areas - [ ] I have made corresponding changes to the documentation - [x] My changes generate no new warnings - [x] I have added tests that prove my fix is effective or that my feature works - [x] New and existing unit tests pass locally with my changes - [ ] I have updated the CHANGELOG.md if applicable ## Screenshots (if applicable) N/A ## Additional Notes Documentation and changelog updates are not applicable for this internal architecture slice. PR-specific GHAS checks will be monitored after opening.
71 lines
2.1 KiB
Python
71 lines
2.1 KiB
Python
"""Tests for pure proxy semantic cache key policy."""
|
|
|
|
from __future__ import annotations
|
|
|
|
from headroom.proxy.semantic_cache import SemanticCache
|
|
from headroom.proxy.semantic_cache_key_policy import (
|
|
compute_semantic_cache_key,
|
|
strip_cache_control,
|
|
)
|
|
|
|
MESSAGES = [{"role": "user", "content": "hello"}]
|
|
MODEL = "claude-haiku-4-5"
|
|
|
|
|
|
def test_strip_cache_control_recurses_through_dicts_and_lists() -> None:
|
|
payload = {
|
|
"system": [
|
|
{"type": "text", "text": "sys", "cache_control": {"type": "ephemeral"}},
|
|
{"nested": {"cache_control": "drop", "value": 1}},
|
|
],
|
|
"cache_control": "drop-root",
|
|
}
|
|
assert strip_cache_control(payload) == {
|
|
"system": [
|
|
{"type": "text", "text": "sys"},
|
|
{"nested": {"value": 1}},
|
|
]
|
|
}
|
|
|
|
|
|
def test_compute_semantic_cache_key_is_stable_for_identical_inputs() -> None:
|
|
kwargs = {"system": "sys", "tools": [{"name": "read"}], "temperature": 0.2}
|
|
assert compute_semantic_cache_key(MESSAGES, MODEL, **kwargs) == compute_semantic_cache_key(
|
|
MESSAGES,
|
|
MODEL,
|
|
**kwargs,
|
|
)
|
|
|
|
|
|
def test_compute_semantic_cache_key_distinguishes_response_shaping_fields() -> None:
|
|
assert compute_semantic_cache_key(
|
|
MESSAGES, MODEL, temperature=0.0
|
|
) != compute_semantic_cache_key(
|
|
MESSAGES,
|
|
MODEL,
|
|
temperature=1.0,
|
|
)
|
|
|
|
|
|
def test_compute_semantic_cache_key_ignores_moved_cache_control_breakpoints() -> None:
|
|
with_breakpoint = [{"type": "text", "text": "sys", "cache_control": {"type": "ephemeral"}}]
|
|
without_breakpoint = [{"type": "text", "text": "sys"}]
|
|
assert compute_semantic_cache_key(
|
|
MESSAGES,
|
|
MODEL,
|
|
system=with_breakpoint,
|
|
) == compute_semantic_cache_key(
|
|
MESSAGES,
|
|
MODEL,
|
|
system=without_breakpoint,
|
|
)
|
|
|
|
|
|
def test_semantic_cache_private_key_wrapper_delegates_to_policy() -> None:
|
|
cache = SemanticCache()
|
|
kwargs = {"system": "sys", "tools": [{"name": "read"}], "temperature": 0.2}
|
|
assert cache._compute_key(MESSAGES, MODEL, **kwargs) == compute_semantic_cache_key(
|
|
MESSAGES,
|
|
MODEL,
|
|
**kwargs,
|
|
)
|