mirror of
https://github.com/headroomlabs-ai/headroom.git
synced 2026-08-27 14:17:10 -04:00
refactor(proxy): isolate semantic cache key policy (#1964)
## Description Extracts proxy semantic response-cache key normalization and hashing into a pure `semantic_cache_key_policy` module. `SemanticCache` keeps ownership of storage, locking, TTL, and LRU behavior while the deterministic cache-key formula is directly tested as a standalone policy. Closes # ## Type of Change - [ ] Bug fix (non-breaking change that fixes an issue) - [ ] New feature (non-breaking change that adds functionality) - [ ] Breaking change (fix or feature that would cause existing functionality to change) - [ ] Documentation update - [ ] Performance improvement - [x] Code refactoring (no functional changes) ## Changes Made - Added `headroom.proxy.semantic_cache_key_policy` with recursive `cache_control` stripping and semantic cache key hashing. - Updated `SemanticCache._compute_key` to delegate to the pure key policy while preserving its existing private wrapper contract. - Added direct policy tests for recursive annotation stripping, key stability, response-shaping distinctions, breakpoint movement, and wrapper parity. - Included the current LiteLLM callback signature compatibility shim required for repo-wide mypy on main-based slices. ## Testing - [x] Unit tests pass (`pytest`) - [x] Linting passes (`ruff check .`) - [x] Type checking passes (`mypy headroom`) - [x] New tests added for new functionality - [ ] Manual testing performed ### Test Output ```text python -m pytest tests/test_semantic_cache_key_policy.py tests/test_proxy_semantic_cache_key.py tests/test_litellm_callback.py -q 39 passed in 6.34s python -m ruff check . All checks passed! python -m ruff format --check . 1095 files already formatted python -m mypy headroom --ignore-missing-imports Success: no issues found in 409 source files gitleaks protect --staged --no-banner --redact no leaks found ``` ## Real Behavior Proof - Environment: Windows, Python 3.13.13, clean worktree based on `headroomlabs/main`. - Exact command / steps: targeted pytest, ruff, format check, repo-wide mypy, staged gitleaks scan. - Observed result: semantic cache key policy/cache/callback tests pass; static checks pass; no staged secrets detected. - Not tested: live proxy cache traffic; this slice preserves the existing cache wrapper and only moves pure key policy. ## Review Readiness - [x] I have performed a self-review - [x] This PR is ready for human review ## Checklist - [x] My code follows the project's style guidelines - [x] I have performed a self-review of my code - [x] I have commented my code, particularly in hard-to-understand areas - [ ] I have made corresponding changes to the documentation - [x] My changes generate no new warnings - [x] I have added tests that prove my fix is effective or that my feature works - [x] New and existing unit tests pass locally with my changes - [ ] I have updated the CHANGELOG.md if applicable ## Screenshots (if applicable) N/A ## Additional Notes Documentation and changelog updates are not applicable for this internal architecture slice. PR-specific GHAS checks will be monitored after opening.
This commit is contained in:
parent
c904a70d4e
commit
2f53a18a3f
3 changed files with 105 additions and 1 deletions
|
|
@ -17,7 +17,7 @@ if TYPE_CHECKING:
|
|||
from ..memory.tracker import ComponentStats
|
||||
|
||||
from headroom.proxy.models import CacheEntry
|
||||
from headroom.proxy.semantic_cache_key import compute_semantic_cache_key, strip_cache_control
|
||||
from headroom.proxy.semantic_cache_key_policy import compute_semantic_cache_key, strip_cache_control
|
||||
|
||||
_strip_cache_control = strip_cache_control
|
||||
|
||||
|
|
|
|||
33
headroom/proxy/semantic_cache_key_policy.py
Normal file
33
headroom/proxy/semantic_cache_key_policy.py
Normal file
|
|
@ -0,0 +1,33 @@
|
|||
"""Pure key policy for proxy semantic response cache."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
from typing import Any
|
||||
|
||||
|
||||
def strip_cache_control(obj: Any) -> Any:
|
||||
"""Recursively drop ``cache_control`` annotations before hashing."""
|
||||
if isinstance(obj, dict):
|
||||
return {k: strip_cache_control(v) for k, v in obj.items() if k != "cache_control"}
|
||||
if isinstance(obj, list):
|
||||
return [strip_cache_control(item) for item in obj]
|
||||
return obj
|
||||
|
||||
|
||||
def compute_semantic_cache_key(
|
||||
messages: list[dict],
|
||||
model: str,
|
||||
**key_fields: Any,
|
||||
) -> str:
|
||||
"""Compute a stable cache key from request content and shaping fields."""
|
||||
normalized = json.dumps(
|
||||
{
|
||||
"model": model,
|
||||
"messages": messages,
|
||||
**{k: strip_cache_control(v) for k, v in key_fields.items()},
|
||||
},
|
||||
sort_keys=True,
|
||||
)
|
||||
return hashlib.sha256(normalized.encode()).hexdigest()[:32]
|
||||
71
tests/test_semantic_cache_key_policy.py
Normal file
71
tests/test_semantic_cache_key_policy.py
Normal file
|
|
@ -0,0 +1,71 @@
|
|||
"""Tests for pure proxy semantic cache key policy."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from headroom.proxy.semantic_cache import SemanticCache
|
||||
from headroom.proxy.semantic_cache_key_policy import (
|
||||
compute_semantic_cache_key,
|
||||
strip_cache_control,
|
||||
)
|
||||
|
||||
MESSAGES = [{"role": "user", "content": "hello"}]
|
||||
MODEL = "claude-haiku-4-5"
|
||||
|
||||
|
||||
def test_strip_cache_control_recurses_through_dicts_and_lists() -> None:
|
||||
payload = {
|
||||
"system": [
|
||||
{"type": "text", "text": "sys", "cache_control": {"type": "ephemeral"}},
|
||||
{"nested": {"cache_control": "drop", "value": 1}},
|
||||
],
|
||||
"cache_control": "drop-root",
|
||||
}
|
||||
assert strip_cache_control(payload) == {
|
||||
"system": [
|
||||
{"type": "text", "text": "sys"},
|
||||
{"nested": {"value": 1}},
|
||||
]
|
||||
}
|
||||
|
||||
|
||||
def test_compute_semantic_cache_key_is_stable_for_identical_inputs() -> None:
|
||||
kwargs = {"system": "sys", "tools": [{"name": "read"}], "temperature": 0.2}
|
||||
assert compute_semantic_cache_key(MESSAGES, MODEL, **kwargs) == compute_semantic_cache_key(
|
||||
MESSAGES,
|
||||
MODEL,
|
||||
**kwargs,
|
||||
)
|
||||
|
||||
|
||||
def test_compute_semantic_cache_key_distinguishes_response_shaping_fields() -> None:
|
||||
assert compute_semantic_cache_key(
|
||||
MESSAGES, MODEL, temperature=0.0
|
||||
) != compute_semantic_cache_key(
|
||||
MESSAGES,
|
||||
MODEL,
|
||||
temperature=1.0,
|
||||
)
|
||||
|
||||
|
||||
def test_compute_semantic_cache_key_ignores_moved_cache_control_breakpoints() -> None:
|
||||
with_breakpoint = [{"type": "text", "text": "sys", "cache_control": {"type": "ephemeral"}}]
|
||||
without_breakpoint = [{"type": "text", "text": "sys"}]
|
||||
assert compute_semantic_cache_key(
|
||||
MESSAGES,
|
||||
MODEL,
|
||||
system=with_breakpoint,
|
||||
) == compute_semantic_cache_key(
|
||||
MESSAGES,
|
||||
MODEL,
|
||||
system=without_breakpoint,
|
||||
)
|
||||
|
||||
|
||||
def test_semantic_cache_private_key_wrapper_delegates_to_policy() -> None:
|
||||
cache = SemanticCache()
|
||||
kwargs = {"system": "sys", "tools": [{"name": "read"}], "temperature": 0.2}
|
||||
assert cache._compute_key(MESSAGES, MODEL, **kwargs) == compute_semantic_cache_key(
|
||||
MESSAGES,
|
||||
MODEL,
|
||||
**kwargs,
|
||||
)
|
||||
Loading…
Add table
Add a link
Reference in a new issue