mirror of
https://github.com/headroomlabs-ai/headroom.git
synced 2026-08-27 14:17:10 -04:00
## Description Extracts provider-neutral output effort decisions into a pure `output_effort_policy` module. `output_shaper` still owns request mutation and labels, while the rank comparisons, legacy thinking clamp, and OpenAI text verbosity eligibility now live behind small deterministic functions. Closes # ## Type of Change - [ ] Bug fix (non-breaking change that fixes an issue) - [ ] New feature (non-breaking change that adds functionality) - [ ] Breaking change (fix or feature that would cause existing functionality to change) - [ ] Documentation update - [ ] Performance improvement - [x] Code refactoring (no functional changes) ## Changes Made - Added `headroom.proxy.output_effort_policy` for effort lowering, legacy thinking budget clamping, and OpenAI text verbosity decisions. - Updated `output_shaper` to delegate those pure decisions while preserving existing labels and request mutation behavior. - Added focused policy tests for effort rank transitions, thinking clamp boundaries, and verbosity creation/lowering. - Included the current LiteLLM callback signature compatibility shim required for repo-wide mypy on main-based slices. ## Testing - [x] Unit tests pass (`pytest`) - [x] Linting passes (`ruff check .`) - [x] Type checking passes (`mypy headroom`) - [x] New tests added for new functionality - [ ] Manual testing performed ### Test Output ```text python -m pytest tests/test_output_effort_policy.py tests/test_output_shaper.py tests/test_litellm_callback.py -q 56 passed in 6.34s python -m ruff check . All checks passed! python -m ruff format --check . 1095 files already formatted python -m mypy headroom --ignore-missing-imports Success: no issues found in 409 source files gitleaks protect --staged --no-banner --redact no leaks found ``` ## Real Behavior Proof - Environment: Windows, Python 3.13.13, clean worktree based on `headroomlabs/main`. - Exact command / steps: targeted pytest, ruff, format check, repo-wide mypy, staged gitleaks scan. - Observed result: output effort policy/shaper/callback tests pass; static checks pass; no staged secrets detected. - Not tested: live provider calls; this slice preserves existing request mutation behavior and only moves pure policy decisions. ## Review Readiness - [x] I have performed a self-review - [x] This PR is ready for human review ## Checklist - [x] My code follows the project's style guidelines - [x] I have performed a self-review of my code - [x] I have commented my code, particularly in hard-to-understand areas - [ ] I have made corresponding changes to the documentation - [x] My changes generate no new warnings - [x] I have added tests that prove my fix is effective or that my feature works - [x] New and existing unit tests pass locally with my changes - [ ] I have updated the CHANGELOG.md if applicable ## Screenshots (if applicable) N/A ## Additional Notes Documentation and changelog updates are not applicable for this internal architecture slice. PR-specific GHAS checks will be monitored after opening.
58 lines
2.1 KiB
Python
58 lines
2.1 KiB
Python
"""Tests for pure output effort policy decisions."""
|
|
|
|
from __future__ import annotations
|
|
|
|
from headroom.proxy.output_effort_policy import (
|
|
LEGACY_THINKING_FLOOR,
|
|
can_create_openai_text_verbosity,
|
|
clamp_legacy_thinking_budget,
|
|
lower_effort_value,
|
|
lower_text_verbosity_value,
|
|
)
|
|
|
|
|
|
def test_lower_effort_value_lowers_known_higher_effort_to_target() -> None:
|
|
assert lower_effort_value("xhigh", "low") == "low"
|
|
assert lower_effort_value("max", "medium") == "medium"
|
|
|
|
|
|
def test_lower_effort_value_keeps_lower_equal_unknown_or_non_string_values() -> None:
|
|
assert lower_effort_value("low", "medium") is None
|
|
assert lower_effort_value("medium", "medium") is None
|
|
assert lower_effort_value("turbo", "low") is None
|
|
assert lower_effort_value("high", "turbo") is None
|
|
assert lower_effort_value(None, "low") is None
|
|
|
|
|
|
def test_clamp_legacy_thinking_budget_only_clamps_enabled_over_floor() -> None:
|
|
assert (
|
|
clamp_legacy_thinking_budget(
|
|
thinking_type="enabled",
|
|
budget_tokens=32_000,
|
|
)
|
|
== LEGACY_THINKING_FLOOR
|
|
)
|
|
assert (
|
|
clamp_legacy_thinking_budget(
|
|
thinking_type="enabled",
|
|
budget_tokens=LEGACY_THINKING_FLOOR,
|
|
)
|
|
is None
|
|
)
|
|
assert clamp_legacy_thinking_budget(thinking_type="adaptive", budget_tokens=32_000) is None
|
|
assert clamp_legacy_thinking_budget(thinking_type="enabled", budget_tokens="32000") is None
|
|
|
|
|
|
def test_can_create_openai_text_verbosity_only_for_gpt5_family() -> None:
|
|
assert can_create_openai_text_verbosity("gpt-5")
|
|
assert can_create_openai_text_verbosity("GPT-5.1")
|
|
assert not can_create_openai_text_verbosity("gpt-4o")
|
|
assert not can_create_openai_text_verbosity(None)
|
|
|
|
|
|
def test_lower_text_verbosity_value_lowers_existing_verbose_values() -> None:
|
|
assert lower_text_verbosity_value("medium") == "low"
|
|
assert lower_text_verbosity_value("high") == "low"
|
|
assert lower_text_verbosity_value("low") is None
|
|
assert lower_text_verbosity_value("chatty") is None
|
|
assert lower_text_verbosity_value(None) is None
|