mirror of
https://github.com/headroomlabs-ai/headroom.git
synced 2026-08-27 14:17:10 -04:00
Compression Summaries:
- New: headroom/transforms/compression_summary.py
- summarize_dropped_items(): categorizes compressed JSON items by
field values (status, type, level, etc.), highlights errors/failures
- summarize_compressed_code(): extracts function names from AST
signatures (language-agnostic: Python, JS, Go, Rust, Java)
- Newline-safe: strips \n from field values to keep markers single-line
- SmartCrusher: CCR markers include categorical summary of dropped items
e.g. "[500 items compressed to 20. Omitted: 87 passed, 2 failed.
Retrieve more: hash=abc123. Expires in 5m.]"
- CodeCompressor: CCR markers list compressed function names from AST
e.g. "[180 tokens compressed. 5 bodies compressed: authenticate().
Retrieve more: hash=abc123. Expires in 5m.]"
- Markers include TTL so LLM knows retrieval window
- Summary escapes { } to prevent .format() crashes
- Uses index-based dropped detection (not id()) for .copy() correctness
Proxy Response Headers:
- Anthropic, OpenAI, and Gemini handlers inject x-headroom-tokens-*
headers for SaaS metering
Multi-Provider Passthrough Routing:
- Detect x-goog-api-key (Gemini) and api-key (Azure OpenAI)
- X-Headroom-Base-URL for explicit upstream URL override
Dockerfile: add build-essential + g++ for hnswlib compilation
Bump version to 0.3.5
Tests: 27 new tests (unit, eval, integration with real API, tool invocation)
271 lines
6.8 KiB
TOML
271 lines
6.8 KiB
TOML
[build-system]
|
|
requires = ["hatchling"]
|
|
build-backend = "hatchling.build"
|
|
|
|
[project]
|
|
name = "headroom-ai"
|
|
version = "0.3.5"
|
|
description = "The Context Optimization Layer for LLM Applications - Cut costs by 50-90%"
|
|
readme = "README.md"
|
|
license = "Apache-2.0"
|
|
requires-python = ">=3.10"
|
|
authors = [
|
|
{ name = "Headroom Contributors" }
|
|
]
|
|
maintainers = [
|
|
{ name = "Headroom Contributors" }
|
|
]
|
|
keywords = [
|
|
"llm",
|
|
"openai",
|
|
"anthropic",
|
|
"claude",
|
|
"gpt",
|
|
"context",
|
|
"token",
|
|
"optimization",
|
|
"compression",
|
|
"caching",
|
|
"proxy",
|
|
"ai",
|
|
"machine-learning",
|
|
]
|
|
classifiers = [
|
|
"Development Status :: 4 - Beta",
|
|
"Intended Audience :: Developers",
|
|
"License :: OSI Approved :: Apache Software License",
|
|
"Operating System :: OS Independent",
|
|
"Programming Language :: Python :: 3",
|
|
"Programming Language :: Python :: 3.10",
|
|
"Programming Language :: Python :: 3.11",
|
|
"Programming Language :: Python :: 3.12",
|
|
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
"Topic :: Software Development :: Libraries :: Python Modules",
|
|
"Typing :: Typed",
|
|
]
|
|
dependencies = [
|
|
"tiktoken>=0.5.0",
|
|
"pydantic>=2.0.0",
|
|
"openai>=2.14.0",
|
|
"sentence-transformers>=5.2.0",
|
|
"litellm>=1.0.0",
|
|
"accelerate>=1.12.0",
|
|
"sentencepiece>=0.2.1",
|
|
"protobuf>=6.33.4",
|
|
"semantic-router>=0.1.12",
|
|
"pillow>=10.0.0", # Image processing for compression
|
|
"datasets>=4.5.0",
|
|
"hnswlib>=0.8.0", # HNSW vector index for memory system
|
|
"click>=8.1.0", # CLI framework
|
|
"rich>=13.0.0", # Rich terminal output
|
|
]
|
|
|
|
[project.optional-dependencies]
|
|
# Semantic relevance scoring with embeddings
|
|
relevance = [
|
|
"sentence-transformers>=2.2.0",
|
|
"numpy>=1.24.0",
|
|
]
|
|
# Proxy server
|
|
proxy = [
|
|
"fastapi>=0.100.0",
|
|
"uvicorn>=0.23.0",
|
|
"httpx[http2]>=0.24.0", # http2 extra enables h2 for HTTP/2 support
|
|
]
|
|
# Report generation
|
|
reports = [
|
|
"jinja2>=3.0.0",
|
|
]
|
|
# ML-based compression (LLMLingua-2)
|
|
llmlingua = [
|
|
"llmlingua>=0.2.0",
|
|
"torch>=2.0.0",
|
|
"transformers>=4.30.0",
|
|
]
|
|
# AST-based code compression (tree-sitter)
|
|
code = [
|
|
"tree-sitter-language-pack>=0.10.0",
|
|
]
|
|
# Agno agent framework integration
|
|
agno = [
|
|
"agno>=1.0.0",
|
|
]
|
|
# AWS Strands Agents SDK integration
|
|
strands = [
|
|
"strands-agents>=0.1.0",
|
|
]
|
|
# MCP server for Claude Code integration (CCR without API access)
|
|
mcp = [
|
|
"mcp>=1.0.0",
|
|
"httpx>=0.24.0",
|
|
]
|
|
# Voice filler detection (training and inference)
|
|
voice = [
|
|
"onnxruntime>=1.16.0", # Fast CPU inference
|
|
"transformers>=4.30.0", # Model loading and tokenization
|
|
"torch>=2.0.0", # Training
|
|
]
|
|
# Voice training only (includes voice deps + training extras)
|
|
voice-train = [
|
|
"headroom-ai[voice]",
|
|
"datasets>=2.14.0", # Data loading
|
|
"accelerate>=0.20.0", # Training acceleration
|
|
]
|
|
# Evaluation framework for testing compression accuracy
|
|
evals = [
|
|
"datasets>=2.14.0", # HuggingFace datasets
|
|
"sentence-transformers>=2.2.0", # Semantic similarity
|
|
"numpy>=1.24.0", # Numerical operations
|
|
"scikit-learn>=1.3.0", # ML metrics
|
|
"anthropic>=0.18.0", # Anthropic API for evals
|
|
"openai>=1.0.0", # OpenAI API for evals
|
|
]
|
|
# Memory system extras (hierarchical memory with vector search)
|
|
memory = [
|
|
"hnswlib>=0.8.0", # HNSW vector index for semantic search
|
|
"sqlite-vec>=0.1.6",
|
|
]
|
|
# AWS Bedrock backend
|
|
bedrock = [
|
|
"boto3>=1.28.0",
|
|
]
|
|
# HTML content extraction (web scraping results)
|
|
html = [
|
|
"trafilatura>=1.6.0",
|
|
]
|
|
# Comprehensive LLM benchmarks (EleutherAI lm-evaluation-harness)
|
|
benchmark = [
|
|
"lm-eval>=0.4.0",
|
|
"openai>=1.0.0",
|
|
"anthropic>=0.18.0",
|
|
]
|
|
# Development dependencies
|
|
dev = [
|
|
"pytest>=7.0.0",
|
|
"pytest-cov>=4.0.0",
|
|
"pytest-asyncio>=0.21.0",
|
|
"ruff>=0.1.0",
|
|
"mypy>=1.0.0",
|
|
"pre-commit>=3.0.0",
|
|
"openai>=1.0.0",
|
|
"anthropic>=0.18.0",
|
|
"ollama>=0.4.0", # For Ollama integration tests (local LLM, no API key needed)
|
|
"langchain-ollama>=0.2.0", # For LangChain+Ollama integration tests
|
|
"hnswlib>=0.8.0", # For memory system tests
|
|
]
|
|
# All optional dependencies
|
|
all = [
|
|
"headroom-ai[relevance,proxy,reports,llmlingua,code,evals,memory,voice,html,benchmark,mcp]",
|
|
]
|
|
|
|
[project.scripts]
|
|
headroom = "headroom.cli:main"
|
|
|
|
[project.urls]
|
|
Homepage = "https://github.com/chopratejas/headroom"
|
|
Documentation = "https://github.com/chopratejas/headroom#readme"
|
|
Repository = "https://github.com/chopratejas/headroom"
|
|
Issues = "https://github.com/chopratejas/headroom/issues"
|
|
Changelog = "https://github.com/chopratejas/headroom/blob/main/CHANGELOG.md"
|
|
|
|
[tool.hatch.build.targets.wheel]
|
|
packages = ["headroom"]
|
|
# Include non-Python files (dashboard templates, etc.)
|
|
artifacts = [
|
|
"headroom/dashboard/templates/*.html",
|
|
]
|
|
|
|
[tool.hatch.build.targets.sdist]
|
|
include = [
|
|
"/headroom",
|
|
"/tests",
|
|
"/LICENSE",
|
|
"/NOTICE",
|
|
"/README.md",
|
|
"/CHANGELOG.md",
|
|
]
|
|
|
|
[tool.ruff]
|
|
target-version = "py310"
|
|
line-length = 100
|
|
|
|
[tool.ruff.lint]
|
|
select = [
|
|
"E", # pycodestyle errors
|
|
"W", # pycodestyle warnings
|
|
"F", # pyflakes
|
|
"I", # isort
|
|
"B", # flake8-bugbear
|
|
"C4", # flake8-comprehensions
|
|
"UP", # pyupgrade
|
|
]
|
|
ignore = [
|
|
"E501", # line too long (handled by formatter)
|
|
"B008", # do not perform function calls in argument defaults
|
|
"B905", # zip without strict parameter
|
|
"UP038", # isinstance(x, (A, B)) is clearer than isinstance(x, A | B)
|
|
]
|
|
|
|
[tool.ruff.lint.isort]
|
|
known-first-party = ["headroom"]
|
|
|
|
[tool.ruff.format]
|
|
quote-style = "double"
|
|
indent-style = "space"
|
|
|
|
[tool.mypy]
|
|
python_version = "3.10"
|
|
warn_return_any = true
|
|
warn_unused_configs = true
|
|
disallow_untyped_defs = true
|
|
ignore_missing_imports = true
|
|
|
|
# Per-module overrides for modules with dynamic typing patterns
|
|
[[tool.mypy.overrides]]
|
|
module = [
|
|
"headroom.proxy.server",
|
|
"headroom.integrations.langchain",
|
|
"headroom.integrations.mcp",
|
|
"headroom.ccr.mcp_server",
|
|
"headroom.relevance.embedding",
|
|
"headroom.reporting.generator",
|
|
]
|
|
disallow_untyped_defs = false
|
|
|
|
[[tool.mypy.overrides]]
|
|
module = [
|
|
"headroom.tokenizers.*",
|
|
"headroom.providers.litellm",
|
|
"headroom.providers.google",
|
|
]
|
|
disallow_untyped_defs = false
|
|
warn_return_any = false
|
|
|
|
# Ignore third-party stubs with syntax errors
|
|
[[tool.mypy.overrides]]
|
|
module = ["mlx.*"]
|
|
ignore_errors = true
|
|
|
|
[tool.pytest.ini_options]
|
|
testpaths = ["tests"]
|
|
python_files = ["test_*.py"]
|
|
python_functions = ["test_*"]
|
|
addopts = "-v --tb=short"
|
|
asyncio_mode = "auto"
|
|
|
|
[tool.coverage.run]
|
|
source = ["headroom"]
|
|
branch = true
|
|
omit = [
|
|
"headroom/cli.py",
|
|
"*/tests/*",
|
|
]
|
|
|
|
[tool.coverage.report]
|
|
exclude_lines = [
|
|
"pragma: no cover",
|
|
"def __repr__",
|
|
"raise NotImplementedError",
|
|
"if TYPE_CHECKING:",
|
|
"if __name__ == .__main__.:",
|
|
]
|