mirror of
https://github.com/headroomlabs-ai/headroom.git
synced 2026-08-27 14:17:10 -04:00
🤖 I have created a release *beep* *boop* --- <details><summary>0.25.0</summary> ## [0.25.0](https://github.com/chopratejas/headroom/compare/v0.24.0...v0.25.0) (2026-06-12) ### Features * add differential network capture harness ([#761](https://github.com/chopratejas/headroom/issues/761)) ([11ab5f8](11ab5f83a1)) * add light mode for dashboard ([#834](https://github.com/chopratejas/headroom/issues/834)) ([c425893](c425893d12)) * add OAuth2 client-credentials upstream-auth proxy extension ([#778](https://github.com/chopratejas/headroom/issues/778)) ([#784](https://github.com/chopratejas/headroom/issues/784)) ([eb2e50f](eb2e50feb2)) * add Vertex AI proxy routing ([#793](https://github.com/chopratejas/headroom/issues/793)) ([3c77e52](3c77e52ce4)) * **cli:** comprehensive help text, validation, and exception handling improvements ([#640](https://github.com/chopratejas/headroom/issues/640)) ([028efab](028efabb4e)) * compression safety rails — error-output protection, pipeline circuit breaker, library inflation guard ([#851](https://github.com/chopratejas/headroom/issues/851)) ([c0cadcc](c0cadccff9)) * **dashboard:** per-model savings breakdown and expected-vs-actual cost on historical charts ([#807](https://github.com/chopratejas/headroom/issues/807)) ([34dafe6](34dafe69d9)) * detect re-served tool results as over-compression waste signal ([#854](https://github.com/chopratejas/headroom/issues/854)) ([5f1d88a](5f1d88ad27)) * **evals:** add zero-cost tool schema compaction integrity eval ([#817](https://github.com/chopratejas/headroom/issues/817)) ([53a08c6](53a08c63bf)) * gated Markdown-KV compaction formatter (serialization-aware output) ([#859](https://github.com/chopratejas/headroom/issues/859)) ([06b2625](06b2625b17)) * **kompress:** warn on unrecognized HEADROOM_KOMPRESS_BACKEND + document backend selection ([#204](https://github.com/chopratejas/headroom/issues/204)) ([6367d0b](6367d0b722)) * **memory:** add opt-in Apple-GPU (MPS) embedding runtime ([#766](https://github.com/chopratejas/headroom/issues/766)) ([c71592d](c71592d421)) * net-cost cache mutation formula on CompressionPolicy ([#856](https://github.com/chopratejas/headroom/issues/856) P1) ([#857](https://github.com/chopratejas/headroom/issues/857)) ([d5f5802](d5f58026e2)) * **plugins:** Hermes agent headroom_retrieve plugin ([#824](https://github.com/chopratejas/headroom/issues/824)) ([058bced](058bcedab8)) * probe-based retention scoring of recorded compression events ([#862](https://github.com/chopratejas/headroom/issues/862)) ([c2106cb](c2106cbdab)) * **proxy:** add CLI opt-outs for CCR injection (compression-only mode) ([#823](https://github.com/chopratejas/headroom/issues/823)) ([693d9d2](693d9d20e2)) * **proxy:** attribute savings history rollups per provider ([#791](https://github.com/chopratejas/headroom/issues/791)) ([0b8b8d9](0b8b8d92de)) * **proxy:** log compressed messages alongside original request ([#261](https://github.com/chopratejas/headroom/issues/261)) ([2269e40](2269e40bde)) * **proxy:** per-project savings breakdown on the dashboard (claude, codex, aider, copilot, cursor) ([#803](https://github.com/chopratejas/headroom/issues/803)) ([914a60a](914a60a2b0)) * support Python 3.14+ via pyo3 abi3 stable ABI ([#516](https://github.com/chopratejas/headroom/issues/516)) ([19eac8e](19eac8e00d)) * switch Kompress default to kompress-v2-base with weight-only int8 ONNX ([#799](https://github.com/chopratejas/headroom/issues/799)) ([74392b2](74392b238e)) * **transforms:** attribute read_lifecycle + smart_crush tags ([#249](https://github.com/chopratejas/headroom/issues/249)) ([8f37426](8f374263d3)) ### Bug Fixes * **anthropic:** CCR exception must re-raise, not silently swallow ([#838](https://github.com/chopratejas/headroom/issues/838)) ([8db5efc](8db5efc6f9)) * **ccr:** key Rust search/diff/log markers with explicit_hash ([#852](https://github.com/chopratejas/headroom/issues/852)) ([bfcb07d](bfcb07d78e)) * **ccr:** make retrieval TTL configurable ([#715](https://github.com/chopratejas/headroom/issues/715)) ([2533f77](2533f7703e)) * **ccr:** skip CCR when model calls headroom_retrieve alongside user tools ([#839](https://github.com/chopratejas/headroom/issues/839)) ([30078f8](30078f8465)) * **ccr:** use shared compression store ([#875](https://github.com/chopratejas/headroom/issues/875)) ([249af6c](249af6cc7b)) * **ci:** correct comments, timeouts, and pip reliability in native e2e workflows ([#878](https://github.com/chopratejas/headroom/issues/878)) ([b716c8c](b716c8c2ee)) * **ci:** pin cosign-installer to v3 (v4 does not exist) ([#774](https://github.com/chopratejas/headroom/issues/774)) ([199d693](199d693f98)) * **codex:** respect CODEX_HOME for wrap config ([#731](https://github.com/chopratejas/headroom/issues/731)) ([96abf38](96abf38b09)) * **content_router:** guard against empty compression output causing Anthropic 400 ([#771](https://github.com/chopratejas/headroom/issues/771)) ([2f9ff07](2f9ff07e6c)) * **copilot:** use responses API for subscription reasoning models ([#647](https://github.com/chopratejas/headroom/issues/647)) ([84ac332](84ac332d14)) * correct preserved-entry index mapping in Gemini content round-trip ([#836](https://github.com/chopratejas/headroom/issues/836)) ([0ffe2b6](0ffe2b6ea4)) * **dashboard:** stable 'Proxy $ Saved' hero tile under --workers > 1 ([#481](https://github.com/chopratejas/headroom/issues/481)) ([fd73b88](fd73b88368)) * don't inject empty tools:[] when client omitted the tools field ([#772](https://github.com/chopratejas/headroom/issues/772)) ([574bbae](574bbae2cb)) * harden Copilot API auth token handling ([#557](https://github.com/chopratejas/headroom/issues/557)) ([6b0c09f](6b0c09ffd5)) * **health:** readyz verifies upstream connectivity, not just process liveness ([#744](https://github.com/chopratejas/headroom/issues/744)) ([5dfb446](5dfb446da1)) * **init:** guard persistent task startup ([#616](https://github.com/chopratejas/headroom/issues/616)) ([9252d85](9252d852c5)) * **init:** normalize Windows hook paths to forward slashes ([#788](https://github.com/chopratejas/headroom/issues/788)) ([6ea6e31](6ea6e31f09)) * **init:** suppress hook recovery output ([#760](https://github.com/chopratejas/headroom/issues/760)) ([b439599](b4395993ae)) * **learn:** claude-cli streams output with idle timeout ([#373](https://github.com/chopratejas/headroom/issues/373)) ([9bff575](9bff5752bb)) * make headroom wrap readiness probe timeout configurable for slow ML imports ([#581](https://github.com/chopratejas/headroom/issues/581)) ([163677b](163677b405)) * **parser:** detect waste signals in Anthropic tool_result content blocks ([#815](https://github.com/chopratejas/headroom/issues/815)) ([929698a](929698af10)) * **proxy:** F4 — trust X-Forwarded-* only behind allow-listed gateway ([d10bd5f](d10bd5f59c)) * **proxy:** lazy-import server to avoid fastapi crash ([#442](https://github.com/chopratejas/headroom/issues/442)) ([93c6937](93c69372e6)) * **proxy:** make CCR multi-worker warning conditional on backend ([#770](https://github.com/chopratejas/headroom/issues/770)) ([d76a729](d76a7296df)) * **proxy:** make Kompress eager preload cache-only so a cold cache can't block startup ([#783](https://github.com/chopratejas/headroom/issues/783)) ([841663d](841663da16)) * **proxy:** restore Codex usage headers on WS and streaming SSE transports ([#577](https://github.com/chopratejas/headroom/issues/577)) ([#794](https://github.com/chopratejas/headroom/issues/794)) ([0ce68de](0ce68dedd7)) * schema compaction must not drop property names that match DROP_KEYS ([#785](https://github.com/chopratejas/headroom/issues/785)) ([ae2122f](ae2122fda8)) * **security:** block DNS-rebinding on /debug/* and /stats/reset via Host-header allowlist ([#605](https://github.com/chopratejas/headroom/issues/605)) ([b4b5025](b4b50253f1)) * **ssl:** upstream httpx client inherits SSL_CERT_FILE, REQUESTS_CA_BUNDLE, NODE_EXTRA_CA_CERTS ([#745](https://github.com/chopratejas/headroom/issues/745)) ([e50fbb3](e50fbb3e0d)) * suppress LiteLLM provider banner before import ([#874](https://github.com/chopratejas/headroom/issues/874)) ([f9384ef](f9384ef4b7)) * **transforms:** use thread-local tree-sitter parsers to prevent pyo3 Unsendable panic ([#604](https://github.com/chopratejas/headroom/issues/604)) ([2ad300a](2ad300aff8)) * **wrap:** track shared proxy clients with markers ([#877](https://github.com/chopratejas/headroom/issues/877)) ([05bd56b](05bd56bcb6)) ### Code Refactoring * extract litellm model resolution to shared utility ([ec7d006](ec7d0065cc)) </details> --- This PR was generated with [Release Please](https://github.com/googleapis/release-please). See [documentation](https://github.com/googleapis/release-please#release-please). Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com>
414 lines
14 KiB
TOML
414 lines
14 KiB
TOML
[build-system]
|
||
requires = ["maturin>=1.5,<2.0"]
|
||
build-backend = "maturin"
|
||
|
||
[project]
|
||
name = "headroom-ai"
|
||
version = "0.25.0"
|
||
description = "The Context Optimization Layer for LLM Applications - Cut costs by 50-90%"
|
||
readme = "README.md"
|
||
license = "Apache-2.0"
|
||
requires-python = ">=3.10"
|
||
authors = [
|
||
{ name = "Headroom Contributors" }
|
||
]
|
||
maintainers = [
|
||
{ name = "Headroom Contributors" }
|
||
]
|
||
keywords = [
|
||
"llm",
|
||
"openai",
|
||
"anthropic",
|
||
"claude",
|
||
"gpt",
|
||
"context",
|
||
"token",
|
||
"optimization",
|
||
"compression",
|
||
"caching",
|
||
"proxy",
|
||
"ai",
|
||
"machine-learning",
|
||
]
|
||
classifiers = [
|
||
"Development Status :: 4 - Beta",
|
||
"Intended Audience :: Developers",
|
||
"License :: OSI Approved :: Apache Software License",
|
||
"Operating System :: OS Independent",
|
||
"Programming Language :: Python :: 3",
|
||
"Programming Language :: Python :: 3.10",
|
||
"Programming Language :: Python :: 3.11",
|
||
"Programming Language :: Python :: 3.12",
|
||
"Programming Language :: Python :: 3.13",
|
||
"Programming Language :: Python :: 3.14",
|
||
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
||
"Topic :: Software Development :: Libraries :: Python Modules",
|
||
"Typing :: Typed",
|
||
]
|
||
dependencies = [
|
||
# Core: lightweight compression (SmartCrusher, ContentRouter, CCR, TOIN)
|
||
"tiktoken>=0.5.0", # Tokenizer for all compressors
|
||
"pydantic>=2.0.0", # Config and data models
|
||
"litellm>=1.86.2,<2.0", # Model registry, pricing, and provider support
|
||
"click>=8.1.0", # CLI framework
|
||
"rich>=13.0.0", # Rich terminal output
|
||
"opentelemetry-api>=1.24.0", # Safe no-op OTEL API for instrumentation
|
||
"ast-grep-cli>=0.30.0", # AST-aware code slicing (CodeCompressor); binary wheel
|
||
"tomli>=2.0.0; python_version < '3.11'", # tomllib backport for helper scripts
|
||
]
|
||
|
||
[project.optional-dependencies]
|
||
# Proxy server (most common install: pip install headroom-ai[proxy])
|
||
proxy = [
|
||
"fastapi>=0.100.0",
|
||
"uvicorn>=0.23.0,<1.0",
|
||
"httpx[http2]>=0.24.0",
|
||
"openai>=2.14.0", # OpenAI API format support
|
||
"mcp>=1.0.0", # MCP server (headroom_compress, retrieve, stats)
|
||
"magika>=0.6.0", # ML content detection for ContentRouter
|
||
"zstandard>=0.20.0", # Decompress zstd request bodies (Codex, etc.)
|
||
"websockets>=13.0", # WebSocket proxy for /v1/responses (Codex gpt-5.4+)
|
||
"onnxruntime>=1.16.0", # Kompress ONNX INT8 text compression (no torch needed)
|
||
"transformers>=4.30.0,<6.0", # Tokenizer only (for Kompress)
|
||
"watchdog>=4.0.0", # File watcher for live code graph reindexing (--code-graph)
|
||
"sqlite-vec>=0.1.6", # Vector index for memory (--memory). Lightweight, no torch.
|
||
]
|
||
# Production ASGI/WSGI server — Unix-only (gunicorn does not support Windows).
|
||
# Kept separate from [proxy] so that dev, CI, and Windows users are not forced
|
||
# to install a non-functional package. Production deployments should use:
|
||
# pip install headroom-ai[proxy,proxy-prod]
|
||
proxy-prod = [
|
||
"headroom-ai[proxy]",
|
||
"gunicorn>=21.0.0; sys_platform != 'win32'",
|
||
]
|
||
# AST-based code compression (tree-sitter)
|
||
code = [
|
||
"tree-sitter-language-pack>=0.10.0",
|
||
]
|
||
# ML-based compression with Kompress (ModernBERT).
|
||
# (The legacy [llmlingua] extra was removed in 0.9.x — no live code path used it.
|
||
# Use [ml] for the supported ML compression dependencies.)
|
||
ml = [
|
||
"torch>=2.0.0",
|
||
"transformers>=4.30.0,<6.0",
|
||
# transformers >= 5.x requires huggingface-hub >= 1.5.0,<2.0; pinning
|
||
# the floor here prevents Kompress from silently falling back to
|
||
# "unavailable" when a sibling install (e.g. `pip install
|
||
# strands-agents`) drags huggingface-hub backwards.
|
||
"huggingface-hub>=1.5.0,<2.0",
|
||
]
|
||
# Memory system (hierarchical memory with vector search)
|
||
memory = [
|
||
"hnswlib>=0.8.0",
|
||
"sqlite-vec>=0.1.6",
|
||
"sentence-transformers>=2.2.0,<6.0",
|
||
]
|
||
# Qdrant + Neo4j memory backend helpers
|
||
memory-stack = [
|
||
"mem0ai>=1.0.0,<2.0",
|
||
"qdrant-client>=1.9.0,<2.0",
|
||
"neo4j>=5.20.0,<7.0",
|
||
]
|
||
# Apple-Silicon GPU (MPS) offload for the memory embedder. Opt in at runtime with
|
||
# HEADROOM_EMBEDDER_RUNTIME=pytorch_mps. macOS-only; intentionally excluded from [all].
|
||
pytorch-mps = [
|
||
"torch>=2.0.0; sys_platform == 'darwin'",
|
||
"sentence-transformers>=2.2.0; sys_platform == 'darwin'",
|
||
]
|
||
# Semantic relevance scoring with embeddings.
|
||
# Uses `fastembed` (BAAI/bge-small-en-v1.5 by default — 33M params,
|
||
# 384 dims, ~30 MB int8-quantized ONNX). Same library + model used by
|
||
# the Rust SmartCrusher (`fastembed` crate), giving byte-equal embeddings
|
||
# across the language boundary. Replaced sentence-transformers in
|
||
# Stage 3c.1 — fastembed is faster (~2-3x), smaller (no torch
|
||
# dependency), and outranks all-MiniLM-L6-v2 on MTEB by ~6 points.
|
||
relevance = [
|
||
"fastembed>=0.4.0",
|
||
"numpy>=1.24.0",
|
||
]
|
||
# Image compression (ML-based routing + OCR)
|
||
#
|
||
# OCR backend uses ONNX Runtime regardless of Python version. The
|
||
# rapidocr ecosystem split into two flavors after 1.4.x:
|
||
# * rapidocr-onnxruntime 1.4.x — bundled-ORT package, capped at
|
||
# Python <3.13 by its requires-python metadata. Drop-in for our
|
||
# existing v1 tuple-shaped API call.
|
||
# * rapidocr 3.x — engine-agnostic core, supports Python 3.13+.
|
||
# Returns a RapidOCROutput dataclass (txts, scores, boxes, ...).
|
||
# Needs `onnxruntime` installed separately to use the ORT backend.
|
||
#
|
||
# `headroom/image/compressor.py` adapts both API shapes at runtime via
|
||
# a try/except cascade. See issue #372 for context.
|
||
image = [
|
||
"pillow>=10.0.0",
|
||
"sentencepiece>=0.1.99", # Required by SigLIP tokenizer (SiglipTokenizer)
|
||
# Python 3.6–3.12: keep the proven ORT-bundled package directly.
|
||
# ~15 MB ONNX models auto-downloaded on first use.
|
||
"rapidocr-onnxruntime>=1.4.0,<2; python_version<'3.13'",
|
||
# Python 3.13+: rapidocr-onnxruntime is unavailable (its wheels
|
||
# declare requires-python<3.13). Use the successor `rapidocr` 3.x
|
||
# core + `onnxruntime` engine; same ORT backend, just split into
|
||
# two packages. Total install size and inference speed unchanged.
|
||
"rapidocr>=3.0,<4; python_version>='3.13'",
|
||
"onnxruntime>=1.7,<2; python_version>='3.13'",
|
||
]
|
||
# Report generation
|
||
reports = [
|
||
"jinja2>=3.0.0",
|
||
]
|
||
# OpenTelemetry metrics export
|
||
otel = [
|
||
"opentelemetry-sdk>=1.24.0",
|
||
"opentelemetry-exporter-otlp-proto-http>=1.24.0",
|
||
]
|
||
# any-llm multi-provider backend (requires Python 3.11+)
|
||
anyllm = [
|
||
"any-llm-sdk>=1.0.0; python_version >= '3.11'",
|
||
]
|
||
# LangChain integration
|
||
langchain = [
|
||
"langchain-core>=1.3.3,<4.0",
|
||
"langchain-openai>=1.1.14,<2.0",
|
||
]
|
||
# Agno agent framework integration
|
||
agno = [
|
||
"agno>=1.0.0",
|
||
]
|
||
# AWS Strands Agents SDK integration
|
||
strands = [
|
||
"strands-agents>=0.1.0",
|
||
]
|
||
# MCP server for Claude Code integration
|
||
mcp = [
|
||
"mcp>=1.0.0",
|
||
"httpx>=0.24.0",
|
||
]
|
||
# Voice filler detection
|
||
voice = [
|
||
"onnxruntime>=1.16.0",
|
||
"transformers>=4.30.0,<6.0",
|
||
"torch>=2.0.0",
|
||
]
|
||
# Voice training (includes voice deps + training extras)
|
||
voice-train = [
|
||
"headroom-ai[voice]",
|
||
"datasets>=2.14.0",
|
||
"accelerate>=0.20.0",
|
||
]
|
||
# Evaluation framework
|
||
evals = [
|
||
"datasets>=2.14.0",
|
||
"sentence-transformers>=2.2.0,<6.0",
|
||
"numpy>=1.24.0",
|
||
"scikit-learn>=1.3.0",
|
||
"anthropic>=0.18.0",
|
||
"openai>=1.0.0",
|
||
]
|
||
# AWS Bedrock backend
|
||
bedrock = [
|
||
"boto3>=1.28.0",
|
||
]
|
||
# HTML content extraction
|
||
html = [
|
||
"trafilatura>=1.6.0",
|
||
]
|
||
# Comprehensive LLM benchmarks
|
||
benchmark = [
|
||
"lm-eval[api]>=0.4.0",
|
||
"openai>=1.0.0",
|
||
"anthropic>=0.18.0",
|
||
]
|
||
# Development dependencies
|
||
dev = [
|
||
"pytest>=7.0.0",
|
||
"pytest-cov>=4.0.0",
|
||
"pytest-asyncio>=0.21.0",
|
||
"ruff>=0.1.0",
|
||
"mypy>=1.0.0",
|
||
"pre-commit>=3.0.0",
|
||
"openai>=1.0.0",
|
||
"anthropic>=0.18.0",
|
||
"litellm>=1.86.2,<2.0",
|
||
"fastapi>=0.100.0",
|
||
"uvicorn>=0.23.0,<1.0",
|
||
"httpx[http2]>=0.24.0",
|
||
"websockets>=13.0",
|
||
"opentelemetry-sdk>=1.24.0",
|
||
"opentelemetry-exporter-otlp-proto-http>=1.24.0",
|
||
"ollama>=0.4.0",
|
||
"langchain-ollama>=0.2.0",
|
||
"hnswlib>=0.8.0",
|
||
"sqlite-vec>=0.1.6",
|
||
"sentence-transformers>=2.2.0,<6.0",
|
||
"numpy>=1.24.0",
|
||
]
|
||
# All optional dependencies (everything you need)
|
||
all = [
|
||
"headroom-ai[proxy,code,ml,memory,relevance,image,reports,otel,evals,voice,html,benchmark,mcp]",
|
||
]
|
||
|
||
[project.scripts]
|
||
headroom = "headroom.cli:main"
|
||
|
||
[project.urls]
|
||
Homepage = "https://headroom-docs.vercel.app"
|
||
Documentation = "https://headroom-docs.vercel.app/docs"
|
||
Repository = "https://github.com/chopratejas/headroom"
|
||
Issues = "https://github.com/chopratejas/headroom/issues"
|
||
Changelog = "https://github.com/chopratejas/headroom/blob/main/CHANGELOG.md"
|
||
# llms.txt convention (llmstxt.org) — point AI agents / LLM crawlers
|
||
# at the auto-generated docs index so they can resolve install paths
|
||
# and entry points without a follow-up fetch.
|
||
"AI / LLM Index" = "https://headroom-docs.vercel.app/llms.txt"
|
||
|
||
# Maturin builds a single wheel containing both the Python source under
|
||
# `headroom/` AND the compiled Rust extension `headroom/_core.so` (cdylib
|
||
# from `crates/headroom-py`). One `pip install headroom-ai` ships everything
|
||
# atomically — no separate `headroom-core-py` package, no chicken-and-egg,
|
||
# no PIP_FIND_LINKS plumbing. Phase A0's runtime fail-loud check still
|
||
# exists but only fires if someone forces an sdist install on a platform
|
||
# without a wheel and the rust toolchain isn't available to compile it.
|
||
# Pin the project's package index to public PyPI. Without this, `uv lock`
|
||
# inherits the developer's user-level `~/.config/uv/uv.toml` index
|
||
# setting — including private/internal mirrors like
|
||
# `pypi.netflix.net/simple` — and bakes those URLs into uv.lock, which
|
||
# then breaks CI on every public runner that can't reach the mirror.
|
||
# Declaring the index in pyproject.toml makes the project authoritative
|
||
# regardless of who runs `uv lock`.
|
||
[[tool.uv.index]]
|
||
name = "pypi"
|
||
url = "https://pypi.org/simple/"
|
||
default = true
|
||
|
||
[tool.maturin]
|
||
# Where the Python package lives. With `python-source = "."` and the
|
||
# package directory `headroom/` at repo root, maturin includes every file
|
||
# under `headroom/` in the wheel — that picks up the dashboard HTML
|
||
# templates and bundled YAML configs. `LICENSE` and `NOTICE` are listed
|
||
# explicitly because maturin sdists do not get the package-directory
|
||
# treatment wheels do, and PEP 639 auto-discovery emits both files into
|
||
# `License-File:` metadata — PyPI rejects sdists whose declared license
|
||
# files are missing from the tarball with `400 License-File X does not
|
||
# exist in distribution file`.
|
||
include = [
|
||
{ path = "LICENSE", format = "sdist" },
|
||
{ path = "NOTICE", format = "sdist" },
|
||
]
|
||
python-source = "."
|
||
module-name = "headroom._core"
|
||
# The cdylib source lives under `crates/headroom-py`. Maturin invokes
|
||
# `cargo build` with this manifest to produce `_core.cdylib`, then injects
|
||
# the resulting `.so` into the wheel at `headroom/_core.so`.
|
||
manifest-path = "crates/headroom-py/Cargo.toml"
|
||
features = ["extension-module"]
|
||
# Forbid building without the cdylib feature — bare `cargo build` won't
|
||
# produce a usable Python extension. Maturin's default `bindings` is "pyo3"
|
||
# which is correct here (see `crates/headroom-py/src/`).
|
||
bindings = "pyo3"
|
||
|
||
[tool.ruff]
|
||
target-version = "py310"
|
||
line-length = 100
|
||
|
||
[tool.ruff.lint]
|
||
select = [
|
||
"E", # pycodestyle errors
|
||
"W", # pycodestyle warnings
|
||
"F", # pyflakes
|
||
"I", # isort
|
||
"B", # flake8-bugbear
|
||
"C4", # flake8-comprehensions
|
||
"UP", # pyupgrade
|
||
]
|
||
ignore = [
|
||
"E501", # line too long (handled by formatter)
|
||
"B008", # do not perform function calls in argument defaults
|
||
"B905", # zip without strict parameter
|
||
]
|
||
|
||
[tool.ruff.lint.isort]
|
||
known-first-party = ["headroom"]
|
||
|
||
[tool.ruff.format]
|
||
quote-style = "double"
|
||
indent-style = "space"
|
||
|
||
[tool.mypy]
|
||
python_version = "3.10"
|
||
warn_return_any = true
|
||
warn_unused_configs = true
|
||
disallow_untyped_defs = true
|
||
ignore_missing_imports = true
|
||
|
||
# Per-module overrides for modules with dynamic typing patterns
|
||
[[tool.mypy.overrides]]
|
||
module = [
|
||
"headroom.proxy.server",
|
||
"headroom.proxy.cost",
|
||
"headroom.proxy.prometheus_metrics",
|
||
"headroom.proxy.semantic_cache",
|
||
"headroom.proxy.rate_limiter",
|
||
"headroom.proxy.request_logger",
|
||
"headroom.proxy.helpers",
|
||
"headroom.integrations.langchain",
|
||
"headroom.integrations.mcp",
|
||
"headroom.ccr.mcp_server",
|
||
"headroom.relevance.embedding",
|
||
"headroom.reporting.generator",
|
||
]
|
||
disallow_untyped_defs = false
|
||
|
||
[[tool.mypy.overrides]]
|
||
module = [
|
||
"headroom.tokenizers.*",
|
||
"headroom.providers.litellm",
|
||
"headroom.providers.google",
|
||
]
|
||
disallow_untyped_defs = false
|
||
warn_return_any = false
|
||
|
||
# Handler mixins use self.* from HeadroomProxy via duck typing — mypy can't resolve these
|
||
[[tool.mypy.overrides]]
|
||
module = ["headroom.proxy.handlers.*"]
|
||
disallow_untyped_defs = false
|
||
ignore_errors = true
|
||
|
||
# Ignore third-party stubs with syntax errors
|
||
[[tool.mypy.overrides]]
|
||
module = ["mlx.*"]
|
||
ignore_errors = true
|
||
|
||
[tool.pytest.ini_options]
|
||
testpaths = ["tests"]
|
||
python_files = ["test_*.py"]
|
||
python_functions = ["test_*"]
|
||
addopts = "-v --tb=short"
|
||
asyncio_mode = "auto"
|
||
filterwarnings = [
|
||
# pyo3 Unsendable parsers emit an unraisable warning when GC drops them on a
|
||
# test-teardown thread; this is a test-harness artifact, not a production issue
|
||
# (production threads are long-lived and drop their parsers on themselves).
|
||
"ignore::pytest.PytestUnraisableExceptionWarning",
|
||
]
|
||
markers = [
|
||
"slow: slow tests (model loads, large fixtures)",
|
||
"real_llm: tests that hit real LLM APIs; skipped unless explicitly enabled",
|
||
"live: opt-in multi-turn tests that hit real upstream APIs; require provider keys",
|
||
]
|
||
|
||
[tool.coverage.run]
|
||
source = ["headroom"]
|
||
branch = true
|
||
omit = [
|
||
"headroom/cli.py",
|
||
"*/tests/*",
|
||
]
|
||
|
||
[tool.coverage.report]
|
||
exclude_lines = [
|
||
"pragma: no cover",
|
||
"def __repr__",
|
||
"raise NotImplementedError",
|
||
"if TYPE_CHECKING:",
|
||
"if __name__ == .__main__.:",
|
||
]
|