From eca5c9422968775f02aeff830952fd20dec1a659 Mon Sep 17 00:00:00 2001 From: chopratejas Date: Sun, 1 Feb 2026 22:12:45 -0800 Subject: [PATCH] Add SQLite vector backend as default and fix HNSW test skipping - Add VectorBackend.AUTO that prefers SQLITE_VEC if available, else HNSW - Add SQLiteVectorIndex configuration options (vector_db_path, vector_cache_size_kb) - Add update_embedding() method to SQLiteVectorIndex for VectorIndex protocol - Update factory to create SQLiteVectorIndex when appropriate - Skip HNSW-specific tests when hnswlib is not available (fixes CI on Python 3.12) --- devto-article.md | 295 ---------------------- headroom/memory/adapters/sqlite_vector.py | 41 ++- headroom/memory/backends/local.py | 7 +- headroom/memory/config.py | 21 +- headroom/memory/factory.py | 48 +++- tests/test_memory_tracker_integration.py | 9 + tests/test_memory_usage_integration.py | 11 + 7 files changed, 126 insertions(+), 306 deletions(-) delete mode 100644 devto-article.md diff --git a/devto-article.md b/devto-article.md deleted file mode 100644 index fb74f8748..000000000 --- a/devto-article.md +++ /dev/null @@ -1,295 +0,0 @@ -# I Was Wasting 85% of My LLM Tokens on JSON Boilerplate - -I recently built an agent to handle some SRE tasks—fetching logs, querying databases, searching code. It worked, but when I looked at the traces, I was annoyed. - -It wasn't just that it was expensive (though the bill was climbing). It was the sheer **inefficiency**. - -I looked at a single tool output—a search for Python files. It was 40,000 tokens. -About 35,000 of those tokens were just `"type": "file"` and `"language": "python"` repeated 2,000 times. - -We are paying premium compute prices to force state-of-the-art models to read standard JSON boilerplate. - -I couldn't find a tool that solved this without breaking the agent, so I wrote one. It's called **Headroom**. It's a context optimization layer that sits between your app and your LLM. It compresses context by ~85% without losing semantic meaning. - -It's open source (Apache-2.0). If you just want the code: -**[github.com/chopratejas/headroom](https://github.com/chopratejas/headroom)** - ---- - -## Why Truncation and Summarization Don't Work - -When your context window fills up, the standard industry solution is **truncation** (chopping off the oldest messages or the middle of the document). - -But for an agent, truncation is dangerous. - -* If you chop the middle of a log file, you might lose the one error line that explains the crash. -* If you chop a file list, you might lose the exact config file the user asked for. - -I tried **summarization** (using a cheaper model to summarize the data first), but that introduced hallucination. I had a summarizer tell me a deployment "looked fine" because it ignored specific error codes in the raw log. - -I needed a third option: **Lossless compression.** Or at least, "intent-lossless." - ---- - -## The Core Idea: Statistical Analysis, Not Blind Truncation - -I realized that 90% of the data in a tool output is just schema scaffolding. The LLM doesn't need to see `status: active` repeated a thousand times. It needs the **anomalies**. - -Headroom's SmartCrusher runs statistical analysis before touching your data: - -**1. Constant Factoring** -If every item in an array has `"type": "file"`, it doesn't repeat that 2,000 times. It extracts constants once. - -**2. Outlier Detection** -It calculates standard deviation of numerical fields. It preserves the spikes—the values that are >2σ from the mean. Those are usually what matters. - -**3. Error Preservation** -Hard rule: never discard strings that look like stack traces, error messages, or failures. Errors are sacred. - -**4. Relevance Scoring** -If you searched for "auth", items containing "auth" get preserved. Uses BM25 + semantic embeddings (hybrid scoring) to match items against the user's query context. - -**5. First/Last Retention** -Always keeps first few and last few items. The LLM expects to see some examples, and recency matters. - -The result: 40,000 tokens → 4,000 tokens. Same information density. No hallucination risk. - ---- - -## CCR: Making Compression Reversible - -Here's the insight that changed everything: **compression should be reversible**. - -I call the architecture **CCR** (Compress-Cache-Retrieve): - -### 1. Compress -SmartCrusher compresses the tool output from 2,000 items to 20. - -### 2. Cache -The original 2,000 items are cached locally (5-minute TTL, LRU eviction). - -### 3. Retrieve -Headroom injects a tool called `headroom_retrieve()` into the LLM's context. If the model looks at the compressed summary and decides it needs more data—maybe the user asked a follow-up question—it can call that tool. Headroom fetches from the cache and returns the relevant items. - -This changes the risk calculus. You can compress aggressively (90%+) because **nothing is ever truly lost**. The model can always "unzip" what it needs. - -I've had conversations like this: - -``` -Turn 1: "Search for all Python files" - → 1000 files returned, compressed to 15 - -Turn 5: "Actually, what was that file handling JWT tokens?" - → LLM calls headroom_retrieve("jwt") - → Returns jwt_handler.py from cached data -``` - -No extra API calls. No "sorry, I don't have that information anymore." - ---- - -## TOIN: The Network Effect - -Here's where it gets interesting. Headroom learns from compression patterns. - -**TOIN** (Tool Output Intelligence Network) tracks—anonymously—what happens after compression: -- Which fields get retrieved most often? -- Which tool types have high retrieval rates? -- What query patterns trigger retrievals? - -This data feeds back into compression recommendations. If TOIN learns that users frequently retrieve `error_code` fields after compression, it tells SmartCrusher to preserve `error_code` more aggressively next time. - -Privacy is built in: -- No actual data values stored -- Tool names are structure hashes -- Field names are SHA256[:8] hashes -- No user identifiers - -The network effect: more users → more compression events → better recommendations for everyone. - ---- - -## Memory: Cross-Conversation Learning - -Agents often need to remember things across conversations. "I prefer dark mode." "My timezone is PST." "I'm working on the auth refactor." - -Headroom has a memory system that extracts and stores these facts automatically. - -Two approaches: - -**Fast Memory (Recommended)** -Zero extra latency. The LLM outputs a `` block inline with its response. Headroom parses it out and stores the memory. - -```python -from headroom.memory import with_fast_memory -client = with_fast_memory(OpenAI(), user_id="alice") - -# Memories extracted automatically from responses -# Injected automatically into future requests -``` - -**Background Memory** -Separate LLM call extracts memories asynchronously. More accurate but adds latency. - -```python -from headroom import with_memory -client = with_memory(OpenAI(), user_id="alice") -``` - -Memories are stored locally (SQLite) and injected into future conversations. The model remembers that Alice prefers dark mode without you managing state. - ---- - -## The Transform Pipeline - -Headroom runs four transforms on each request: - -### 1. CacheAligner -LLM providers offer cached token pricing (Anthropic: 90% off, OpenAI: 50% off). But caching only works if your prompt prefix is stable. - -Problem: your system prompt probably has a timestamp. `Current time: 2024-01-15 10:32:45`. That breaks caching. - -CacheAligner extracts dynamic content and moves it to the end, stabilizing the prefix. Same information, better cache hits. - -### 2. SmartCrusher -The statistical compression engine. Analyzes arrays, detects patterns, preserves anomalies, factors constants. - -### 3. ContentRouter -Different content needs different compression. Code isn't JSON isn't logs isn't prose. - -ContentRouter uses ML-based content detection to route data to specialized compressors: -- **Code** → AST-aware compression (tree-sitter) -- **JSON** → SmartCrusher -- **Logs** → LogCompressor (clusters similar messages) -- **Text** → Optional LLMLingua integration (20x compression, adds latency) - -### 4. RollingWindow -When context exceeds the model limit, something has to go. RollingWindow drops oldest tool calls + responses together (never orphans data), preserves system prompt and recent turns. - ---- - -## Three Ways to Use It - -### Option 1: Proxy Server (Zero Code Changes) - -```bash -pip install headroom-ai -headroom proxy --port 8787 -``` - -Point your OpenAI client to `http://localhost:8787/v1`. Done. - -```python -from openai import OpenAI -client = OpenAI(base_url="http://localhost:8787/v1") -# No other changes -``` - -Works with Claude Code, Cursor, any OpenAI-compatible client. - -### Option 2: SDK Wrapper - -```python -from headroom import HeadroomClient -from openai import OpenAI - -client = HeadroomClient(OpenAI()) - -response = client.chat.completions.create( - model="gpt-4o", - messages=[...], - headroom_mode="optimize" # or "audit" or "simulate" -) -``` - -Three modes: -- **audit**: Observe only. Logs what would be optimized, doesn't change anything. -- **optimize**: Apply compression. This is what saves tokens. -- **simulate**: Dry run. Returns the optimized messages without calling the API. - -Start with `audit` to see potential savings, then flip to `optimize` when you're confident. - -### Option 3: Framework Integrations - -**LangChain:** -```python -from langchain_openai import ChatOpenAI -from headroom.integrations.langchain import HeadroomChatModel - -base_model = ChatOpenAI(model="gpt-4o") -model = HeadroomChatModel(base_model, mode="optimize") - -# Use in any chain or agent -chain = prompt | model | parser -``` - -**Agno:** -```python -from agno.agent import Agent -from headroom.integrations.agno import HeadroomAgnoModel - -model = HeadroomAgnoModel(original_model, mode="optimize") -agent = Agent(model=model, tools=[...]) -``` - -**MCP (Model Context Protocol):** -```python -from headroom.integrations.mcp import compress_tool_result - -# Compress any tool result before returning to LLM -compressed = compress_tool_result(tool_name, result_data) -``` - ---- - -## Real Numbers - -I've been running this in production for months. Here's what the token reduction looks like: - -| Workload | Before | After | Savings | -|----------|--------|-------|---------| -| Log Analysis | 22,000 | 3,300 | 85% | -| Code Search | 45,000 | 4,500 | 90% | -| Database Queries | 18,000 | 2,700 | 85% | -| Long Conversations | 80,000 | 32,000 | 60% | - -Latency overhead: 3-5ms per request. No extra LLM calls. - ---- - -## What's Coming Next - -This is actively maintained. On the roadmap: - -**More Frameworks** -- CrewAI integration -- AutoGen integration -- Semantic Kernel integration - -**Managed Storage** -- Cloud-hosted TOIN backend (opt-in) -- Cross-device memory sync -- Team-shared compression patterns - -**Better Compression** -- Domain-specific profiles (SRE, coding, data analysis) -- Custom compressor plugins -- Streaming compression for real-time tools - ---- - -## Why I Built This - -I'm a believer that we're in the "optimization phase" of the AI hype cycle. Getting things to work is table stakes; getting them to work cheaply and reliably is the actual engineering work. - -Headroom is my attempt to fix the "context bloat" problem properly. Not with heuristics or truncation, but with statistical analysis and reversible compression. - -It runs entirely locally. No data leaves your machine (except to OpenAI/Anthropic as usual). Apache-2.0 licensed. - -**Repo:** [github.com/chopratejas/headroom](https://github.com/chopratejas/headroom) - -If you find bugs or have ideas, open an issue. I'm actively maintaining this. - ---- - -*Tags: #llm #ai #python #openai #anthropic #agents #optimization* diff --git a/headroom/memory/adapters/sqlite_vector.py b/headroom/memory/adapters/sqlite_vector.py index e340b6e9d..fa65705ec 100644 --- a/headroom/memory/adapters/sqlite_vector.py +++ b/headroom/memory/adapters/sqlite_vector.py @@ -647,6 +647,45 @@ class SQLiteVectorIndex: return self._deserialize_f32(row[0], self._dimension) + async def update_embedding(self, memory_id: str, embedding: np.ndarray) -> bool: + """Update the embedding for an indexed memory. + + Args: + memory_id: The unique identifier of the memory. + embedding: The new embedding vector. + + Returns: + True if updated, False if memory not found in index. + """ + embedding = np.asarray(embedding, dtype=np.float32) + if embedding.shape[0] != self._dimension: + raise ValueError( + f"Embedding dimension {embedding.shape[0]} does not match " + f"index dimension {self._dimension}" + ) + + with self._lock: + with self._get_conn() as conn: + # Get rowid for the memory + row = conn.execute( + "SELECT rowid FROM vec_metadata WHERE memory_id = ?", + (memory_id,), + ).fetchone() + + if row is None: + return False + + rowid = row[0] + + # Update the embedding + conn.execute( + "UPDATE vec_embeddings SET embedding = ? WHERE rowid = ?", + (self._serialize_f32(embedding), rowid), + ) + conn.commit() + + return True + def clear(self) -> None: """Clear all entries from the index.""" with self._lock: @@ -705,6 +744,6 @@ class SQLiteVectorIndex: with self._get_conn() as conn: conn.execute("VACUUM") - def close(self) -> None: + async def close(self) -> None: """Close the index (cleanup).""" pass # Connection-per-request pattern, nothing to close diff --git a/headroom/memory/backends/local.py b/headroom/memory/backends/local.py index cb6eb1235..9346c4097 100644 --- a/headroom/memory/backends/local.py +++ b/headroom/memory/backends/local.py @@ -2,7 +2,8 @@ Provides a fully local memory backend using embedded databases: - SQLite for memory storage -- HNSW for vector search +- SQLite-vec for vector search (bounded, persistent) - preferred +- HNSW for vector search (fallback if sqlite-vec unavailable) - FTS5 for text search - SQLite graph for relationships (bounded memory, persistent) @@ -64,9 +65,9 @@ class LocalBackend: This backend provides a fully local memory system with: - SQLite for memory storage (MemoryStore) - - HNSW for vector search (VectorIndex) + - SQLite-vec for vector search (VectorIndex) - bounded, persistent - FTS5 for text search (TextIndex) - - In-memory graph for relationships (GraphStore) + - SQLite graph for relationships (GraphStore) - bounded, persistent All operations are performed locally with no network calls, making it suitable for: diff --git a/headroom/memory/config.py b/headroom/memory/config.py index eb16b63bd..49a15823c 100644 --- a/headroom/memory/config.py +++ b/headroom/memory/config.py @@ -2,7 +2,7 @@ Provides configuration options for all pluggable components: - Storage backends (SQLite, future: PostgreSQL, DynamoDB) -- Vector index backends (HNSW, future: FAISS, Pinecone) +- Vector index backends (SQLITE_VEC recommended, HNSW fallback) - Text index backends (FTS5, future: Elasticsearch) - Embedder backends (local sentence-transformers, OpenAI, Ollama) - Caching options @@ -26,8 +26,9 @@ class StoreBackend(Enum): class VectorBackend(Enum): """Supported vector index backends.""" - HNSW = "hnsw" - # Future: FAISS = "faiss", PINECONE = "pinecone" + AUTO = "auto" # Auto-select: SQLITE_VEC if available, else HNSW + SQLITE_VEC = "sqlite_vec" # SQLite-based, bounded memory, recommended + HNSW = "hnsw" # hnswlib-based, unbounded unless max_entries set class TextBackend(Enum): @@ -57,11 +58,16 @@ class MemoryConfig: store_backend: Which storage backend to use for memory persistence. db_path: Path to the database file (for file-based backends like SQLite). - vector_backend: Which vector index backend to use for similarity search. + vector_backend: Which vector index backend to use (AUTO, SQLITE_VEC, HNSW). + AUTO (default) selects SQLITE_VEC if available, else HNSW. vector_dimension: Dimension of embedding vectors. + vector_db_path: Path to vector index database (for SQLITE_VEC). Derived from + db_path if None. + vector_cache_size_kb: SQLite page cache size for vector index (8MB default). hnsw_ef_construction: HNSW index build-time accuracy parameter. hnsw_m: HNSW maximum number of connections per node. hnsw_ef_search: HNSW search-time accuracy parameter. + hnsw_max_entries: Maximum entries for HNSW (None = unbounded). text_backend: Which text index backend to use for full-text search. @@ -90,11 +96,16 @@ class MemoryConfig: db_path: Path = field(default_factory=lambda: Path("headroom_memory.db")) # Vector index - vector_backend: VectorBackend = VectorBackend.HNSW + vector_backend: VectorBackend = VectorBackend.AUTO # Auto-select best available vector_dimension: int = 384 + vector_db_path: Path | None = ( + None # For SQLite-based vector index (derived from db_path if None) + ) + vector_cache_size_kb: int = 8192 # SQLite page cache size (8MB default) hnsw_ef_construction: int = 200 hnsw_m: int = 16 hnsw_ef_search: int = 50 + hnsw_max_entries: int | None = None # Max entries for HNSW (None = unbounded) # Text index text_backend: TextBackend = TextBackend.FTS5 diff --git a/headroom/memory/factory.py b/headroom/memory/factory.py index c5316d9a5..0db58ff5d 100644 --- a/headroom/memory/factory.py +++ b/headroom/memory/factory.py @@ -139,9 +139,52 @@ def _create_vector_index(config: MemoryConfig) -> VectorIndex: A VectorIndex implementation based on config.vector_backend. Raises: - ValueError: If the vector backend is not supported. + ValueError: If the vector backend is not supported or unavailable. """ - if config.vector_backend == VectorBackend.HNSW: + backend = config.vector_backend + + # AUTO: prefer SQLITE_VEC if available, else HNSW + if backend == VectorBackend.AUTO: + from headroom.memory.adapters import SQLITE_VEC_AVAILABLE + + if SQLITE_VEC_AVAILABLE: + backend = VectorBackend.SQLITE_VEC + else: + backend = VectorBackend.HNSW + + if backend == VectorBackend.SQLITE_VEC: + from headroom.memory.adapters import SQLITE_VEC_AVAILABLE + + if not SQLITE_VEC_AVAILABLE: + raise ValueError( + "sqlite-vec is not available. Install with: pip install sqlite-vec\n" + "Or use vector_backend=VectorBackend.HNSW" + ) + + from headroom.memory.adapters.sqlite_vector import SQLiteVectorIndex + + # Derive vector db path from main db path if not specified + if config.vector_db_path: + vector_db_path = config.vector_db_path + else: + # "memory.db" -> "memory_vectors.db" + vector_db_path = config.db_path.parent / f"{config.db_path.stem}_vectors.db" + + return SQLiteVectorIndex( + dimension=config.vector_dimension, + db_path=vector_db_path, + page_cache_size_kb=config.vector_cache_size_kb, + ) + + if backend == VectorBackend.HNSW: + from headroom.memory.adapters import HNSW_AVAILABLE + + if not HNSW_AVAILABLE: + raise ValueError( + "hnswlib is not available. Install with: pip install hnswlib\n" + "Or use vector_backend=VectorBackend.SQLITE_VEC" + ) + from headroom.memory.adapters.hnsw import HNSWVectorIndex return HNSWVectorIndex( @@ -149,6 +192,7 @@ def _create_vector_index(config: MemoryConfig) -> VectorIndex: ef_construction=config.hnsw_ef_construction, m=config.hnsw_m, ef_search=config.hnsw_ef_search, + max_entries=config.hnsw_max_entries, ) raise ValueError(f"Unknown vector backend: {config.vector_backend}") diff --git a/tests/test_memory_tracker_integration.py b/tests/test_memory_tracker_integration.py index fbaa4cf88..6be8ef62c 100644 --- a/tests/test_memory_tracker_integration.py +++ b/tests/test_memory_tracker_integration.py @@ -10,6 +10,14 @@ import pytest from headroom.memory.tracker import MemoryTracker +# Check HNSW availability for skipping tests +try: + from headroom.memory.adapters.hnsw import _check_hnswlib_available + + HNSW_AVAILABLE = _check_hnswlib_available() +except ImportError: + HNSW_AVAILABLE = False + class TestCompressionStoreMemoryTracking: """Tests for CompressionStore memory tracking integration.""" @@ -211,6 +219,7 @@ class TestGraphStoreMemoryTracking: assert final_stats.entry_count == 100 +@pytest.mark.skipif(not HNSW_AVAILABLE, reason="hnswlib not available") class TestHNSWVectorIndexMemoryTracking: """Tests for HNSWVectorIndex memory tracking integration.""" diff --git a/tests/test_memory_usage_integration.py b/tests/test_memory_usage_integration.py index d6903cae8..991279a01 100644 --- a/tests/test_memory_usage_integration.py +++ b/tests/test_memory_usage_integration.py @@ -24,6 +24,14 @@ from dotenv import load_dotenv load_dotenv() +# Check HNSW availability for skipping tests +try: + from headroom.memory.adapters.hnsw import _check_hnswlib_available + + HNSW_AVAILABLE = _check_hnswlib_available() +except ImportError: + HNSW_AVAILABLE = False + def get_process_memory_mb() -> float: """Get current process memory in MB.""" @@ -124,6 +132,7 @@ class TestMemorySystemIntegration: print(f"\nTotal tracked memory: {report.total_tracked_mb:.4f} MB") print(f"Process RSS: {report.process.rss_mb:.1f} MB") + @pytest.mark.skipif(not HNSW_AVAILABLE, reason="hnswlib not available") @pytest.mark.asyncio async def test_hnsw_vector_index_memory_growth(self): """Test that HNSW vector index memory is tracked as vectors are added.""" @@ -450,6 +459,7 @@ class TestCombinedMemoryTracking: MemoryTracker.reset() reset_batch_context_store() + @pytest.mark.skipif(not HNSW_AVAILABLE, reason="hnswlib not available") @pytest.mark.asyncio async def test_all_components_memory_tracking(self): """Test memory tracking with all components active.""" @@ -563,6 +573,7 @@ class TestCombinedMemoryTracking: total_from_components = sum(c.size_bytes for c in report.components.values()) assert report.total_tracked_bytes == total_from_components + @pytest.mark.skipif(not HNSW_AVAILABLE, reason="hnswlib not available") @pytest.mark.asyncio async def test_memory_budget_enforcement(self): """Test that budget enforcement works correctly."""