mirror of
https://github.com/headroomlabs-ai/headroom.git
synced 2026-08-27 14:17:10 -04:00
Adds crates/headroom-core/src/transforms/code_compressor.rs (1,882 lines): the AST-aware CodeCompressor ported to Rust on tree-sitter, with grammars for Python, JavaScript, TypeScript, Go, Rust, Java, C and C++. Parity-only, like #1153. Nothing calls it: the only references outside the module are the pub mod / pub use declarations in transforms/mod.rs, and live_zone.rs still routes SourceCode to a no-op. The pyo3 bridge is untouched and no Python source changes, so the engine is unreachable from the shipped package. #1155 wires it into live-zone dispatch. Every grammar is pinned with '=' to the exact version of the corresponding Python tree-sitter-<lang> PyPI wheel. Same version on crates.io and PyPI means the same grammar.js, hence the same generated parser.c, hence node-for-node identical ASTs — the precondition for byte-parity. A canary over 9 samples x 8 languages confirmed identical node-type and line-span trees at these pins; bumping any pin requires re-running it and re-recording the fixtures. Ships 30 recorded parity fixtures, a CodeCompressorComparator in headroom-parity, and scripts/record_code_compressor_fixtures.py. Verified byte-identical to the recorded Python output: [code_aware_compressor] total=30 matched=30 skipped=0 diffed=0 Full harness on the merge result: 227 fixtures, 182 matched, 45 skipped (cache_aligner + ccr stubs), 0 diffed, exit 0 — with kompress at 21/21 under ONNX Runtime 1.24.4 (see #2591). Also verified cargo check -p headroom-core --no-default-features passes, so the static-musl path stays intact.
40 lines
2.1 KiB
JSON
40 lines
2.1 KiB
JSON
{
|
|
"config": {
|
|
"ccr_ttl": 300,
|
|
"compress_comments": true,
|
|
"docstring_mode": "first_line",
|
|
"enable_ccr": false,
|
|
"fallback_to_kompress": false,
|
|
"language_hint": null,
|
|
"max_body_lines": 5,
|
|
"min_tokens_for_compression": 100,
|
|
"preserve_decorators": true,
|
|
"preserve_imports": true,
|
|
"preserve_signatures": true,
|
|
"preserve_type_annotations": true,
|
|
"semantic_analysis": true,
|
|
"target_compression_rate": 0.2
|
|
},
|
|
"input": "from collections import defaultdict\n\n\ndef build_index(records):\n index = defaultdict(list)\n for rec in records:\n key = rec.get(\"id\")\n if key is None:\n continue\n index[key].append(rec)\n if len(index[key]) > 100:\n index[key] = index[key][:100]\n return index\n\n\ndef merge(a, b):\n out = dict(a)\n for k, v in b.items():\n out[k] = v\n return out\n",
|
|
"input_sha256": "921876ad5dd26cf623a97a7668ec4a1026f9f8b4e2d97af154943f29482c160c",
|
|
"output": {
|
|
"cache_key": null,
|
|
"compressed": "from collections import defaultdict\n\ndef build_index(records):\n index = defaultdict(list)\n # [8 lines omitted]\n pass\ndef merge(a, b):\n out = dict(a)\n # [3 lines omitted]\n pass",
|
|
"compressed_bodies": 0,
|
|
"compressed_tokens": 48,
|
|
"compression_ratio": 0.46601941747572817,
|
|
"language": "python",
|
|
"language_confidence": 1.0,
|
|
"original": "from collections import defaultdict\n\n\ndef build_index(records):\n index = defaultdict(list)\n for rec in records:\n key = rec.get(\"id\")\n if key is None:\n continue\n index[key].append(rec)\n if len(index[key]) > 100:\n index[key] = index[key][:100]\n return index\n\n\ndef merge(a, b):\n out = dict(a)\n for k, v in b.items():\n out[k] = v\n return out\n",
|
|
"original_tokens": 103,
|
|
"preserved_imports": 1,
|
|
"preserved_signatures": 2,
|
|
"symbol_scores": {
|
|
"build_index": 0.5,
|
|
"merge": 0.5
|
|
},
|
|
"syntax_valid": true
|
|
},
|
|
"recorded_at": "2026-06-19T00:15:16.319542+00:00",
|
|
"transform": "code_aware_compressor"
|
|
}
|