mirror of
https://github.com/headroomlabs-ai/headroom.git
synced 2026-08-27 14:17:10 -04:00
Adds crates/headroom-core/src/transforms/code_compressor.rs (1,882 lines): the AST-aware CodeCompressor ported to Rust on tree-sitter, with grammars for Python, JavaScript, TypeScript, Go, Rust, Java, C and C++. Parity-only, like #1153. Nothing calls it: the only references outside the module are the pub mod / pub use declarations in transforms/mod.rs, and live_zone.rs still routes SourceCode to a no-op. The pyo3 bridge is untouched and no Python source changes, so the engine is unreachable from the shipped package. #1155 wires it into live-zone dispatch. Every grammar is pinned with '=' to the exact version of the corresponding Python tree-sitter-<lang> PyPI wheel. Same version on crates.io and PyPI means the same grammar.js, hence the same generated parser.c, hence node-for-node identical ASTs — the precondition for byte-parity. A canary over 9 samples x 8 languages confirmed identical node-type and line-span trees at these pins; bumping any pin requires re-running it and re-recording the fixtures. Ships 30 recorded parity fixtures, a CodeCompressorComparator in headroom-parity, and scripts/record_code_compressor_fixtures.py. Verified byte-identical to the recorded Python output: [code_aware_compressor] total=30 matched=30 skipped=0 diffed=0 Full harness on the merge result: 227 fixtures, 182 matched, 45 skipped (cache_aligner + ccr stubs), 0 diffed, exit 0 — with kompress at 21/21 under ONNX Runtime 1.24.4 (see #2591). Also verified cargo check -p headroom-core --no-default-features passes, so the static-musl path stays intact.
37 lines
2.6 KiB
JSON
37 lines
2.6 KiB
JSON
{
|
|
"config": {
|
|
"ccr_ttl": 300,
|
|
"compress_comments": true,
|
|
"docstring_mode": "first_line",
|
|
"enable_ccr": false,
|
|
"fallback_to_kompress": false,
|
|
"language_hint": null,
|
|
"max_body_lines": 5,
|
|
"min_tokens_for_compression": 100,
|
|
"preserve_decorators": true,
|
|
"preserve_imports": true,
|
|
"preserve_signatures": true,
|
|
"preserve_type_annotations": true,
|
|
"semantic_analysis": true,
|
|
"target_compression_rate": 0.2
|
|
},
|
|
"input": "package main\n\nimport (\n\t\"fmt\"\n\t\"strings\"\n)\n\ntype Processor struct {\n\tName string\n\tCount int\n}\n\nfunc (p *Processor) Process(items []string) []string {\n\tresults := make([]string, 0, len(items))\n\tfor _, item := range items {\n\t\tif item == \"\" {\n\t\t\tcontinue\n\t\t}\n\t\tclean := strings.ToLower(strings.TrimSpace(item))\n\t\tresults = append(results, clean)\n\t\tp.Count++\n\t}\n\treturn results\n}\n\nfunc main() {\n\tp := &Processor{Name: \"main\"}\n\tfmt.Println(p.Process([]string{\"a\", \"b\"}))\n}\n\n// variant 1",
|
|
"input_sha256": "716a5a2380907e7165334f0d0ce4c4502d6756647f8cb24dacf9b4882c1f8ede",
|
|
"output": {
|
|
"cache_key": null,
|
|
"compressed": "package main\n\nimport (\n\t\"fmt\"\n\t\"strings\"\n)\n\ntype Processor struct {\n\tName string\n\tCount int\n}\n\nfunc (p *Processor) Process(items []string) []string {\n\tresults := make([]string, 0, len(items))\n\tfor _, item := range items {\n\t\tif item == \"\" {\n\t\t\tcontinue\n\t\t}\n\t\tclean := strings.ToLower(strings.TrimSpace(item))\n\t\tresults = append(results, clean)\n\t\tp.Count++\n\t}\n\treturn results\n}\n\nfunc main() {\n\tp := &Processor{Name: \"main\"}\n\tfmt.Println(p.Process([]string{\"a\", \"b\"}))\n}\n\n// variant 1",
|
|
"compressed_bodies": 0,
|
|
"compressed_tokens": 120,
|
|
"compression_ratio": 1.0,
|
|
"language": "go",
|
|
"language_confidence": 1.0,
|
|
"original": "package main\n\nimport (\n\t\"fmt\"\n\t\"strings\"\n)\n\ntype Processor struct {\n\tName string\n\tCount int\n}\n\nfunc (p *Processor) Process(items []string) []string {\n\tresults := make([]string, 0, len(items))\n\tfor _, item := range items {\n\t\tif item == \"\" {\n\t\t\tcontinue\n\t\t}\n\t\tclean := strings.ToLower(strings.TrimSpace(item))\n\t\tresults = append(results, clean)\n\t\tp.Count++\n\t}\n\treturn results\n}\n\nfunc main() {\n\tp := &Processor{Name: \"main\"}\n\tfmt.Println(p.Process([]string{\"a\", \"b\"}))\n}\n\n// variant 1",
|
|
"original_tokens": 120,
|
|
"preserved_imports": 0,
|
|
"preserved_signatures": 0,
|
|
"symbol_scores": {},
|
|
"syntax_valid": true
|
|
},
|
|
"recorded_at": "2026-06-19T00:15:16.342039+00:00",
|
|
"transform": "code_aware_compressor"
|
|
}
|