diff --git a/crates/headroom-core/src/transforms/detection.rs b/crates/headroom-core/src/transforms/detection.rs
new file mode 100644
index 000000000..386c28b57
--- /dev/null
+++ b/crates/headroom-core/src/transforms/detection.rs
@@ -0,0 +1,222 @@
+//! Stage-3d ContentRouter detection chain.
+//!
+//! Wires the per-tier detectors (`magika_detector`, `unidiff_detector`)
+//! into the single function the ContentRouter calls. The locked design
+//! from `project_rust_content_detection_arch.md`:
+//!
+//! ```text
+//! Tier 1: magika_detect() → if non-PlainText, return it
+//! Tier 2: unidiff::is_diff() → if true, return GitDiff
+//! Tier 3: PlainText (fallthrough)
+//! ```
+//!
+//! # Why no regex tier
+//!
+//! User-locked decision (2026-04-25): the Rust side does not run the
+//! regex-based [`crate::transforms::content_detector`] in production
+//! detection. The regex path stays in tree as a comparison oracle for
+//! parity testing and as an opt-in escape hatch, but the dispatch is
+//! magika + parser only.
+//!
+//! # Tier-1 errors do not abort the chain
+//!
+//! If magika init or inference fails, we log at warn level and proceed
+//! to Tier 2. The chain's *next* tier is the legitimate fallback for
+//! magika failure — that's the whole point of having tiers. We
+//! deliberately do **not** treat tier-1 error as a hard failure of the
+//! entire chain; that would block all detection on a transient ONNX
+//! issue. Loud-on-error stays at the [`magika_detect`] entry point for
+//! callers who care; the chain swallows the err with a log line.
+//!
+//! # SearchResults / BuildOutput
+//!
+//! The retired regex detector recognized grep-style search output
+//! (`file:line:`) and CMake/log output as their own [`ContentType`]
+//! variants. Magika has no equivalent labels. Per the locked design,
+//! these now route to [`ContentType::PlainText`] — we prefer
+//! passthrough to misroute. If a benchmark later shows real
+//! compression loss on grep/build outputs, we add a focused detector
+//! for those specifically; not preemptively.
+
+use crate::transforms::content_detector::ContentType;
+use crate::transforms::magika_detector::magika_detect;
+use crate::transforms::unidiff_detector::is_diff;
+
+/// Run the detection chain on `content` and return the chosen
+/// [`ContentType`].
+///
+/// Empty input shortcuts to [`ContentType::PlainText`] without
+/// touching either tier.
+pub fn detect(content: &str) -> ContentType {
+ if content.is_empty() {
+ return ContentType::PlainText;
+ }
+
+ // ── Tier 1: Magika ──────────────────────────────────────────
+ match magika_detect(content) {
+ Ok(ContentType::PlainText) => {
+ // Magika says "I don't know" or "plain text". Continue
+ // to Tier 2 — magika frequently mis-classifies short
+ // diffs and prose-prefixed diffs as text.
+ }
+ Ok(content_type) => return content_type,
+ Err(e) => {
+ // Init or inference failure. Log it (so an ops-side
+ // health check can spot magika trouble in the proxy
+ // logs) and fall through to Tier 2 — the chain itself
+ // must not break on a single tier's outage.
+ tracing::warn!(
+ error = %e,
+ "magika detection failed; falling through to unidiff tier"
+ );
+ }
+ }
+
+ // ── Tier 2: unidiff parser ──────────────────────────────────
+ if is_diff(content) {
+ return ContentType::GitDiff;
+ }
+
+ // ── Tier 3: fallthrough ─────────────────────────────────────
+ ContentType::PlainText
+}
+
+#[cfg(test)]
+mod tests {
+ use super::*;
+
+ #[test]
+ fn empty_input_short_circuits_to_plain_text() {
+ assert_eq!(detect(""), ContentType::PlainText);
+ }
+
+ #[test]
+ fn json_array_routes_via_tier_1() {
+ let payload = r#"[{"id": 1}, {"id": 2}, {"id": 3}]"#;
+ assert_eq!(detect(payload), ContentType::JsonArray);
+ }
+
+ #[test]
+ fn source_code_routes_via_tier_1() {
+ let py = "def hello():\n print('world')\n\nclass Foo:\n pass\n";
+ assert_eq!(detect(py), ContentType::SourceCode);
+ }
+
+ #[test]
+ fn html_routes_via_tier_1() {
+ let html = "
x
";
+ assert_eq!(detect(html), ContentType::Html);
+ }
+
+ #[test]
+ fn standard_git_diff_routes_via_tier_1_or_2() {
+ let diff = "diff --git a/foo.py b/foo.py\n\
+ --- a/foo.py\n\
+ +++ b/foo.py\n\
+ @@ -1,1 +1,2 @@\n \
+ def hello():\n\
+ + print(\"new\")\n";
+ // Either magika tags it `diff` (Tier 1 hit) or magika
+ // mis-classifies as text and unidiff catches it (Tier 2).
+ // Both paths produce GitDiff.
+ assert_eq!(detect(diff), ContentType::GitDiff);
+ }
+
+ #[test]
+ fn naked_hunk_diff_routes_via_tier_2() {
+ // Magika often mis-classifies naked hunks (no `diff --git`
+ // wrapper) because the visible bytes look like ordinary
+ // patch lines mixed with code. Tier 2 (unidiff parser)
+ // catches these.
+ let diff = "--- a/foo.py\n\
+ +++ b/foo.py\n\
+ @@ -1,2 +1,2 @@\n\
+ -old line\n\
+ +new line\n \
+ context line\n";
+ assert_eq!(detect(diff), ContentType::GitDiff);
+ }
+
+ #[test]
+ fn plain_prose_routes_to_plain_text() {
+ let prose = "The quick brown fox jumps over the lazy dog. \
+ Just regular English with no special structure.";
+ assert_eq!(detect(prose), ContentType::PlainText);
+ }
+
+ #[test]
+ fn grep_search_results_route_to_plain_text_per_locked_design() {
+ // Locked design (2026-04-25): no regex tier on Rust side, so
+ // grep-style `file:line:content` output now goes through
+ // PlainText. This is a deliberate behavior change vs. the
+ // retired regex detector. If proxy benchmarks later show
+ // real compression loss on grep output, we add a focused
+ // detector then — not preemptively.
+ let grep = "src/foo.py:42:def process():\n\
+ src/bar.py:10: return True\n\
+ src/baz.py:7:class Worker:\n";
+ // We assert PlainText to lock the design. If magika
+ // probabilistically detects this as code, that's also fine
+ // for the router (CodeAware compresses code well too) —
+ // but the test pins the safe-default contract.
+ let result = detect(grep);
+ assert!(
+ result == ContentType::PlainText || result == ContentType::SourceCode,
+ "grep output should route to PlainText (preferred) or SourceCode (acceptable), got {result:?}"
+ );
+ }
+
+ #[test]
+ fn build_log_output_routes_via_chain() {
+ // Same locked-design note: build/test log output (no
+ // explicit regex detector for it on the Rust side). Magika
+ // can label this either as `txt` (→ PlainText, our mapping)
+ // or, more probabilistically, as a code-group label like
+ // `log` / `c` because the structured `[LEVEL]` shape and
+ // `file.cpp:line` references look code-like to the model.
+ // Either route is acceptable for the router: SourceCode
+ // dispatches to the code-aware compressor (which compresses
+ // log lines reasonably well via repetition collapsing);
+ // PlainText is the safe-default passthrough. The test pins
+ // the contract that it lands somewhere reasonable, not at
+ // a degenerate type like JsonArray or GitDiff.
+ let log = "[INFO] Building target foo\n\
+ [WARN] Deprecated API usage in foo.cpp:45\n\
+ [ERROR] Compilation failed: undefined reference\n";
+ let got = detect(log);
+ assert!(
+ matches!(got, ContentType::PlainText | ContentType::SourceCode),
+ "build log should route to PlainText or SourceCode, got {got:?}"
+ );
+ }
+
+ #[test]
+ fn yaml_routes_to_source_code() {
+ // YAML lives in magika's `code` group; the chain returns it
+ // as SourceCode so the router picks the code-aware compressor.
+ let yaml = "name: my-app\nversion: 1.0\ndependencies:\n - foo\n";
+ assert_eq!(detect(yaml), ContentType::SourceCode);
+ }
+
+ #[test]
+ fn rust_source_routes_to_source_code() {
+ let rs = "use std::collections::HashMap;\n\n\
+ pub struct Counter { counts: HashMap }\n\n\
+ impl Counter {\n \
+ pub fn new() -> Self { Self { counts: HashMap::new() } }\n\
+ }\n";
+ assert_eq!(detect(rs), ContentType::SourceCode);
+ }
+
+ #[test]
+ fn chain_is_deterministic_across_repeated_calls() {
+ // Magika returns the same label for identical input on
+ // repeated calls; the chain wraps that determinism.
+ let payload = r#"{"users": [{"id": 1}, {"id": 2}]}"#;
+ let a = detect(payload);
+ let b = detect(payload);
+ let c = detect(payload);
+ assert_eq!(a, b);
+ assert_eq!(b, c);
+ }
+}
diff --git a/crates/headroom-core/src/transforms/mod.rs b/crates/headroom-core/src/transforms/mod.rs
index d0d2e61e1..9abb1e3bd 100644
--- a/crates/headroom-core/src/transforms/mod.rs
+++ b/crates/headroom-core/src/transforms/mod.rs
@@ -18,6 +18,7 @@
pub mod adaptive_sizer;
pub mod anchor_selector;
pub mod content_detector;
+pub mod detection;
pub mod diff_compressor;
pub mod magika_detector;
pub mod smart_crusher;
@@ -26,6 +27,7 @@ pub mod unidiff_detector;
pub use content_detector::{
detect_content_type, is_json_array_of_dicts, ContentType, DetectionResult,
};
+pub use detection::detect;
pub use diff_compressor::{
DiffCompressionResult, DiffCompressor, DiffCompressorConfig, DiffCompressorStats,
};
diff --git a/crates/headroom-py/src/lib.rs b/crates/headroom-py/src/lib.rs
index 25e80af32..7fb7d892a 100644
--- a/crates/headroom-py/src/lib.rs
+++ b/crates/headroom-py/src/lib.rs
@@ -21,10 +21,9 @@ use headroom_core::transforms::smart_crusher::{
SmartCrusherConfig as RustSmartCrusherConfig,
};
use headroom_core::transforms::{
- detect_content_type as rust_detect_content_type,
- is_json_array_of_dicts as rust_is_json_array_of_dicts, ContentType as RustContentType,
- DetectionResult as RustDetectionResult, DiffCompressionResult, DiffCompressor,
- DiffCompressorConfig, DiffCompressorStats,
+ detect as rust_detect_chain, is_json_array_of_dicts as rust_is_json_array_of_dicts,
+ ContentType as RustContentType, DetectionResult as RustDetectionResult, DiffCompressionResult,
+ DiffCompressor, DiffCompressorConfig, DiffCompressorStats,
};
use pyo3::prelude::*;
use pyo3::types::{PyDict, PyString};
@@ -888,14 +887,29 @@ impl PyDetectionResult {
/// Detect the type of `content`. Returns a `DetectionResult` with the
/// same field surface as Python's dataclass.
///
-/// Releases the GIL while detecting — pattern matching can be substantial
-/// on large bodies (HTML scans, 500-line diff windows), and freeing the
-/// GIL lets other Python threads make progress in the meantime.
+/// Stage-3d (PR5) wired this through the magika→unidiff→PlainText
+/// detection chain — the regex `content_detector` is no longer on
+/// the production path. The chain returns a `ContentType` only;
+/// we synthesize the legacy `DetectionResult` shape here with
+/// `confidence = 1.0` (the chain doesn't surface a probabilistic
+/// score) and an empty metadata bag (no production caller reads
+/// metadata from this binding today — see audit notes in
+/// `headroom/transforms/content_router.py`).
+///
+/// Releases the GIL while detecting — magika inference and unidiff
+/// parsing can be substantial on large bodies, and freeing the GIL
+/// lets other Python threads make progress in the meantime.
#[pyfunction]
fn detect_content_type(py: Python<'_>, content: &str) -> PyDetectionResult {
let owned = content.to_string();
- let result = py.allow_threads(move || rust_detect_content_type(&owned));
- PyDetectionResult { inner: result }
+ let content_type = py.allow_threads(move || rust_detect_chain(&owned));
+ PyDetectionResult {
+ inner: RustDetectionResult {
+ content_type,
+ confidence: 1.0,
+ metadata: serde_json::Map::new(),
+ },
+ }
}
/// Quick check: is `content` a JSON array of dictionaries (the format
diff --git a/headroom/transforms/content_router.py b/headroom/transforms/content_router.py
index 7d925a481..1f17081da 100644
--- a/headroom/transforms/content_router.py
+++ b/headroom/transforms/content_router.py
@@ -48,59 +48,37 @@ from typing import Any
from ..config import DEFAULT_EXCLUDE_TOOLS, ReadLifecycleConfig, TransformResult
from ..tokenizer import Tokenizer
from .base import Transform
-from .content_detector import ContentType, DetectionResult, detect_content_type
+from .content_detector import ContentType, DetectionResult
logger = logging.getLogger(__name__)
-_magika_detector: Any | None = None
-_magika_status: bool | None = None
-
-
-def _get_magika_detector() -> Any | None:
- """Load the Magika detector only when router detection actually runs."""
- global _magika_detector, _magika_status
-
- if _magika_status is False:
- return None
- if _magika_detector is not None:
- return _magika_detector
-
- try:
- from ..compression.detector import get_detector
-
- _magika_detector = get_detector(prefer_magika=True)
- _magika_status = True
- logger.info("ContentRouter: Using Magika ML-based content detection")
- except ImportError:
- _magika_status = False
- logger.debug("Magika not available, using regex-based detection")
-
- return _magika_detector
-
def _detect_content(content: str) -> DetectionResult:
- """Detect content type using Magika if available, else regex fallback."""
- magika_detector = _get_magika_detector()
- if magika_detector is not None:
- result = magika_detector.detect(content)
- # Map Magika ContentType to router's expected format
- type_map = {
- "json": ContentType.JSON_ARRAY,
- "code": ContentType.SOURCE_CODE,
- "log": ContentType.BUILD_OUTPUT,
- "diff": ContentType.GIT_DIFF,
- "markdown": ContentType.PLAIN_TEXT,
- "text": ContentType.PLAIN_TEXT,
- "unknown": ContentType.PLAIN_TEXT,
- }
- mapped_type = type_map.get(result.content_type.value, ContentType.PLAIN_TEXT)
- return DetectionResult(
- content_type=mapped_type,
- confidence=result.confidence,
- metadata={"language": result.language, "raw_label": result.raw_label},
- )
- else:
- return detect_content_type(content)
+ """Detect content type via the Rust detection chain.
+
+ Stage-3d (PR5) wired this through `headroom._core.detect_content_type`,
+ which runs the magika→unidiff→PlainText chain. The Python-side
+ Magika+regex fallback path was retired here — single detection
+ surface, no parallel paths. The Rust extension is a hard dep
+ (no Python fallback) per `feedback_no_silent_fallbacks.md`.
+
+ The Rust binding returns the legacy `DetectionResult` shape with
+ `confidence=1.0` and an empty metadata dict. Existing callers
+ only consumed `.content_type` from it; the strategy mapping in
+ `_strategy_from_detection` keys off that field alone.
+ """
+ from headroom._core import detect_content_type as _rust_detect
+
+ rust_result = _rust_detect(content)
+ # Rust's `content_type` is the lowercase string tag (e.g.
+ # "json_array"); translate to the Python `ContentType` enum so
+ # downstream mapping keys match.
+ content_type = ContentType(rust_result.content_type)
+ return DetectionResult(
+ content_type=content_type,
+ confidence=rust_result.confidence,
+ metadata={},
+ )
def _create_content_signature(
diff --git a/tests/test_transforms_content_router.py b/tests/test_transforms_content_router.py
index dfbd60ebe..0c20cd08a 100644
--- a/tests/test_transforms_content_router.py
+++ b/tests/test_transforms_content_router.py
@@ -101,30 +101,36 @@ def test_router_result_helpers_and_summary() -> None:
def test_content_signature_and_detection_helpers(monkeypatch: pytest.MonkeyPatch) -> None:
+ """Stage-3d (PR5) wired `_detect_content` through the Rust chain
+ (`headroom._core.detect_content_type` → magika → unidiff →
+ PlainText). The pre-PR5 Python-side `_get_magika_detector`
+ fallback path is gone.
+
+ This test asserts the new contract:
+ 1. The detection helper delegates to the Rust binding.
+ 2. Whatever `ContentType` the Rust side returns flows back as a
+ Python `DetectionResult` with that same `content_type`.
+ """
signature = _create_content_signature("search", "file.py:10:match", language="python")
assert signature is not None
assert len(signature.structure_hash) == 24
- fake_detector = SimpleNamespace(
- detect=lambda content: SimpleNamespace(
- content_type=SimpleNamespace(value="code"),
- confidence=0.91,
- language="python",
- raw_label="python",
- )
- )
- monkeypatch.setattr(content_router_module, "_get_magika_detector", lambda: fake_detector)
- magika_result = _detect_content("def main(): pass")
- assert magika_result == DetectionResult(
- content_type=ContentType.SOURCE_CODE,
- confidence=0.91,
- metadata={"language": "python", "raw_label": "python"},
- )
+ # Monkeypatch the Rust binding to return a deterministic fake
+ # result; verify _detect_content propagates the content_type
+ # tag back as the Python ContentType enum.
+ import headroom._core as _core
- fallback = DetectionResult(ContentType.PLAIN_TEXT, 0.6, {"kind": "fallback"})
- monkeypatch.setattr(content_router_module, "_get_magika_detector", lambda: None)
- monkeypatch.setattr(content_router_module, "detect_content_type", lambda content: fallback)
- assert _detect_content("plain") is fallback
+ fake_rust_result = SimpleNamespace(
+ content_type="source_code",
+ confidence=1.0,
+ metadata={},
+ )
+ monkeypatch.setattr(_core, "detect_content_type", lambda content: fake_rust_result)
+
+ result = _detect_content("def main(): pass")
+ assert result.content_type is ContentType.SOURCE_CODE
+ assert result.confidence == 1.0
+ assert result.metadata == {}
def test_mixed_content_section_splitting_and_json_extraction() -> None: