headroom/tests/test_cli/test_install_cli.py

Ignoring revisions in .git-blame-ignore-revs. Click here to bypass and see the normal blame view.

590 lines
21 KiB
Python
Raw Normal View History

from __future__ import annotations
import click
from click.testing import CliRunner
from headroom.cli.main import main
def test_install_apply_starts_service_supervisor(monkeypatch) -> None:
runner = CliRunner()
calls: list[str] = []
class Manifest:
profile = "default"
preset = "persistent-service"
runtime_kind = "python"
supervisor_kind = "service"
scope = "user"
health_url = "http://127.0.0.1:8787/readyz"
targets = ["claude", "codex"]
mutations = []
artifacts = []
manifest = Manifest()
monkeypatch.setattr("headroom.cli.install.build_manifest", lambda **_: manifest)
monkeypatch.setattr("headroom.cli.install.load_manifest", lambda profile: None)
monkeypatch.setattr("headroom.cli.install.apply_mutations", lambda deployment: [])
monkeypatch.setattr("headroom.cli.install.install_supervisor", lambda deployment: [])
monkeypatch.setattr(
"headroom.cli.install.save_manifest", lambda deployment: calls.append("save")
)
monkeypatch.setattr(
"headroom.cli.install.start_supervisor", lambda deployment: calls.append("start_service")
)
monkeypatch.setattr(
"headroom.cli.install.start_detached_agent", lambda profile: calls.append("start_agent")
)
monkeypatch.setattr(
"headroom.cli.install.start_persistent_docker",
lambda deployment: calls.append("start_docker"),
)
monkeypatch.setattr(
"headroom.cli.install.wait_ready", lambda deployment, timeout_seconds=45: True
)
result = runner.invoke(main, ["install", "apply"])
assert result.exit_code == 0, result.output
assert "Installed persistent deployment 'default'" in result.output
assert "Targets: claude, codex" in result.output
assert calls == ["save", "start_service"]
def test_install_status_includes_backend_from_health_probe(monkeypatch) -> None:
runner = CliRunner()
class Manifest:
profile = "default"
preset = "persistent-service"
runtime_kind = "python"
supervisor_kind = "service"
scope = "user"
port = 8787
backend = "anthropic"
health_url = "http://127.0.0.1:8787/readyz"
monkeypatch.setattr("headroom.cli.install.load_manifest", lambda profile: Manifest())
monkeypatch.setattr("headroom.cli.install.runtime_status", lambda manifest: "running")
monkeypatch.setattr("headroom.cli.install.probe_ready", lambda url: True)
monkeypatch.setattr(
"headroom.cli.install.probe_json",
lambda url: {"config": {"backend": "anthropic"}},
)
result = runner.invoke(main, ["install", "status"])
assert result.exit_code == 0, result.output
assert "Status: running" in result.output
assert "Healthy: yes" in result.output
assert "Backend: anthropic" in result.output
def test_install_restart_uses_internal_helpers(monkeypatch) -> None:
runner = CliRunner()
calls: list[str] = []
class Manifest:
profile = "default"
preset = "persistent-service"
runtime_kind = "python"
supervisor_kind = "service"
scope = "user"
health_url = "http://127.0.0.1:8787/readyz"
monkeypatch.setattr("headroom.cli.install.load_manifest", lambda profile: Manifest())
monkeypatch.setattr(
"headroom.cli.install.stop_supervisor", lambda manifest: calls.append("stop_supervisor")
)
monkeypatch.setattr(
"headroom.cli.install.stop_runtime", lambda manifest: calls.append("stop_runtime")
)
monkeypatch.setattr(
"headroom.cli.install.start_supervisor", lambda manifest: calls.append("start_supervisor")
)
monkeypatch.setattr(
"headroom.cli.install.wait_ready", lambda manifest, timeout_seconds=45: True
)
result = runner.invoke(main, ["install", "restart"])
assert result.exit_code == 0, result.output
assert "Restarted deployment 'default'." in result.output
assert calls == ["stop_supervisor", "stop_runtime", "start_supervisor"]
def test_install_apply_rejects_invalid_profile() -> None:
runner = CliRunner()
result = runner.invoke(main, ["install", "apply", "--profile", "../bad"])
assert result.exit_code != 0
assert "Invalid profile name '../bad'" in result.output
def test_install_apply_rejects_provider_scope_targets_without_support() -> None:
runner = CliRunner()
result = runner.invoke(
main,
["install", "apply", "--scope", "provider", "--providers", "manual", "--target", "copilot"],
)
assert result.exit_code != 0
feat: headroom wrap opencode / unwrap opencode CLI (#1105) ## Summary This PR implements transparent `headroom wrap opencode` support without asking users to edit OpenCode provider URLs, choose an extra CLI flag, or maintain a static provider list. The wrapper now lives at the runtime transport boundary: OpenCode keeps its user/provider config, while Headroom intercepts outbound provider traffic in-process and routes it through the local Headroom proxy. ## What changed ### Transparent OpenCode wrapping - `headroom wrap opencode` injects the `headroom-opencode` plugin through `OPENCODE_CONFIG_CONTENT`. - Existing OpenCode provider URLs are preserved. We do not rewrite user config URLs to point at Headroom. - Existing `OPENAI_BASE_URL` and `ANTHROPIC_BASE_URL` env vars are preserved. - Local OpenCode traffic, localhost traffic, and Headroom proxy traffic bypass the shim to avoid loops. ### Runtime transport interception - Added an OpenCode plugin transport shim that wraps: - `globalThis.fetch` - `http.request` / `http.get` - `https.request` / `https.get` - External provider calls are routed to the local Headroom proxy. - The original upstream origin is passed through `x-headroom-base-url`, so the proxy can forward to the real provider without changing OpenCode config. - External `http2.connect` is blocked loudly instead of allowing direct provider traffic to leak outside Headroom. ### Live provider additions Provider coverage is no longer based on a static config scan. Because routing happens at outbound request time, providers added mid-session are routed through Headroom automatically as long as they use the covered Node transport paths. ### Subagent and child-process coverage - The parent OpenCode plugin sets a packaged Node preload shim through `NODE_OPTIONS=--import=.../hook-shim/handler.js`. - The transport shim patches `child_process.spawn`, `exec`, `execFile`, and `fork` so child Node processes receive the Headroom preload even when OpenCode passes a custom `env`. - The child-process shim fails closed if it loads without `HEADROOM_OPENCODE_TRANSPORT_PROXY_URL`. - This closes the subagent leak path where a child Node process could otherwise start without Headroom transport interception. ## Why this goes beyond PR #1089 PR #1089 improves OpenCode provider registration, but it still focuses on provider config shape. This PR moves the enforcement boundary to runtime transport interception. This PR goes further because: - No provider URL rewriting is required. - New providers added mid-session are covered automatically. - Subagents and child Node processes inherit the Headroom transport shim. - Direct external HTTP/2 paths fail loudly instead of leaking. - The wrap remains transparent to the user's OpenCode provider config. - The wrapper is fail-closed for unsupported child-process preload state. ## Additional robustness fixes While validating the change in Docker, the full Python suite exposed unrelated Linux/container robustness issues. These are fixed in this PR so the suite is green: - Binary cache handling now treats cache paths under a non-writable existing parent as unavailable, including when tests run as root in Docker. - `release_version.py` honors `MANUAL_VER` before git calls so direct script execution works outside a `.git` checkout. - Test logger isolation now resets relevant Headroom child loggers so proxy logging setup cannot poison later `caplog` tests. - The scanner missing-path test now uses a guaranteed missing `tmp_path` child instead of relying on `/nonexistent/path`. ## Validation All implementation validation was run inside Docker. - Full Python suite from a fresh Docker copy: `6605 passed, 523 skipped`. - Ruff on changed Python/OpenCode paths: passed. - OpenCode plugin typecheck: passed. - OpenCode plugin tests: `9 passed`. - OpenCode plugin build: passed. - Hook shim preload smoke test: passed. ## Notes This PR intentionally does not add a CLI option. `headroom wrap opencode` means full wrap. Either Headroom wraps OpenCode transparently, or the path fails loudly instead of silently leaking provider traffic. --------- Co-authored-by: Rudimar Ronsoni <6081613+rudironsoni@users.noreply.github.com>
2026-06-22 18:07:12 +02:00
assert "Provider scope supports only claude, codex, openclaw, and opencode" in result.output
fix(opencode): write local MCP config (#1381) ## Description Fixes the OpenCode config corruption reported in #1380 for wrap, MCP registration, and provider-scope install paths. OpenCode MCP entries are local stdio servers, not remote HTTP endpoints. This changes Headroom's OpenCode MCP serialization to write `type: "local"` with `command: ["headroom", "mcp", "serve"]`, uses OpenCode's `environment` field for MCP env vars, and still reads the older `env` key for compatibility. This also stops provider-only OpenCode config injection from creating a fake `http://127.0.0.1:<port>/mcp` entry, so `headroom wrap opencode --no-mcp` no longer leaves `mcp.headroom` behind. Finally, the install CLI/docs now accept and document `--target opencode` with provider scope. This does not change the broader `headroom mcp status/uninstall` behavior from #1380; that looks like a separate follow-up. ## Type of Change - [x] Bug fix (non-breaking change that fixes an issue) - [ ] New feature (non-breaking change that adds functionality) - [ ] Breaking change (fix or feature that would cause existing functionality to change) - [x] Documentation update - [ ] Performance improvement - [ ] Code refactoring (no functional changes) ## Changes Made - Write OpenCode MCP entries as local stdio config instead of remote `/mcp` config. - Use `environment` for OpenCode MCP env vars while continuing to read legacy `env` entries. - Stop OpenCode provider injection/persistent provider install from adding MCP config. - Keep `--no-mcp` from writing `mcp.headroom` while preserving other MCP entries such as Serena. - Allow `headroom install apply --target opencode` at the CLI layer. - Update OpenCode docs and changelog. ## Testing - [x] Focused unit tests pass - [x] Linting passes (`ruff check .`) - [x] Formatting passes (`ruff format --check .`) - [x] Type checking passes (`mypy headroom`) - [x] New tests added for the fixed behavior - [x] Manual testing performed ### Test Output ```text $ pytest tests/test_mcp_registry_opencode.py tests/test_cli/test_wrap_opencode.py tests/test_providers_opencode_config.py tests/test_providers_opencode_install.py tests/test_cli/test_install_cli.py tests/test_install/test_providers.py Pytest: 164 passed $ uvx ruff check . All checks passed! $ uvx ruff format --check . 986 files already formatted $ uvx mypy --config-file pyproject.toml headroom Success: no issues found in 398 source files ``` ## Real Behavior Proof - Environment: macOS local worktree at `/Users/vinaygupta/Desktop/git/headroom-fix-opencode-mcp-config`; branch `fix-opencode-mcp-config`; commit `aea96208`. - Exact command / steps: ran the focused OpenCode/installer regression suite plus Ruff lint/format checks and mypy commands shown above. - Observed result: the focused tests pass and cover OpenCode MCP serialization as `type: "local"`, `command: ["headroom", "mcp", "serve"]`, `environment` env vars, `--no-mcp` not writing `mcp.headroom`, provider-scope install not adding MCP config, and `install apply --target opencode` being accepted. - Not tested: full `pytest` locally, because collection requires the native `headroom._core` extension in this worktree. Attempting the project runner hit a local native build failure first: `esaxx-rs` failed compiling `src/esaxx.cpp` with `fatal error: 'cstdint' file not found`. The broader generic `headroom mcp status/uninstall` behavior from #1380 is intentionally left for a follow-up. ## Review Readiness - [x] I have performed a self-review - [x] This PR is ready for human review Scope note: generic `mcp status/uninstall` support from #1380 is intentionally left as a separate follow-up PR.
2026-06-26 12:23:54 -05:00
def test_install_apply_accepts_opencode_target(monkeypatch) -> None:
runner = CliRunner()
captured: dict[str, object] = {}
class Manifest:
profile = "default"
preset = "persistent-service"
runtime_kind = "python"
supervisor_kind = "service"
scope = "provider"
health_url = "http://127.0.0.1:8787/readyz"
targets = ["opencode"]
mutations = []
artifacts = []
manifest = Manifest()
def fake_build_manifest(**kwargs):
captured.update(kwargs)
return manifest
monkeypatch.setattr("headroom.cli.install.build_manifest", fake_build_manifest)
monkeypatch.setattr("headroom.cli.install.load_manifest", lambda profile: None)
monkeypatch.setattr("headroom.cli.install.apply_mutations", lambda deployment: [])
monkeypatch.setattr("headroom.cli.install.install_supervisor", lambda deployment: [])
monkeypatch.setattr("headroom.cli.install.save_manifest", lambda deployment: None)
monkeypatch.setattr("headroom.cli.install.start_supervisor", lambda deployment: None)
monkeypatch.setattr("headroom.cli.install.start_detached_agent", lambda profile: None)
monkeypatch.setattr(
"headroom.cli.install.wait_ready", lambda deployment, timeout_seconds=45: True
)
result = runner.invoke(
main,
[
"install",
"apply",
"--scope",
"provider",
"--providers",
"manual",
"--target",
"opencode",
],
)
assert result.exit_code == 0, result.output
assert captured["targets"] == ["opencode"]
assert "Targets: opencode" in result.output
def test_install_apply_restores_previous_deployment_after_failed_update(monkeypatch) -> None:
runner = CliRunner()
calls: list[str] = []
class Manifest:
def __init__(self, profile: str, targets: list[str]) -> None:
self.profile = profile
self.preset = "persistent-service"
self.runtime_kind = "python"
self.supervisor_kind = "service"
self.scope = "user"
self.health_url = "http://127.0.0.1:8787/readyz"
self.targets = targets
self.mutations = []
self.artifacts = []
new_manifest = Manifest("default", ["claude"])
existing_manifest = Manifest("default", ["codex"])
monkeypatch.setattr("headroom.cli.install.build_manifest", lambda **_: new_manifest)
monkeypatch.setattr("headroom.cli.install.load_manifest", lambda profile: existing_manifest)
monkeypatch.setattr(
"headroom.cli.install.apply_mutations",
lambda deployment: calls.append(f"apply:{','.join(deployment.targets)}") or [],
)
monkeypatch.setattr(
"headroom.cli.install.install_supervisor",
lambda deployment: calls.append(f"supervisor:{','.join(deployment.targets)}") or [],
)
monkeypatch.setattr(
"headroom.cli.install.save_manifest",
lambda deployment: calls.append(f"save:{','.join(deployment.targets)}"),
)
monkeypatch.setattr(
"headroom.cli.install.stop_supervisor",
lambda deployment: calls.append(f"stop-supervisor:{','.join(deployment.targets)}"),
)
monkeypatch.setattr(
"headroom.cli.install.stop_runtime",
lambda deployment: calls.append(f"stop-runtime:{','.join(deployment.targets)}"),
)
monkeypatch.setattr(
"headroom.cli.install.remove_supervisor",
lambda deployment: calls.append(f"remove-supervisor:{','.join(deployment.targets)}"),
)
monkeypatch.setattr(
"headroom.cli.install.revert_mutations",
lambda deployment: calls.append(f"revert:{','.join(deployment.targets)}"),
)
monkeypatch.setattr(
"headroom.cli.install.delete_manifest",
lambda profile: calls.append(f"delete:{profile}"),
)
def _start(deployment) -> None:
calls.append(f"start:{','.join(deployment.targets)}")
if deployment is new_manifest:
raise click.ClickException("boom")
monkeypatch.setattr("headroom.cli.install._start_deployment", _start)
result = runner.invoke(main, ["install", "apply"])
assert result.exit_code != 0
assert "Restoring previous deployment 'default'" in result.output
assert calls == [
"stop-supervisor:codex",
"stop-runtime:codex",
"remove-supervisor:codex",
"revert:codex",
"delete:default",
"apply:claude",
"supervisor:claude",
"save:claude",
"start:claude",
"stop-supervisor:claude",
"stop-runtime:claude",
"remove-supervisor:claude",
"revert:claude",
"delete:default",
"apply:codex",
"supervisor:codex",
"save:codex",
"start:codex",
]
def test_install_start_rejects_task_lifecycle(monkeypatch) -> None:
runner = CliRunner()
class Manifest:
profile = "default"
preset = "persistent-task"
runtime_kind = "python"
supervisor_kind = "task"
scope = "user"
health_url = "http://127.0.0.1:8787/readyz"
monkeypatch.setattr("headroom.cli.install.load_manifest", lambda profile: Manifest())
result = runner.invoke(main, ["install", "start"])
assert result.exit_code != 0
assert "headroom install start" in result.output
def test_install_apply_uses_docker_runtime_for_persistent_docker(monkeypatch) -> None:
runner = CliRunner()
calls: list[str] = []
class Manifest:
profile = "default"
preset = "persistent-docker"
runtime_kind = "docker"
supervisor_kind = "none"
scope = "user"
health_url = "http://127.0.0.1:8787/readyz"
targets: list[str] = []
mutations = []
artifacts = []
monkeypatch.setattr("headroom.cli.install.build_manifest", lambda **_: Manifest())
monkeypatch.setattr("headroom.cli.install.load_manifest", lambda profile: None)
monkeypatch.setattr("headroom.cli.install.apply_mutations", lambda deployment: [])
monkeypatch.setattr("headroom.cli.install.install_supervisor", lambda deployment: [])
monkeypatch.setattr("headroom.cli.install.save_manifest", lambda deployment: None)
monkeypatch.setattr(
"headroom.cli.install.start_persistent_docker",
lambda deployment: calls.append("start_docker"),
)
monkeypatch.setattr(
"headroom.cli.install.wait_ready", lambda deployment, timeout_seconds=45: True
)
fix(cli): harden all CLI surfaces + fix docs accuracy (#1491) ## Summary Full CLI audit + documentation accuracy pass. All 5 commits on this branch: ### CLI Hardening (4 commits) - **Clean errors instead of tracebacks**: corrupt manifests, missing Docker, malformed JSONL, bad `--profile`, invalid env-var values all now raise `click.ClickException` with helpful messages - **Range validation**: ~25 numeric flags across 10 files now use `click.IntRange`/`FloatRange` — `--port 0`, `--hours -1`, `--limit 0` etc. produce clean usage errors instead of silent wrong behavior - **Flag combination warnings**: conflicting combos (`--no-rate-limit` + `--rpm`, `--no-optimize` + `--target-ratio`, `--telemetry` + `--no-telemetry`) emit yellow warnings on stderr - **`memory --db-path` default fixed**: was resolving to `headroom_memory.db` (wrong bare file); now uses project store `./.headroom/memory.db` if present, else `~/.headroom/memory.db` - **`memory list --search` + filters**: `--scope`/`--session`/`--since` were silently ignored when `--search` was also set; now filters are applied to search results - **`learn --verbosity --apply` now works**: the output shaper is off by default (`HEADROOM_OUTPUT_SHAPER`); `--apply` now hot-enables it via `POST /admin/runtime-env` on a running proxy, or prints explicit `export HEADROOM_OUTPUT_SHAPER=1` instructions when no proxy is running - **`perf --hours` overflow**: `1e9` hours no longer raises `OverflowError`; treated as "all data" - **`evals memory --categories` invalid input**: `abc,1,2` now raises `BadParameter` instead of a raw `ValueError` traceback ### Documentation (1 commit, 20 files) Corrected factual errors found by 3 parallel audit agents across root docs, wiki, and the published Fumadocs site: **Critical (caused runtime errors or wrong behavior if followed):** - `simulation.mdx`: `plan.transforms_applied` -> `plan.transforms`; `plan.savings_percent` -> computed from available fields (both raised `AttributeError`) - `shared-context.mdx`: `import { SharedContext } from "headroom"` -> `"headroom-ai"` (5x `ImportError`) - `claude-code-azure-foundry.mdx`: `pip install headroom` -> `pip install headroom-ai` - `api-reference.mdx` + `configuration.mdx`: `from headroom import GoogleProvider` -> `from headroom.providers import GoogleProvider` - `ccr.mdx`: CCR TTL default 300s -> 1800s (30 min) **Fabricated flags removed:** - `wiki/proxy.md` + `wiki/cli.md`: `--no-intelligent-context`, `--no-intelligent-scoring`, `--no-compress-first` (none exist); replaced with real CCR flags - `wiki/configuration.md`: `--no-ccr-responses`, `--no-ccr-expansion` (none exist); replaced with real flags - `wiki/troubleshooting.md`, `wiki/metrics.md`, `docs/troubleshooting.mdx`: `headroom proxy --log-level debug` (flag doesn't exist) **Stale content corrected:** - `llms.txt`: telemetry stated as enabled-by-default (it's opt-in); wrap list had 5 tools (now 11) - `README.md`: compatibility matrix added 5 missing `wrap` targets; `unwrap`, `doctor`, `init`/`install`, savings-analytics now mentioned - `SECURITY.md`: supported version table showed 0.2.x (current: 0.27.x) - `wiki/learn.md`: 5 missing flags added; verbosity shaper-off behavior documented - `wiki/quickstart.md`: "Configuration Reference" linked to `api.md` (wrong) -> `configuration.md` - `CacheAlignerConfig.enabled` default corrected: `True` -> `False` - `opencode.mdx`: `--port` default wrong ("random") -> 8787; `openai` backend removed - `CONTRIBUTING.md`: broken Markdown table cell fixed - `docs/meta.json`: `claude-code-azure-foundry` added to nav (was unreachable orphan page) - `configuration.mdx`: SDK modes vs proxy `--mode` now clearly distinguished ## Test plan - [x] `python -m pytest tests/ -x -q` — 857 passed, 0 failures - [x] 41-combination CLI smoke test (all flag combos across 8 commands) — 0 tracebacks - [x] `ruff check` on all modified Python files — clean - [x] Docs changes are removals/corrections of fabricated or stale content; no new claims introduced
2026-06-27 14:48:43 -07:00
# _start_deployment guards the persistent-docker preset with
# `shutil.which("docker")`. Fake docker as present so the test exercises the
# runtime-selection path itself rather than the host's docker install —
# otherwise it passes on dev machines with Docker but fails on CI runners
# (e.g. macos-latest) that have no docker on PATH.
monkeypatch.setattr(
"headroom.cli.install.shutil.which",
lambda name, *args, **kwargs: "/usr/local/bin/docker" if name == "docker" else None,
)
result = runner.invoke(main, ["install", "apply", "--preset", "persistent-docker"])
assert result.exit_code == 0, result.output
assert calls == ["start_docker"]
def test_install_remove_continues_when_runtime_teardown_errors(monkeypatch) -> None:
runner = CliRunner()
calls: list[str] = []
class Manifest:
profile = "default"
preset = "persistent-service"
runtime_kind = "python"
supervisor_kind = "service"
scope = "user"
health_url = "http://127.0.0.1:8787/readyz"
monkeypatch.setattr("headroom.cli.install.load_manifest", lambda profile: Manifest())
monkeypatch.setattr(
"headroom.cli.install.stop_supervisor",
lambda manifest: (_ for _ in ()).throw(RuntimeError("boom")),
)
monkeypatch.setattr(
"headroom.cli.install.stop_runtime",
lambda manifest: (_ for _ in ()).throw(RuntimeError("boom")),
)
monkeypatch.setattr(
"headroom.cli.install.remove_supervisor", lambda manifest: calls.append("remove_supervisor")
)
monkeypatch.setattr(
"headroom.cli.install.revert_mutations", lambda manifest: calls.append("revert")
)
monkeypatch.setattr(
"headroom.cli.install.delete_manifest", lambda profile: calls.append("delete")
)
result = runner.invoke(main, ["install", "remove"])
assert result.exit_code == 0, result.output
assert calls == ["remove_supervisor", "revert", "delete"]
def test_install_agent_ensure_reports_already_healthy(monkeypatch) -> None:
runner = CliRunner()
class Manifest:
profile = "default"
health_url = "http://127.0.0.1:8787/readyz"
monkeypatch.setattr("headroom.cli.install.load_manifest", lambda profile: Manifest())
monkeypatch.setattr("headroom.cli.install.probe_ready", lambda url: True)
result = runner.invoke(main, ["install", "agent", "ensure"])
assert result.exit_code == 0, result.output
assert "already healthy" in result.output
def test_install_agent_run_exits_with_foreground_status(monkeypatch) -> None:
runner = CliRunner()
class Manifest:
profile = "default"
health_url = "http://127.0.0.1:8787/readyz"
monkeypatch.setattr("headroom.cli.install.load_manifest", lambda profile: Manifest())
monkeypatch.setattr("headroom.cli.install.run_foreground", lambda manifest: 7)
result = runner.invoke(main, ["install", "agent", "run"])
assert result.exit_code == 7
fix(install): guard install_agent_ensure against duplicate runtime spawns (#1301) ## Type of Change - [x] Bug fix (non-breaking change which fixes an issue) ## Description `install_agent_ensure` in `cli/install.py` only checked `probe_ready(health_url)`. If the proxy was alive-but-not-ready (e.g. during cold start while tokenizers load — ~38s on Windows), `probe_ready` returned false and it unconditionally called `_start_deployment` → `start_detached_agent`, spawning a **second runtime** without: 1. acquiring `acquire_runtime_start_lock` 2. checking `runtime_status` 3. stopping the existing instance Two proxies then contend for `127.0.0.1:<port>`; only one can bind, and the deployment ends up wedged (never ready). Every subsequent ensure spawns yet another runtime → restart storm. By contrast, the hook path `cli/init.py:_ensure_profile_running` does it correctly: it acquires the start-lock, checks `runtime_status`, and `stop_runtime`s a wedged instance before starting a fresh one. Closes #1151. ## Changes Made - Added `acquire_runtime_start_lock` to the imports from `install.runtime` in `headroom/cli/install.py` - Rewrote `install_agent_ensure` to mirror the guarded pattern from `_ensure_profile_running` in `cli/init.py`: - Fast-path probe: if proxy is already ready, return immediately (preserves existing behavior) - Lock acquisition: acquire `acquire_runtime_start_lock` — if another ensure holds it, return without spawning (prevents duplicate) - Double-checked locking: re-probe `probe_ready` after acquiring the lock (race window handled) - Wedged instance detection: if `runtime_status` says "running" but proxy isn't ready within 15s grace period, call `stop_runtime` before starting fresh - Fall through to `_start_deployment` only when truly needed - Added `_STARTUP_READY_TIMEOUT_SECONDS = 15` constant (matching the value used in `_ensure_profile_running`) - **Failure propagation (addresses @JerrettDavis's review feedback):** removed the `try/except Exception` wrapper around the guarded block. `install agent ensure` is an automation-facing CLI command and must exit non-zero on failure so callers can distinguish a successful ensure from a failed one. The `init.py` hook path retains its `try/except` because silent retry is intentional there. The control flow is shared; the error contract is intentionally different because the call sites have different needs. - Added 5 regression tests in `tests/test_cli/test_install_cli.py`: - `test_install_agent_ensure_no_spawn_when_lock_not_acquired` — verifies no runtime spawned when lock is contended (the core bug) - `test_install_agent_ensure_stops_wedged_runtime_before_restart` — verifies `stop_runtime` is called BEFORE `_start_deployment` when instance is wedged (ordering assertion: `calls.index("stop") < calls.index("start_deployment")`) - `test_install_agent_ensure_starts_when_stopped_and_lock_acquired` — verifies the normal start path including the real `_start_deployment` → `start_detached_agent` wiring - `test_install_agent_ensure_no_duplicate_spawn_after_lock_recheck` — verifies double-checked locking prevents duplicate when proxy becomes ready between initial probe and lock acquisition - `test_install_agent_ensure_propagates_start_deployment_failure` — **new** regression test for the failure-propagation fix: monkeypatches `_start_deployment` to raise `click.ClickException("simulated start failure")` and asserts both `exit_code != 0` and that the error message survives in output ## Testing - [x] Unit tests pass (`pytest`) - [x] Linting passes (`ruff check .`) - [x] Type checking passes (`mypy headroom`) - [x] New tests added for new functionality - [x] Manual testing performed ### Test Output ``` $ uv run python -m pytest tests/test_cli/test_install_cli.py -v --tb=short tests/test_cli/test_install_cli.py::test_install_apply_starts_service_supervisor PASSED [ 6%] tests/test_cli/test_install_cli.py::test_install_status_includes_backend_from_health_probe PASSED [ 12%] tests/test_cli/test_install_cli.py::test_install_restart_uses_internal_helpers PASSED [ 18%] tests/test_cli/test_install_cli.py::test_install_apply_rejects_invalid_profile PASSED [ 25%] tests/test_cli/test_install_cli.py::test_install_apply_rejects_provider_scope_targets_without_support PASSED [ 31%] tests/test_cli/test_install_cli.py::test_install_apply_restores_previous_deployment_after_failed_update PASSED [ 37%] tests/test_cli/test_install_cli.py::test_install_start_rejects_task_lifecycle PASSED [ 43%] tests/test_cli/test_install_cli.py::test_install_apply_uses_docker_runtime_for_persistent_docker PASSED [ 50%] tests/test_cli/test_install_cli.py::test_install_remove_continues_when_runtime_teardown_errors PASSED [ 56%] tests/test_cli/test_install_cli.py::test_install_agent_ensure_reports_already_healthy PASSED [ 62%] tests/test_cli/test_install_cli.py::test_install_agent_run_exits_with_foreground_status PASSED [ 68%] tests/test_cli/test_install_cli.py::test_install_agent_ensure_no_spawn_when_lock_not_acquired PASSED [ 75%] tests/test_cli/test_install_cli.py::test_install_agent_ensure_stops_wedged_runtime_before_restart PASSED [ 81%] tests/test_cli/test_install_cli.py::test_install_agent_ensure_starts_when_stopped_and_lock_acquired PASSED [ 87%] tests/test_cli/test_install_cli.py::test_install_agent_ensure_no_duplicate_spawn_after_lock_recheck PASSED [ 93%] tests/test_cli/test_install_cli.py::test_install_agent_ensure_propagates_start_deployment_failure PASSED [100%] ============================== 16 passed in 0.29s ============================== ``` ``` $ uv run ruff check headroom/cli/install.py tests/test_cli/test_install_cli.py All checks passed! $ uv run ruff format --check headroom/cli/install.py tests/test_cli/test_install_cli.py 2 files already formatted $ uv run mypy headroom/cli/install.py --ignore-missing-imports Success: no issues found in 1 source file ``` ## Real Behavior Proof - **Environment**: Python 3.11.14, Linux 6.17.0, headroom dev environment (uv-synced), rebased onto `upstream/main` at `3be2526b` - **Exact command / steps**: `uv run python -m pytest tests/test_cli/test_install_cli.py -v --tb=short` (and the ruff + mypy commands above) - **Observed result**: All 16 tests pass (11 existing + 5 new regression tests). The 5 new tests verify: (1) no-spawn when the lock is contended, (2) `stop_runtime` ordering before `_start_deployment` on a wedged instance, (3) normal start path, (4) double-checked locking after the lock is acquired, (5) failure propagation when `_start_deployment` raises — this last test is the regression for @JerrettDavis's review feedback. ruff check, ruff format --check, and mypy all pass clean. - **Not tested**: Live deployment with concurrent `install agent ensure` invocations on Windows (only unit tests with monkeypatched runtime functions). The fix mirrors the proven pattern from `_ensure_profile_running` which is already battle-tested in the init hook path. ## Review Readiness - [x] I have performed a self-review of my own code - [x] This PR is ready for human review ## Checklist - [x] My code follows the project's style guidelines - [x] I have performed a self-review of my code - [x] I have commented my code, particularly in hard-to-understand areas - [x] I have made corresponding changes to the documentation - [x] My changes generate no new warnings - [x] I have added tests that prove my fix is effective or that my feature works - [x] New and existing unit tests pass locally with my changes - [x] I have updated the CHANGELOG.md if applicable ## Screenshots (if applicable) N/A ## Additional Notes **Triage of labels on this PR:** - `status: needs author action` — **stale**. The 4 `Real Behavior Proof` fields (`Environment`, `Exact command / steps`, `Observed result`, `Not tested`) are all present in this body. The bot snapshot was taken before the body was filled in. Requesting the label be dropped on the next bot run. - `status: ci failing` — **CI env, not caused by this PR.** `install-native (macos-latest)` and `wrap-native (macos-latest)` fail during the editable Rust/Python extension build with `ld: library 'clang_rt.osx' not found`, which is before this command path runs. @JerrettDavis confirmed this is not caused by the PR. All Linux jobs, all unit/integration/E2E jobs, lint, commitlint, template check, and Docker E2E jobs are green. `mergeable: MERGEABLE` is the actual gate. **CHANGELOG:** not updated — this is a single bug fix in an unreleased section, and the maintainers have not requested CHANGELOG entries for individual PRs in past PRs in this repo. Happy to add an entry under `## Unreleased` if requested.
2026-06-24 09:54:28 -05:00
def test_install_agent_ensure_no_spawn_when_lock_not_acquired(monkeypatch) -> None:
"""Ensure does not spawn a runtime when the start lock is contended."""
runner = CliRunner()
calls: list[str] = []
class Manifest:
profile = "default"
health_url = "http://127.0.0.1:8787/readyz"
monkeypatch.setattr("headroom.cli.install.load_manifest", lambda profile: Manifest())
monkeypatch.setattr("headroom.cli.install.probe_ready", lambda url: False)
import contextlib
@contextlib.contextmanager
def fake_lock(profile):
yield False
monkeypatch.setattr("headroom.cli.install.acquire_runtime_start_lock", fake_lock)
monkeypatch.setattr(
"headroom.cli.install.start_detached_agent",
lambda profile: calls.append("start_agent"),
)
monkeypatch.setattr(
"headroom.cli.install.start_persistent_docker",
lambda manifest: calls.append("start_docker"),
)
result = runner.invoke(main, ["install", "agent", "ensure"])
assert result.exit_code == 0, result.output
assert "already in progress" in result.output
assert calls == []
def test_install_agent_ensure_stops_wedged_runtime_before_restart(monkeypatch) -> None:
"""Ensure stops a wedged runtime (running but not ready) before starting fresh."""
runner = CliRunner()
calls: list[str] = []
class Manifest:
profile = "default"
health_url = "http://127.0.0.1:8787/readyz"
preset = "persistent-task"
supervisor_kind = "none"
monkeypatch.setattr("headroom.cli.install.load_manifest", lambda profile: Manifest())
monkeypatch.setattr("headroom.cli.install.probe_ready", lambda url: False)
monkeypatch.setattr("headroom.cli.install.runtime_status", lambda manifest: "running")
monkeypatch.setattr("headroom.cli.install.wait_ready", lambda manifest, timeout_seconds: False)
monkeypatch.setattr("headroom.cli.install.stop_runtime", lambda manifest: calls.append("stop"))
monkeypatch.setattr(
"headroom.cli.install.start_detached_agent",
lambda profile: calls.append("start_agent"),
)
monkeypatch.setattr(
"headroom.cli.install.start_persistent_docker",
lambda manifest: calls.append("start_docker"),
)
import contextlib
@contextlib.contextmanager
def fake_lock(profile):
yield True
monkeypatch.setattr("headroom.cli.install.acquire_runtime_start_lock", fake_lock)
monkeypatch.setattr(
"headroom.cli.install._start_deployment", lambda manifest: calls.append("start_deployment")
)
result = runner.invoke(main, ["install", "agent", "ensure"])
assert result.exit_code == 0, result.output
# stop must come before start_deployment — that's the bug guard.
assert calls.index("stop") < calls.index("start_deployment")
assert "start_agent" not in calls
assert "start_docker" not in calls
def test_install_agent_ensure_starts_when_stopped_and_lock_acquired(monkeypatch) -> None:
"""Ensure starts a runtime when none is running and lock is acquired."""
runner = CliRunner()
calls: list[str] = []
class Manifest:
profile = "default"
health_url = "http://127.0.0.1:8787/readyz"
preset = "persistent-task"
supervisor_kind = "none"
monkeypatch.setattr("headroom.cli.install.load_manifest", lambda profile: Manifest())
monkeypatch.setattr("headroom.cli.install.probe_ready", lambda url: False)
monkeypatch.setattr("headroom.cli.install.runtime_status", lambda manifest: "stopped")
monkeypatch.setattr(
"headroom.cli.install.start_detached_agent",
lambda profile: calls.append("start_agent"),
)
monkeypatch.setattr(
"headroom.cli.install.start_persistent_docker",
lambda manifest: calls.append("start_docker"),
)
import contextlib
@contextlib.contextmanager
def fake_lock(profile):
yield True
monkeypatch.setattr("headroom.cli.install.acquire_runtime_start_lock", fake_lock)
monkeypatch.setattr("headroom.cli.install.wait_ready", lambda manifest, timeout_seconds: True)
result = runner.invoke(main, ["install", "agent", "ensure"])
assert result.exit_code == 0, result.output
assert calls == ["start_agent"]
def test_install_agent_ensure_no_duplicate_spawn_after_lock_recheck(monkeypatch) -> None:
"""Ensure does not spawn if proxy becomes ready between initial probe and lock."""
runner = CliRunner()
calls: list[str] = []
class Manifest:
profile = "default"
health_url = "http://127.0.0.1:8787/readyz"
# First probe_ready (before lock) returns False, second (after lock) returns True
probe_results = iter([False, True])
monkeypatch.setattr("headroom.cli.install.load_manifest", lambda profile: Manifest())
monkeypatch.setattr("headroom.cli.install.probe_ready", lambda url: next(probe_results))
monkeypatch.setattr(
"headroom.cli.install.start_detached_agent",
lambda profile: calls.append("start_agent"),
)
import contextlib
@contextlib.contextmanager
def fake_lock(profile):
yield True
monkeypatch.setattr("headroom.cli.install.acquire_runtime_start_lock", fake_lock)
result = runner.invoke(main, ["install", "agent", "ensure"])
assert result.exit_code == 0, result.output
assert "already healthy" in result.output
assert calls == []
def test_install_agent_ensure_propagates_start_deployment_failure(monkeypatch) -> None:
"""Ensure must exit non-zero and surface the error when _start_deployment fails.
Regression for review feedback on PR #1301: the previous implementation wrapped
the guarded block in `except Exception` and returned normally, which made
a failed ensure indistinguishable from a successful one. Automation callers
need a non-zero exit code to detect that the deployment did not come up.
"""
runner = CliRunner()
class Manifest:
profile = "default"
health_url = "http://127.0.0.1:8787/readyz"
preset = "persistent-task"
supervisor_kind = "none"
monkeypatch.setattr("headroom.cli.install.load_manifest", lambda profile: Manifest())
monkeypatch.setattr("headroom.cli.install.probe_ready", lambda url: False)
monkeypatch.setattr("headroom.cli.install.runtime_status", lambda manifest: "stopped")
import contextlib
@contextlib.contextmanager
def fake_lock(profile):
yield True
monkeypatch.setattr("headroom.cli.install.acquire_runtime_start_lock", fake_lock)
def boom(manifest):
raise click.ClickException("simulated start failure")
monkeypatch.setattr("headroom.cli.install._start_deployment", boom)
result = runner.invoke(main, ["install", "agent", "ensure"])
assert result.exit_code != 0, f"expected non-zero exit, got {result.exit_code}: {result.output}"
assert "simulated start failure" in result.output