diff --git a/.claude/protected-paths.txt b/.claude/protected-paths.txt index cf74dd50..24470498 100644 --- a/.claude/protected-paths.txt +++ b/.claude/protected-paths.txt @@ -30,9 +30,18 @@ ship-spec eval health-check retro +firstmate # --- Orchestrator scripts (.oh/scripts/) --- .oh/scripts/ralph.sh +.oh/scripts/firstmate.sh +# The two `lib/` entries below are load-bearing shared infra reusable by ANY +# executor (the herdr -> tmux -> foreground runner ladder, and the slug + +# four-file task-folder contract firstmate.sh shares with ralph.sh's wording). +# Whole-line entries only — no inline comments, since consumers match with +# `grep -Fxq`. +.oh/scripts/lib/session-runner.sh +.oh/scripts/lib/task-contract.sh .oh/scripts/cron-runtime.ts .oh/scripts/sandbox-healthcheck.sh .oh/scripts/link-providers.sh diff --git a/.oh/evals/RESULTS.md b/.oh/evals/RESULTS.md index 946d1433..6bc47197 100644 --- a/.oh/evals/RESULTS.md +++ b/.oh/evals/RESULTS.md @@ -6,103 +6,107 @@ probe id; git history is the time series.** Schema and exit-code semantics are i | probe | tier | last-run (UTC) | status | source | |-------|------|----------------|--------|--------| -| ablate-state-machine | A | 2026-08-11 05:39 | PASS | issue #645 — one locked versioned ablation recovery owner | -| advisor-monitored-loop | A | 2026-08-11 05:39 | PASS | conversation 2026-06-19 (advisor-monitored ralph loop pattern, issue #257) | -| agent-browser-cli | A | 2026-08-11 05:39 | PASS | .oh/memory/MEMORY.md 2026-06-07 (agent-browser 0.8.5 CLI) | -| artifact-contract-audit | A | 2026-08-11 05:39 | PASS | issue #583/#645 — production /audit implementation Gate 1 behavior | -| audit-context-shared-memory | A | 2026-08-11 05:39 | PASS | issue #645 — shared-memory context ablation routing | -| audit-dispatcher-contract | A | 2026-08-11 05:39 | PASS | issue #645 — audit consolidation public taxonomy | -| audit-implementation-behavior | A | 2026-08-11 05:39 | PASS | issue #645 — implementation root/repo/browser behavior | -| audit-pr-acquire | A | 2026-08-11 05:39 | PASS | issue #645 — production PR acquisition behavior | -| audit-pr-classifier | A | 2026-08-11 05:39 | PASS | issue #645 — deterministic focused and queue PR classifier | -| audit-run-root-contract | A | 2026-08-11 05:39 | PASS | issue #645 — executable immutable audit root/run/log correlation | -| audit-shellcheck-coverage | A | 2026-08-11 05:39 | PASS | issue #645 — private audit scripts require release and CI lint coverage | -| audit-stale-references | A | 2026-08-11 05:39 | PASS | issue #645 — clean-breaking audit migration | -| autopilot-executor-toggle | A | 2026-08-11 05:39 | PASS | conversation 2026-06-13 (autopilot executor); 2026-06-27 (ralph-default flip) | -| autopilot-merged-pr-reference-dedupe | A | 2026-08-11 05:39 | PASS | issue #468 — autopilot must not rebuild open tickets whose development PRs already merged | -| autopilot-no-pr-session-close | A | 2026-08-11 05:39 | PASS | issue #209 (autopilot no-PR tmux session closure) 2026-06-16 | -| autopilot-open-pr-reference-dedupe | A | 2026-08-11 05:39 | PASS | issue #437 — autopilot must not start duplicate work when open PRs reference the same issue without linked-PR metadata | -| autopilot-pi-agent | A | 2026-08-11 05:39 | PASS | issue #116 (autopilot Pi tmux alignment) 2026-06-14; issue #118 (attachable Pi TUI tmux) 2026-06-14; issue #126 (kept Pi overlap lock release) 2026-06-14; issue #142 (worktree-by-default, skip→worktree) 2026-06-14 | -| autopilot-preflight-gate | A | 2026-08-11 05:39 | SKIPPED | issue #194 (deterministic autopilot caps preflight gate) 2026-06-15 | -| autopilot-upstream-default | A | 2026-08-11 05:39 | PASS | issue #420 — future autopilots must target canonical repo, not personal fork | -| autopilot-worktree-log-root | A | 2026-08-11 05:39 | PASS | issue #152 (persist autopilot worktree logs) 2026-06-15 | -| boot-lint-glob | A | 2026-08-11 05:39 | PASS | issue #90, issue #120 | -| builder-skill-consolidation | A | 2026-08-11 05:39 | PASS | issue #643 — consolidate artifact builders behind one /builder dispatcher | -| capability-benchmark-schema | A | 2026-08-11 05:39 | PASS | issue #167 — capability benchmark instrument | -| cc-safety-net-wiring | A | 2026-08-11 05:39 | REGRESSION | .oh/tasks/cc-safety-net/prd.json US-007 2026-07-19 | -| clean-restore | A | 2026-08-11 05:39 | PASS | issue #63 (autopilot-stray-wip-guard) 2026-06-12; issue #81 (owned-paths-zsh-split) 2026-06-13 | -| cleanup-tasks-scoped-guard | A | 2026-08-11 05:39 | PASS | issue #85 | -| cleanup-tasks-worktree-grooming | A | 2026-08-11 05:39 | PASS | issue #168; issue #327 | -| codex-stale-response-retry | A | 2026-08-11 05:39 | PASS | issue #506 — Codex previous_response_not_found RCA | -| cron-claude-codex-fallback | A | 2026-08-11 05:39 | PASS | conversation 2026-06-12 (default Codex fallback for crons) | -| cron-watchdog | A | 2026-08-11 05:39 | PASS | issues #130/#453 (cron runtime watchdog + legacy system-cron reaping) 2026-06-19 | -| curl-bash-safe-alternatives | A | 2026-08-11 05:39 | PASS | vet-run/vet integration — public curl|bash examples need review-first alternatives | -| datasets-schema | A | 2026-08-11 05:39 | PASS | issue #196 — .oh/evals/datasets verifiable trajectory corpus (Repo2RLEnv-inspired) | -| debugmcp-availability | A | 2026-08-11 05:39 | SKIPPED | issue #297 — DebugMCP MCP debug-server availability | -| delegate-model-effort-policy | A | 2026-08-11 05:39 | PASS | conversation 2026-07-11 (delegate model inheritance and thinking policy) | -| devtcp-hook | A | 2026-08-11 05:39 | PASS | .oh/memory/MEMORY.md 2026-06-10 (zsh /dev/tcp) | -| docker-inspect-env-guard | A | 2026-08-11 05:39 | PASS | operator directive 2026-08-08 (agents keep the docker socket, but must | -| docs-build-fast-path | A | 2026-08-11 05:39 | PASS | #455 — docs builds must stay out of fast harness/eval/release gates; #536 — docs site externalized to openharness-web; docs markdown relocated to .oh/docs/ | -| drift-check-cron-staleness-glob | A | 2026-08-11 05:39 | PASS | issue #98; issue #225 (restart-required cron frontmatter/config drift) | -| entrypoint-pnpm-manifest-fingerprint | A | 2026-08-11 05:39 | PASS | issue #521 (manifest-aware sandbox installs) 2026-07-01 | -| eval-ci-gate | A | 2026-08-11 05:39 | PASS | #103 — eval probe suite gated in CI | -| eval-gate | A | 2026-08-11 05:39 | PASS | .oh/memory/MEMORY.md 2026-06-11 (eval-gate) | -| eval-results-atomic | A | 2026-08-11 05:39 | PASS | issue #83 (eval-results-atomic-write) | -| eval-runner-exit | A | 2026-08-11 05:39 | PASS | .oh/memory/MEMORY.md 2026-06-11 (eval-runner-exit) #29 | -| execution-target-contract | A | 2026-08-11 05:39 | PASS | issue #733 (ExecutionTarget contract + Docker Compose adapter) 2026-08-10 | -| first-mate-charter | A | 2026-08-11 05:39 | PASS | .oh/tasks/first-mate-charter/ (issue #660) — First Mate role charter + advisor prompt pack must stay present, tracked, resolvable, and effort-vocabulary-aligned with /delegate | -| get-oh-bootstrap | A | 2026-08-11 05:39 | PASS | get-oh.sh bootstrap — the Node-bootstrapping host-side path to the standalone `oh` CLI (also on npm as @mifune/openharness; see oh-npm-package.sh) | -| git-skill | A | 2026-08-11 05:39 | PASS | conversation 2026-06-15 — rules are not always supported; git workflow must be the /git skill | -| harness-audit-empty-output-gate | A | 2026-08-11 05:39 | PASS | issue #246 — /audit harness must fail closed on empty auditor outputs | -| harness-audit-memory-path | A | 2026-08-11 05:39 | PASS | issue #183 — /audit harness must inspect the active worktree, not a hardcoded root | -| harness-audit-shared-memory | A | 2026-08-11 05:39 | PASS | issue #432 — /audit harness must load durable memory from shared log root in cron worktrees | -| harness-ci-core-paths | A | 2026-08-11 05:39 | PASS | #165 — core sandbox config files must trigger harness CI | -| harness-ci-hooks-paths | A | 2026-08-11 05:39 | PASS | issue #202 — credential/security hook changes must trigger harness CI | -| health-check-docker-stats | A | 2026-08-11 05:39 | PASS | .oh/memory/MEMORY.md 2026-06-10 (docker stats vs ps Size) | -| heartbeat-logging-contract | A | 2026-08-11 05:39 | PASS | issue #447 (heartbeat log append hardening) 2026-06-18 | -| locked-append-critical-path | A | 2026-08-11 05:39 | PASS | issue #204 (lock shared runtime log appends) 2026-06-15 | -| markitdown-wiki-ingest | A | 2026-08-11 05:39 | PASS | issue #649 — pinned local-document normalization contract for /wiki ingest | -| memory-gitignore-claim | A | 2026-08-11 05:39 | PASS | issue #101 | -| memory-log-locked-append | A | 2026-08-11 05:39 | PASS | issue #476 and #645 — memory appends are locked and audit has one log owner | -| next-dev-prod | A | 2026-08-11 05:39 | REGRESSION | .oh/memory/MEMORY.md 2026-06-04 | -| oh-devcontainer-restructure | A | 2026-08-11 05:39 | PASS | consolidate devcontainer — .oh/devcontainer/ folded back into .devcontainer/ | -| oh-image-only-deploy | A | 2026-08-11 05:39 | PASS | .oh/tasks/image-only-deploy/prd.json US-004 (issue #609, Flavor B image-only deploy) | -| oh-init-scaffold | A | 2026-08-11 05:39 | PASS | issue #531 Phase 2 | -| oh-npm-package | A | 2026-08-11 05:39 | PASS | npm publish path for the standalone `oh` CLI (@mifune/openharness) — alternative to get-oh.sh | -| oh-payload-manifest | A | 2026-08-11 05:39 | PASS | issue #531 follow-on (.oh payload manifest — oh update ships a declared allowlist) | -| oh-sandbox-image-mode | A | 2026-08-11 05:39 | PASS | conversation 2026-07-05 (basic Docker deployment — prebuilt-image mode) | -| oh-shipped-repo-overridable | A | 2026-08-11 05:39 | PASS | issue #531 follow-on (de-hardcode residual — shipped .oh shell scripts keep the upstream repo overridable) | -| oh-standalone-lifecycle | A | 2026-08-11 05:39 | PASS | issue #564 | -| oh-update | A | 2026-08-11 05:39 | PASS | issue #531 Phase 3 (oh update — upgrade only the .oh control plane) | -| operator-config-guard | A | 2026-08-11 05:39 | PASS | operator directives 2026-08-06 (.config/ and settings.local.json are operator-only) | -| owned-surface-guard | A | 2026-08-11 05:39 | PASS | issue #63 (autopilot-stray-wip-guard) 2026-06-12; issue #81 (owned-paths-zsh-split) 2026-06-13 | -| pnpm-audit-ci-gate | A | 2026-08-11 05:39 | PASS | issue #171 — pnpm security audits must run in CI | -| post-bridge-publish-confirmation | A | 2026-08-11 05:39 | PASS | #523 — post-bridge live publishing requires an explicit final confirmation gate | -| prd-output-path-contract | A | 2026-08-11 05:39 | PASS | .oh/memory/MEMORY.md 2026-06-19 | -| project-root-seam | A | 2026-08-11 05:39 | PASS | issue #531 Phase 1 (OH_PROJECT_ROOT project-root seam) 2026-06-26 | -| prompt-miner-log-root-worktree | A | 2026-08-11 05:39 | PASS | .oh/skills/prompt-miner/scripts/render-log-entry.sh | -| prompt-miner-schema-compat | A | 2026-08-11 05:39 | PASS | issue #253 — prompt-miner JSONL schema-drift guard | -| prompt-miner-symlink-entrypoint | A | 2026-08-11 05:39 | PASS | issue #663 — prompt-miner engine no-ops via the documented .claude/skills symlink | -| prompt-miner-weakness-record | A | 2026-08-11 05:39 | PASS | issue #580 — prompt-miner weakness-record (WH-xxx) cluster output | -| ralph-fallback-order | A | 2026-08-11 05:39 | PASS | conversation 2026-06-12 (Ralph default fallback order) | -| repo-map-contract | A | 2026-08-11 05:39 | PASS | issue #464 — repo map must optimize orientation without adding a tree dependency or unmeasured performance claims | -| retro-deterministic-contract | A | 2026-08-11 05:39 | PASS | issue #443 — /retro deterministic output and self-contained helper contract | -| rl-delegation-write-worker | A | 2026-08-11 05:39 | PASS | .oh/memory/MEMORY.md 2026-06-10 (rl-delegation) #57 | -| rlm-context-budget | A | 2026-08-11 05:39 | PASS | .oh/tasks/rlm-weighted-trajectories/prd.json US-006 | -| sandbox-boot-guard-ci | A | 2026-08-11 05:39 | PASS | issue #449 (sandbox image build CI guard) 2026-06-19 | -| ship-spec-ready-finalization | A | 2026-08-11 05:39 | PASS | issue #134 — /ship-spec must finalize ready PRs after gates, not stop at draft scaffold | -| skill-paths | A | 2026-08-11 05:39 | PASS | issue #43 — stale path references; extended by issue #69 — apps/->packages/ rename guard | -| skills-dir-clean | A | 2026-08-11 05:39 | PASS | conversation 2026-06-29 — Pi parses every top-level `.md` in the skills | -| skills-vendored | A | 2026-08-11 05:39 | PASS | absorb .mifune submodule into .oh — the skills/agents/hooks pack is vendored | -| slack-admin-command-surface | A | 2026-08-11 05:39 | PASS | issue #354 — Slack bridge docs must distinguish Pi /msg-bridge commands from Slack DM admin text handlers | -| spec-family-contract | A | 2026-08-11 05:39 | PASS | conversation 2026-06-19 (spec-* family split, issue #265); consolidated into /spec dispatcher 2026-06-23 (one skill, args) | -| submitted-by-trailers | A | 2026-08-11 05:39 | PASS | conversation 2026-06-12 (commit attribution trailers) | -| sync-skill-contract | A | 2026-08-11 05:39 | PASS | issue #331 — /sync dispatcher skill (bidirectional origin↔upstream sync) | -| watchdog-completed-session-reap | A | 2026-08-11 05:39 | PASS | issue #235 (completed autopilot PR session reaping) | -| watchdog-draft-prs | A | 2026-08-11 05:39 | PASS | conversation 2026-06-15 (generic watchdog + stale draft PR recovery) | -| watchdog-stuck-sessions | A | 2026-08-11 05:39 | PASS | issue #240 (Codex zero-credit stuck autopilot sessions) 2026-06-17 | -| weigh-scorer-contract | A | 2026-08-11 05:39 | PASS | .oh/tasks/rlm-weighted-trajectories/prd.json US-003 (2026-06-27) | -| wiki-readme-index | A | 2026-08-11 05:39 | PASS | issue #132 — wiki README index drift guard | -| workflow-boundaries | A | 2026-08-11 05:39 | PASS | conversation 2026-06-19 (workflow consolidation, issue #259) | +| ablate-state-machine | A | 2026-08-13 03:03 | PASS | issue #645 — one locked versioned ablation recovery owner | +| advisor-monitored-loop | A | 2026-08-13 03:03 | PASS | conversation 2026-06-19 (advisor-monitored ralph loop pattern, issue #257) | +| agent-browser-cli | A | 2026-08-13 03:03 | PASS | .oh/memory/MEMORY.md 2026-06-07 (agent-browser 0.8.5 CLI) | +| artifact-contract-audit | A | 2026-08-13 03:03 | PASS | issue #583/#645 — production /audit implementation Gate 1 behavior | +| audit-context-shared-memory | A | 2026-08-13 03:03 | PASS | issue #645 — shared-memory context ablation routing | +| audit-dispatcher-contract | A | 2026-08-13 03:03 | PASS | issue #645 — audit consolidation public taxonomy | +| audit-implementation-behavior | A | 2026-08-13 03:03 | PASS | issue #645 — implementation root/repo/browser behavior | +| audit-pr-acquire | A | 2026-08-13 03:03 | PASS | issue #645 — production PR acquisition behavior | +| audit-pr-classifier | A | 2026-08-13 03:03 | PASS | issue #645 — deterministic focused and queue PR classifier | +| audit-run-root-contract | A | 2026-08-13 03:03 | PASS | issue #645 — executable immutable audit root/run/log correlation | +| audit-shellcheck-coverage | A | 2026-08-13 03:03 | PASS | issue #645 — private audit scripts require release and CI lint coverage | +| audit-stale-references | A | 2026-08-13 03:03 | PASS | issue #645 — clean-breaking audit migration | +| autopilot-executor-toggle | A | 2026-08-13 03:03 | PASS | conversation 2026-06-13 (autopilot executor); 2026-06-27 (ralph-default flip) | +| autopilot-merged-pr-reference-dedupe | A | 2026-08-13 03:03 | PASS | issue #468 — autopilot must not rebuild open tickets whose development PRs already merged | +| autopilot-no-pr-session-close | A | 2026-08-13 03:03 | PASS | issue #209 (autopilot no-PR tmux session closure) 2026-06-16 | +| autopilot-open-pr-reference-dedupe | A | 2026-08-13 03:03 | PASS | issue #437 — autopilot must not start duplicate work when open PRs reference the same issue without linked-PR metadata | +| autopilot-pi-agent | A | 2026-08-13 03:03 | PASS | issue #116 (autopilot Pi tmux alignment) 2026-06-14; issue #118 (attachable Pi TUI tmux) 2026-06-14; issue #126 (kept Pi overlap lock release) 2026-06-14; issue #142 (worktree-by-default, skip→worktree) 2026-06-14 | +| autopilot-preflight-gate | A | 2026-08-13 03:03 | SKIPPED | issue #194 (deterministic autopilot caps preflight gate) 2026-06-15 | +| autopilot-upstream-default | A | 2026-08-13 03:03 | PASS | issue #420 — future autopilots must target canonical repo, not personal fork | +| autopilot-worktree-log-root | A | 2026-08-13 03:03 | PASS | issue #152 (persist autopilot worktree logs) 2026-06-15 | +| boot-lint-glob | A | 2026-08-13 03:03 | PASS | issue #90, issue #120 | +| builder-skill-consolidation | A | 2026-08-13 03:03 | PASS | issue #643 — consolidate artifact builders behind one /builder dispatcher | +| capability-benchmark-schema | A | 2026-08-13 03:03 | PASS | issue #167 — capability benchmark instrument | +| cc-safety-net-wiring | A | 2026-08-13 03:03 | PASS | .oh/tasks/cc-safety-net/prd.json US-007 2026-07-19 | +| clean-restore | A | 2026-08-13 03:03 | PASS | issue #63 (autopilot-stray-wip-guard) 2026-06-12; issue #81 (owned-paths-zsh-split) 2026-06-13 | +| cleanup-tasks-scoped-guard | A | 2026-08-13 03:03 | PASS | issue #85 | +| cleanup-tasks-worktree-grooming | A | 2026-08-13 03:03 | PASS | issue #168; issue #327 | +| codex-stale-response-retry | A | 2026-08-13 03:03 | PASS | issue #506 — Codex previous_response_not_found RCA | +| cron-claude-codex-fallback | A | 2026-08-13 03:03 | PASS | conversation 2026-06-12 (default Codex fallback for crons) | +| cron-watchdog | A | 2026-08-13 03:03 | PASS | issues #130/#453 (cron runtime watchdog + legacy system-cron reaping) 2026-06-19 | +| curl-bash-safe-alternatives | A | 2026-08-13 03:03 | PASS | vet-run/vet integration — public curl|bash examples need review-first alternatives | +| datasets-schema | A | 2026-08-13 03:03 | PASS | issue #196 — .oh/evals/datasets verifiable trajectory corpus (Repo2RLEnv-inspired) | +| debugmcp-availability | A | 2026-08-13 03:03 | SKIPPED | issue #297 — DebugMCP MCP debug-server availability | +| delegate-model-effort-policy | A | 2026-08-13 03:03 | PASS | conversation 2026-07-11 (delegate model inheritance and thinking policy) | +| devtcp-hook | A | 2026-08-13 03:03 | PASS | .oh/memory/MEMORY.md 2026-06-10 (zsh /dev/tcp) | +| docker-inspect-env-guard | A | 2026-08-13 03:03 | PASS | operator directive 2026-08-08 (agents keep the docker socket, but must | +| docs-build-fast-path | A | 2026-08-13 03:03 | PASS | #455 — docs builds must stay out of fast harness/eval/release gates; #536 — docs site externalized to openharness-web; docs markdown relocated to .oh/docs/ | +| drift-check-cron-staleness-glob | A | 2026-08-13 03:03 | PASS | issue #98; issue #225 (restart-required cron frontmatter/config drift) | +| entrypoint-pnpm-manifest-fingerprint | A | 2026-08-13 03:03 | PASS | issue #521 (manifest-aware sandbox installs) 2026-07-01 | +| eval-ci-gate | A | 2026-08-13 03:03 | PASS | #103 — eval probe suite gated in CI | +| eval-gate | A | 2026-08-13 03:03 | PASS | .oh/memory/MEMORY.md 2026-06-11 (eval-gate) | +| eval-results-atomic | A | 2026-08-13 03:03 | PASS | issue #83 (eval-results-atomic-write) | +| eval-runner-exit | A | 2026-08-13 03:03 | PASS | .oh/memory/MEMORY.md 2026-06-11 (eval-runner-exit) #29 | +| execution-target-contract | A | 2026-08-13 03:03 | PASS | issue #733 (ExecutionTarget contract + Docker Compose adapter) 2026-08-10 | +| first-mate-charter | A | 2026-08-13 03:03 | PASS | .oh/tasks/first-mate-charter/ (issue #660) — First Mate role charter + advisor prompt pack must stay present, tracked, resolvable, and effort-vocabulary-aligned with /delegate | +| firstmate-executor-contract | A | 2026-08-13 03:03 | PASS | .oh/tasks/firstmate-executor/ (issue #746) — the opt-in firstmate build executor is additive to ralph and shares its terminal interface | +| get-oh-bootstrap | A | 2026-08-13 03:03 | PASS | get-oh.sh bootstrap — the Node-bootstrapping host-side path to the standalone `oh` CLI (also on npm as @mifune/openharness; see oh-npm-package.sh) | +| git-skill | A | 2026-08-13 03:03 | PASS | conversation 2026-06-15 — rules are not always supported; git workflow must be the /git skill | +| harness-audit-empty-output-gate | A | 2026-08-13 03:03 | PASS | issue #246 — /audit harness must fail closed on empty auditor outputs | +| harness-audit-memory-path | A | 2026-08-13 03:03 | PASS | issue #183 — /audit harness must inspect the active worktree, not a hardcoded root | +| harness-audit-shared-memory | A | 2026-08-13 03:03 | PASS | issue #432 — /audit harness must load durable memory from shared log root in cron worktrees | +| harness-ci-core-paths | A | 2026-08-13 03:03 | PASS | #165 — core sandbox config files must trigger harness CI | +| harness-ci-hooks-paths | A | 2026-08-13 03:03 | PASS | issue #202 — credential/security hook changes must trigger harness CI | +| health-check-docker-stats | A | 2026-08-13 03:03 | PASS | .oh/memory/MEMORY.md 2026-06-10 (docker stats vs ps Size) | +| heartbeat-logging-contract | A | 2026-08-13 03:03 | PASS | issue #447 (heartbeat log append hardening) 2026-06-18 | +| locked-append-critical-path | A | 2026-08-13 03:03 | PASS | issue #204 (lock shared runtime log appends) 2026-06-15 | +| markitdown-wiki-ingest | A | 2026-08-13 03:03 | PASS | issue #649 — pinned local-document normalization contract for /wiki ingest | +| memory-gitignore-claim | A | 2026-08-13 03:03 | PASS | issue #101 | +| memory-log-locked-append | A | 2026-08-13 03:03 | PASS | issue #476 and #645 — memory appends are locked and audit has one log owner | +| next-dev-prod | A | 2026-08-13 03:03 | SKIPPED | .oh/memory/MEMORY.md 2026-06-04 | +| oh-devcontainer-restructure | A | 2026-08-13 03:03 | PASS | consolidate devcontainer — .oh/devcontainer/ folded back into .devcontainer/ | +| oh-image-only-deploy | A | 2026-08-13 03:03 | PASS | .oh/tasks/image-only-deploy/prd.json US-004 (issue #609, Flavor B image-only deploy) | +| oh-init-scaffold | A | 2026-08-13 03:03 | PASS | issue #531 Phase 2 | +| oh-npm-package | A | 2026-08-13 03:03 | PASS | npm publish path for the standalone `oh` CLI (@mifune/openharness) — alternative to get-oh.sh | +| oh-payload-manifest | A | 2026-08-13 03:03 | PASS | issue #531 follow-on (.oh payload manifest — oh update ships a declared allowlist) | +| oh-sandbox-image-mode | A | 2026-08-13 03:03 | PASS | conversation 2026-07-05 (basic Docker deployment — prebuilt-image mode) | +| oh-shipped-repo-overridable | A | 2026-08-13 03:03 | PASS | issue #531 follow-on (de-hardcode residual — shipped .oh shell scripts keep the upstream repo overridable) | +| oh-standalone-lifecycle | A | 2026-08-13 03:03 | PASS | issue #564 | +| oh-update | A | 2026-08-13 03:03 | PASS | issue #531 Phase 3 (oh update — upgrade only the .oh control plane) | +| operator-config-guard | A | 2026-08-13 03:03 | PASS | operator directives 2026-08-06 (.config/ and settings.local.json are operator-only) | +| owned-surface-guard | A | 2026-08-13 03:03 | PASS | issue #63 (autopilot-stray-wip-guard) 2026-06-12; issue #81 (owned-paths-zsh-split) 2026-06-13 | +| pnpm-audit-ci-gate | A | 2026-08-13 03:03 | PASS | issue #171 — pnpm security audits must run in CI | +| post-bridge-publish-confirmation | A | 2026-08-13 03:03 | PASS | #523 — post-bridge live publishing requires an explicit final confirmation gate | +| prd-output-path-contract | A | 2026-08-13 03:03 | PASS | .oh/memory/MEMORY.md 2026-06-19 | +| project-root-seam | A | 2026-08-13 03:03 | PASS | issue #531 Phase 1 (OH_PROJECT_ROOT project-root seam) 2026-06-26 | +| prompt-miner-log-root-worktree | A | 2026-08-13 03:03 | PASS | .oh/skills/prompt-miner/scripts/render-log-entry.sh | +| prompt-miner-schema-compat | A | 2026-08-13 03:03 | PASS | issue #253 — prompt-miner JSONL schema-drift guard | +| prompt-miner-symlink-entrypoint | A | 2026-08-13 03:03 | PASS | issue #663 — prompt-miner engine no-ops via the documented .claude/skills symlink | +| prompt-miner-weakness-record | A | 2026-08-13 03:03 | PASS | issue #580 — prompt-miner weakness-record (WH-xxx) cluster output | +| protected-paths-resolve | A | 2026-08-13 03:03 | PASS | issue #753 — .claude/protected-paths.txt named 7 paths that did not exist. | +| ralph-fallback-order | A | 2026-08-13 03:03 | PASS | conversation 2026-06-12 (Ralph default fallback order) | +| repo-map-contract | A | 2026-08-13 03:03 | PASS | issue #464 — repo map must optimize orientation without adding a tree dependency or unmeasured performance claims | +| retro-deterministic-contract | A | 2026-08-13 03:03 | PASS | issue #443 — /retro deterministic output and self-contained helper contract | +| rl-delegation-write-worker | A | 2026-08-13 03:03 | PASS | .oh/memory/MEMORY.md 2026-06-10 (rl-delegation) #57 | +| rlm-context-budget | A | 2026-08-13 03:03 | PASS | .oh/tasks/rlm-weighted-trajectories/prd.json US-006 | +| sandbox-boot-guard-ci | A | 2026-08-13 03:03 | PASS | issue #449 (sandbox image build CI guard) 2026-06-19 | +| session-runner-ladder | A | 2026-08-13 03:03 | PASS | .oh/tasks/firstmate-executor/ (issue #746) — the shared herdr -> tmux -> foreground runner ladder and its safety gates | +| ship-spec-ready-finalization | A | 2026-08-13 03:03 | PASS | issue #134 — /ship-spec must finalize ready PRs after gates, not stop at draft scaffold | +| skill-paths | A | 2026-08-13 03:03 | PASS | issue #43 — stale path references; extended by issue #69 — apps/->packages/ rename guard | +| skills-dir-clean | A | 2026-08-13 03:03 | PASS | conversation 2026-06-29 — Pi parses every top-level `.md` in the skills | +| skills-vendored | A | 2026-08-13 03:03 | PASS | absorb .mifune submodule into .oh — the skills/agents/hooks pack is vendored | +| slack-admin-command-surface | A | 2026-08-13 03:03 | PASS | issue #354 — Slack bridge docs must distinguish Pi /msg-bridge commands from Slack DM admin text handlers | +| spec-family-contract | A | 2026-08-13 03:03 | PASS | conversation 2026-06-19 (spec-* family split, issue #265); consolidated into /spec dispatcher 2026-06-23 (one skill, args) | +| ste-checker-contract | A | 2026-08-13 03:03 | PASS | issue #750 PR audit — the /ste checker had four fail-open paths (unclosed | +| submitted-by-trailers | A | 2026-08-13 03:03 | PASS | conversation 2026-06-12 (commit attribution trailers) | +| sync-skill-contract | A | 2026-08-13 03:03 | PASS | issue #331 — /sync dispatcher skill (bidirectional origin↔upstream sync) | +| watchdog-completed-session-reap | A | 2026-08-13 03:03 | PASS | issue #235 (completed autopilot PR session reaping) | +| watchdog-draft-prs | A | 2026-08-13 03:03 | PASS | conversation 2026-06-15 (generic watchdog + stale draft PR recovery) | +| watchdog-stuck-sessions | A | 2026-08-13 03:03 | PASS | issue #240 (Codex zero-credit stuck autopilot sessions) 2026-06-17 | +| weigh-scorer-contract | A | 2026-08-13 03:03 | PASS | .oh/tasks/rlm-weighted-trajectories/prd.json US-003 (2026-06-27) | +| wiki-readme-index | A | 2026-08-13 03:03 | PASS | issue #132 — wiki README index drift guard | +| workflow-boundaries | A | 2026-08-13 03:03 | PASS | conversation 2026-06-19 (workflow consolidation, issue #259) | diff --git a/.oh/evals/probes/autopilot-executor-toggle.sh b/.oh/evals/probes/autopilot-executor-toggle.sh index 2a714867..4ee3e68d 100755 --- a/.oh/evals/probes/autopilot-executor-toggle.sh +++ b/.oh/evals/probes/autopilot-executor-toggle.sh @@ -28,12 +28,15 @@ fi missing=() # Executor toggle: default ship-spec (Advisor-monitored ralph build), explicit -# delegate-advisor + inline ralph opt-ins. -grep -F 'argument-hint:' "$SKILL" | grep -Fq '[--executor=ship-spec|delegate-advisor|ralph]' || missing+=("argument hint includes executor toggle") +# delegate-advisor + inline ralph + firstmate opt-ins. +grep -F 'argument-hint:' "$SKILL" | grep -Fq '[--executor=ship-spec|delegate-advisor|ralph|firstmate]' || missing+=("argument hint includes executor toggle") grep -Fq 'EXECUTOR="${AUTOPILOT_EXECUTOR:-ship-spec}"' "$SKILL" || missing+=("AUTOPILOT_EXECUTOR default ship-spec") grep -Fq '*--executor=ship-spec*) EXECUTOR=ship-spec' "$SKILL" || missing+=("CLI --executor=ship-spec toggle") grep -Fq '*--executor=delegate-advisor*) EXECUTOR=delegate-advisor' "$SKILL" || missing+=("CLI --executor=delegate-advisor toggle") grep -Fq '*--executor=ralph*) EXECUTOR=ralph' "$SKILL" || missing+=("CLI --executor=ralph toggle") +grep -Fq '*--executor=firstmate*) EXECUTOR=firstmate' "$SKILL" || missing+=("CLI --executor=firstmate toggle") +# The two-part edit: the case arm alone is inert while the validation list rejects firstmate. +grep -Fq 'case "$EXECUTOR" in ship-spec|delegate-advisor|ralph|firstmate)' "$SKILL" || missing+=("firstmate accepted by the AUTOPILOT_EXECUTOR validation list") grep -Fq '.oh/scripts/ralph.sh "$SLUG"' "$SKILL" || missing+=("Ralph inline fallback still launches .oh/scripts/ralph.sh") grep -Fq '#### `ralph` fallback' "$SKILL" || missing+=("Ralph inline fallback section") @@ -111,5 +114,5 @@ if (( ${#missing[@]} )); then exit 1 fi -echo "PASS: autopilot defaults to ship-spec (Advisor-monitored ralph; /delegate optional inside), exact goal, defers the build to /ship-spec, delegate-advisor + inline ralph opt-ins, /ship-spec default executor ralph, safe tmux naming, dedupe guard, active-marker cleanup, dry-run guard" >&2 +echo "PASS: autopilot defaults to ship-spec (Advisor-monitored ralph; /delegate optional inside), exact goal, defers the build to /ship-spec, delegate-advisor + inline ralph + firstmate opt-ins, /ship-spec default executor ralph, safe tmux naming, dedupe guard, active-marker cleanup, dry-run guard" >&2 exit 0 diff --git a/.oh/evals/probes/firstmate-executor-contract.sh b/.oh/evals/probes/firstmate-executor-contract.sh new file mode 100755 index 00000000..1d099628 --- /dev/null +++ b/.oh/evals/probes/firstmate-executor-contract.sh @@ -0,0 +1,174 @@ +#!/usr/bin/env bash +# tier: A +# source: .oh/tasks/firstmate-executor/ (issue #746) — the opt-in firstmate build executor is additive to ralph and shares its terminal interface +# desc: .oh/scripts/firstmate.sh exists and is executable; the whole-line `STATUS: COMPLETE` +# sentinel and the `| tee` launch pipe survive on the executor surface; BOTH executor +# toggles (SHIP_SPEC_EXECUTOR and AUTOPILOT_EXECUTOR) carry a firstmate arm AND a ralph +# arm with ralph still the ship-spec default; the session-prompt template's ordered +# anchor keywords keep the advisor prompt pack's relative step order while .oh/prompts/ +# stays zero-diff; CLAUDE.md is still a symlink to AGENTS.md; and .oh/scripts/ralph.sh +# still exists — the default executor is retained indefinitely, never replaced. +# +# The single-quoted grep patterns below are pinned LITERALS — the dollar signs are part of +# the text being searched for, not expansions this file wants performed. +# shellcheck disable=SC2016 +set -u + +ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/../../.." && pwd)" +FIRSTMATE="$ROOT/.oh/scripts/firstmate.sh" +RUNNER="$ROOT/.oh/scripts/lib/session-runner.sh" +RALPH="$ROOT/.oh/scripts/ralph.sh" +TEMPLATE="$ROOT/.oh/skills/firstmate/templates/session-prompt.md" +SHIP="$ROOT/.claude/skills/ship-spec/SKILL.md" +AUTOPILOT="$ROOT/.claude/skills/autopilot/SKILL.md" + +# No SKIPPED path: this probe ships in the same commit as the executor it pins, so a missing +# artifact is a REGRESSION rather than a not-applicable run. That is the point of assertion (7). +missing=() + +# --- (1) the executor entrypoint exists and is executable ------------------- +if [ ! -f "$FIRSTMATE" ]; then + missing+=(".oh/scripts/firstmate.sh absent (the firstmate executor entrypoint is gone)") +elif [ ! -x "$FIRSTMATE" ]; then + missing+=(".oh/scripts/firstmate.sh is not executable") +fi + +# --- (2) the invariant terminal interface + the logging pipe ---------------- +# All three executors terminate on the same sentinel: the WHOLE LINE +# `STATUS: COMPLETE` in .oh/tasks//progress.txt. +if [ -f "$FIRSTMATE" ]; then + grep -Fq 'STATUS: COMPLETE' "$FIRSTMATE" \ + || missing+=("firstmate.sh does not name the STATUS: COMPLETE sentinel") +fi +if [ -f "$RUNNER" ]; then + grep -Fq '^STATUS: COMPLETE$' "$RUNNER" \ + || missing+=("session-runner.sh: the whole-line sentinel match ^STATUS: COMPLETE\$ is gone (a substring match would fire on prose)") +fi + +# The `| tee` pipe lives in the shared session-runner library that firstmate.sh sources — +# every launch branch pipes 2>&1 into the per-slug log. Accept it anywhere on the executor +# surface (entrypoint OR library) so a future refactor that moves the launch is not a +# false regression, but require it on at least one of the two. +# +# Full-line comments are excluded: both files DOCUMENT the pipe in their headers, so a +# whole-file grep would stay green after the actual pipe was deleted. +tee_found=0 +for f in "$FIRSTMATE" "$RUNNER"; do + [ -f "$f" ] || continue + grep -v '^[[:space:]]*#' "$f" | grep -Fq '| tee' && tee_found=1 +done +[ "$tee_found" -eq 1 ] \ + || missing+=("the | tee launch pipe is absent from both firstmate.sh and lib/session-runner.sh (sessions would run unlogged)") + +# The entrypoint must actually reach the shared ladder, or the two assertions above pin +# files that no longer form one surface. +if [ -f "$FIRSTMATE" ]; then + grep -Fq 'lib/session-runner.sh' "$FIRSTMATE" \ + || missing+=("firstmate.sh no longer sources .oh/scripts/lib/session-runner.sh (the executor and the ladder have diverged)") +fi + +# --- (3) BOTH executor toggles carry a firstmate arm AND a ralph arm -------- +if [ ! -f "$SHIP" ]; then + missing+=(".claude/skills/ship-spec/SKILL.md absent — cannot verify the SHIP_SPEC_EXECUTOR toggle") +else + grep -Fq '*--executor=firstmate*) SHIP_SPEC_EXECUTOR=firstmate' "$SHIP" \ + || missing+=("ship-spec: no *--executor=firstmate*) SHIP_SPEC_EXECUTOR=firstmate case arm") + grep -Fq '*--executor=ralph*) SHIP_SPEC_EXECUTOR=ralph' "$SHIP" \ + || missing+=("ship-spec: the *--executor=ralph*) SHIP_SPEC_EXECUTOR=ralph case arm is gone") + # ralph stays the DEFAULT — the firstmate arm is opt-in, never a flip. + grep -Fxq 'SHIP_SPEC_EXECUTOR="${SHIP_SPEC_EXECUTOR:-ralph}"' "$SHIP" \ + || missing+=("ship-spec: the ralph default line SHIP_SPEC_EXECUTOR=\${SHIP_SPEC_EXECUTOR:-ralph} is not byte-identical") + ship_case="$(grep -F 'case "$SHIP_SPEC_EXECUTOR" in' "$SHIP" | head -1)" + for arm in ralph firstmate; do + printf '%s\n' "$ship_case" | grep -Fq "$arm" \ + || missing+=("ship-spec: the SHIP_SPEC_EXECUTOR validation list does not accept '$arm'") + done +fi + +if [ ! -f "$AUTOPILOT" ]; then + missing+=(".claude/skills/autopilot/SKILL.md absent — cannot verify the AUTOPILOT_EXECUTOR toggle") +else + grep -Fq '*--executor=firstmate*) EXECUTOR=firstmate' "$AUTOPILOT" \ + || missing+=("autopilot: no *--executor=firstmate*) EXECUTOR=firstmate case arm") + grep -Fq '*--executor=ralph*) EXECUTOR=ralph' "$AUTOPILOT" \ + || missing+=("autopilot: the *--executor=ralph*) EXECUTOR=ralph case arm is gone") + ap_case="$(grep -F 'case "$EXECUTOR" in' "$AUTOPILOT" | head -1)" + for arm in ralph firstmate; do + printf '%s\n' "$ap_case" | grep -Fq "$arm" \ + || missing+=("autopilot: the AUTOPILOT_EXECUTOR validation list does not accept '$arm' (EXECUTOR=$arm would fail hard)") + done +fi + +# --- (4) step-order equivalence, asserted MECHANICALLY ---------------------- +# The ordered anchor-keyword list below was derived at authoring time from +# .oh/prompts/advisor/implement.yml and .oh/prompts/advisor/pr.yml and is recorded verbatim +# in the session-prompt template's own contract header (US-002). "Equivalence" means exactly +# this: these anchors appear in the template BODY in the same relative order, compared by +# first occurrence. Nothing fuzzy, no markdown-vs-YAML similarity judgement. +# +# ORDERING SCOPE IS BODY-ONLY. The header records the list in the asserted order, so +# including it would make this check vacuous; the template ends its header with the literal +# `END CONTRACT HEADER -->` marker for exactly this reason. +ANCHORS=( + 'dependency graph' + '/compact' + 'acceptanceCriteria' + 'passes: true' + '/audit implementation' + 'evidence.md' + '/retro' + 'Ready PR' +) +if [ ! -f "$TEMPLATE" ]; then + missing+=(".oh/skills/firstmate/templates/session-prompt.md absent (the session prompt the executor renders is gone)") +else + body="$(awk '/END CONTRACT HEADER -->/{f=1; next} f' "$TEMPLATE")" + if [ -z "$body" ]; then + missing+=("session-prompt.md: no 'END CONTRACT HEADER -->' marker — the body/header ordering scope cannot be resolved") + else + prev_off=-1 + prev_anchor="" + for a in "${ANCHORS[@]}"; do + off="$(printf '%s\n' "$body" | grep -Fbo -m1 -e "$a" | head -1 | cut -d: -f1)" + if [ -z "$off" ]; then + missing+=("session-prompt.md body: step-order anchor '$a' absent") + continue + fi + if [ "$off" -le "$prev_off" ]; then + missing+=("session-prompt.md body: step-order anchor '$a' precedes '$prev_anchor' (advisor pack order broken)") + fi + prev_off="$off" + prev_anchor="$a" + done + fi +fi + +# --- (5) .oh/prompts/ is untouched — the template is a derivative, not an edit +if git -C "$ROOT" rev-parse --git-dir >/dev/null 2>&1; then + git -C "$ROOT" diff --quiet -- .oh/prompts/ \ + || missing+=(".oh/prompts/ has uncommitted changes — the session-prompt template must derive from the advisor pack, never edit it") +else + missing+=("not a git repository at $ROOT — cannot verify .oh/prompts/ is zero-diff") +fi + +# --- (6) CLAUDE.md is still a symlink to AGENTS.md ------------------------- +# A severed alias (an independently written CLAUDE.md) must surface as a REGRESSION rather +# than silently passing a diff that happens to be empty at write time. +link="$(readlink "$ROOT/CLAUDE.md" 2>/dev/null)" || link="" +[ "$link" = "AGENTS.md" ] \ + || missing+=("CLAUDE.md is not a symlink to AGENTS.md (readlink printed '${link:-}')") + +# --- (7) ralph.sh still exists — the regression tripwire ------------------- +# firstmate is an ADDITIVE third executor. If ralph.sh is ever deleted this probe must go +# red rather than quietly reporting a green firstmate contract over a broken default. +[ -f "$RALPH" ] \ + || missing+=(".oh/scripts/ralph.sh absent — ralph is the DEFAULT executor and is retained indefinitely") + +if [ "${#missing[@]}" -gt 0 ]; then + printf 'REGRESSION: firstmate executor contract broken:\n' >&2 + printf ' - %s\n' "${missing[@]}" >&2 + exit 1 +fi + +echo "PASS: firstmate.sh executable + sentinel/tee intact, both executor toggles carry firstmate+ralph arms, session-prompt anchors in advisor-pack order, .oh/prompts/ zero-diff, CLAUDE.md->AGENTS.md, ralph.sh retained" >&2 +exit 0 diff --git a/.oh/evals/probes/session-runner-ladder.sh b/.oh/evals/probes/session-runner-ladder.sh new file mode 100755 index 00000000..33269943 --- /dev/null +++ b/.oh/evals/probes/session-runner-ladder.sh @@ -0,0 +1,241 @@ +#!/usr/bin/env bash +# tier: A +# source: .oh/tasks/firstmate-executor/ (issue #746) — the shared herdr -> tmux -> foreground runner ladder and its safety gates +# desc: .oh/scripts/lib/session-runner.sh resolves the ladder herdr -> tmux -> foreground; herdr +# health is pinned to the two literal fields `status: running` and `compatible: yes`; +# runner_detect carries the nesting guard BEFORE any probe pane and the execution-context +# fingerprint gate whose mismatch degrades to tmux with the reason logged; resolve_timeout_ms +# is the single session-budget source with the 14400000 default and bounds the tmux/foreground +# poll loop; every exit path runs runner_teardown, removes the per-slug lock and appends +# FIRSTMATE-INCOMPLETE; every herdr launch passes --no-focus; teardown is `pane close` (0.7.4 +# has no agent stop/kill verb); the sourceable library sets no file-scope shell options; and +# the herdr commands that would disturb a shared server stay absent while `herdr agent get` +# (the liveness oracle) stays present. +# +# The single-quoted grep patterns below are pinned LITERALS — the dollar signs are part of +# the text being searched for, not expansions this file wants performed. +# shellcheck disable=SC2016 +set -u + +ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/../../.." && pwd)" +RUNNER="$ROOT/.oh/scripts/lib/session-runner.sh" + +# No SKIPPED path: this probe ships in the same commit as the library it pins, so an absent +# library is a REGRESSION, not a not-applicable run. +missing=() + +if [ ! -f "$RUNNER" ]; then + printf 'REGRESSION: .oh/scripts/lib/session-runner.sh absent — the shared runner ladder is gone\n' >&2 + exit 1 +fi + +# Extracts one top-level function body: from `() {` at column 0 through the closing +# brace at column 0. Every assertion that cares about WHERE a literal lives uses this rather +# than grepping the whole file, so a matching line in the header block cannot satisfy it. +fn_body() { # + awk -v pat="$1() {" 'index($0, pat) == 1 { f = 1 } f { print } f && $0 == "}" { exit }' "$RUNNER" +} + +# Filters stdin down to lines that actually RUN something: full-line comments and +# diagnostic-only lines (printf/echo/runner_log) are dropped. Every assertion about a guard +# or an invocation pipes through this, because prose that merely NAMES a command must not be +# able to satisfy a check that the command is called — the failure mode that lets a deleted +# guard pass while its explanatory comment survives. +code_only() { + grep -vE '^[[:space:]]*#' | grep -vE '^[[:space:]]*(printf|echo|runner_log)[[:space:]]' +} + +# --- (1) the five public entrypoints plus the budget helper exist ----------- +for fn in runner_detect runner_launch runner_verify_cwd runner_alive runner_teardown resolve_timeout_ms; do + grep -qE "^$fn\(\) \{" "$RUNNER" || missing+=("session-runner.sh: function $fn is missing") +done + +detect="$(fn_body runner_detect)" +eligible="$(fn_body runner_herdr_eligible)" +watch="$(fn_body runner_watch)" +teardown="$(fn_body runner_teardown)" +abort="$(fn_body runner_abort)" + +# --- (2) ladder order: herdr -> tmux -> foreground ------------------------- +grep -qE 'herdr[^A-Za-z]+->[^A-Za-z]+tmux[^A-Za-z]+->[^A-Za-z]+foreground' "$RUNNER" \ + || missing+=("session-runner.sh: the ladder order herdr -> tmux -> foreground is not declared") + +# Mechanical, not documentary: inside runner_detect's AUTOMATIC ladder tail (comment lines +# stripped, so the declaring comment cannot satisfy this), herdr must be reached before tmux +# before foreground. +ladder="$(printf '%s\n' "$detect" | awk '/Automatic ladder/{f=1; next} f' | grep -v '^[[:space:]]*#')" +if [ -z "$ladder" ]; then + missing+=("runner_detect: no automatic-ladder branch found (only explicit-request handling remains)") +else + h_at="$(printf '%s\n' "$ladder" | grep -n 'herdr' | head -1 | cut -d: -f1)" + t_at="$(printf '%s\n' "$ladder" | grep -n 'tmux' | head -1 | cut -d: -f1)" + f_at="$(printf '%s\n' "$ladder" | grep -n 'foreground' | head -1 | cut -d: -f1)" + if [ -z "$h_at" ] || [ -z "$t_at" ] || [ -z "$f_at" ]; then + missing+=("runner_detect automatic ladder: one of herdr/tmux/foreground is not a rung at all") + elif [ "$h_at" -ge "$t_at" ] || [ "$t_at" -ge "$f_at" ]; then + missing+=("runner_detect automatic ladder: rungs are not in herdr -> tmux -> foreground order (herdr@$h_at tmux@$t_at foreground@$f_at)") + fi +fi + +# --- (3) the health predicate is the two literal `herdr status` fields ------ +# There is no single "healthy" flag in herdr status output, so these two literals ARE the +# whole predicate. A vague check would let binary-up/server-down select herdr. +if [ -z "$eligible" ]; then + missing+=("session-runner.sh: runner_herdr_eligible is missing — herdr eligibility has no gate") +else + # Each literal must appear in an actual grep of the `herdr status` output, not merely in + # the ineligibility MESSAGE that names it — otherwise deleting the test while keeping its + # error string would pass. + for field in 'status: running' 'compatible: yes'; do + printf '%s\n' "$eligible" | grep -F "$field" | grep -q 'grep' \ + || missing+=("runner_herdr_eligible: the literal field '$field' is not TESTED (naming it in a message is not a health check)") + done +fi + +# --- (4) NESTING GUARD before any probe pane ------------------------------- +# HERDR_ENV / HERDR_PANE_ID are herdr 0.7.4's own in-pane markers, inherited by every child +# of a pane. The permanent detection path must never itself nest a pane, so the guard has to +# run BEFORE the probe-pane launch, not after it. +if [ -n "$eligible" ]; then + eligible_code="$(printf '%s\n' "$eligible" | code_only)" + guard_at="$(printf '%s\n' "$eligible_code" | grep -n 'HERDR_ENV' | head -1 | cut -d: -f1)" + probe_at="$(printf '%s\n' "$eligible_code" | grep -n 'runner_probe_fingerprint' | head -1 | cut -d: -f1)" + if [ -z "$guard_at" ]; then + missing+=("runner_herdr_eligible: no HERDR_ENV nesting guard (the detection path could nest a herdr pane)") + elif [ -n "$probe_at" ] && [ "$guard_at" -ge "$probe_at" ]; then + missing+=("runner_herdr_eligible: the HERDR_ENV nesting guard runs AFTER the probe pane (guard@$guard_at probe@$probe_at)") + fi +fi + +# --- (5) EXECUTION-CONTEXT GATE: fingerprint compare, degrade, log --------- +# herdr panes may be host processes driving a mounted socket while this caller runs inside +# the sandbox. AGENTS.md requires building and testing to happen INSIDE the sandbox, so +# same-environment execution is proven by a probe-pane fingerprint, never assumed. +for fn in runner_local_fingerprint runner_probe_fingerprint; do + grep -qE "^$fn\(\) \{" "$RUNNER" || missing+=("session-runner.sh: the execution-context gate helper $fn is gone") +done +if [ -n "$eligible" ]; then + printf '%s\n' "$eligible" | code_only | grep -Fq 'runner_local_fingerprint' \ + || missing+=("runner_herdr_eligible: no caller-side fingerprint is gathered to compare against") + printf '%s\n' "$eligible" | code_only | grep -q '"\$probe_fp" != "\$caller_fp"\|"\$caller_fp" != "\$probe_fp"' \ + || missing+=("runner_herdr_eligible: the probe/caller fingerprints are never compared — an out-of-environment herdr would be selected") + printf '%s\n' "$eligible" | grep -Fq 'fingerprint mismatch' \ + || missing+=("runner_herdr_eligible: a fingerprint mismatch sets no named ineligibility reason") +fi +# The mismatch degrades DOWN the ladder and says why, in the firstmate log. +printf '%s\n' "$detect" | grep -Fq 'degrading to tmux' \ + || missing+=("runner_detect: an ineligible herdr does not degrade to tmux with a logged reason") +printf '%s\n' "$detect" | grep -Fq 'RUNNER_INELIGIBLE_REASON' \ + || missing+=("runner_detect: the degrade reason is not carried into the log (a silent degrade is unauditable)") +# The probe pane is torn down on BOTH verdicts — no pane is left behind either way. +probe_fn="$(fn_body runner_probe_fingerprint)" +printf '%s\n' "$probe_fn" | code_only | grep -Fq 'herdr pane close' \ + || missing+=("runner_probe_fingerprint: the probe pane is not closed (the gate leaks a pane per detection)") + +# --- (6) session budget: one source, the 14400000 default, bounded polling -- +grep -Fq 'RUNNER_DEFAULT_TIMEOUT_MS=14400000' "$RUNNER" \ + || missing+=("session-runner.sh: the 14400000 (4h) session-budget default literal is gone") +if [ -z "$watch" ]; then + missing+=("session-runner.sh: runner_watch is missing — nothing bounds the session") +else + printf '%s\n' "$watch" | grep -Fq 'resolve_timeout_ms' \ + || missing+=("runner_watch: the budget does not come from resolve_timeout_ms (a second budget source can diverge)") + printf '%s\n' "$watch" | grep -Eq 'deadline=.*budget_ms' \ + || missing+=("runner_watch: the poll deadline is not derived from the resolved budget") + printf '%s\n' "$watch" | grep -Eq 'while .*deadline' \ + || missing+=("runner_watch: the tmux/foreground poll loop is not bounded by the deadline (it could run unbounded)") + printf '%s\n' "$watch" | grep -Fq -- '--timeout "$budget_ms"' \ + || missing+=("runner_watch: the herdr wait output timeout is not the resolved budget") +fi + +# --- (7) every exit path: teardown, lock removal, FIRSTMATE-INCOMPLETE ------ +# A stale lock permanently wedges that slug, so no ending may leave one behind. +if [ -z "$abort" ]; then + missing+=("session-runner.sh: runner_abort is missing — there is no single exit path") +else + printf '%s\n' "$abort" | grep -Fq 'runner_teardown' \ + || missing+=("runner_abort: does not call runner_teardown") + printf '%s\n' "$abort" | grep -Fq 'runner_lock_path' \ + || missing+=("runner_abort: does not resolve the per-slug lock path") + printf '%s\n' "$abort" | grep -Eq 'rm -[a-z]* "\$lock"' \ + || missing+=("runner_abort: does not remove the lock (a stale lock wedges the slug permanently)") + printf '%s\n' "$abort" | grep -Fq 'FIRSTMATE-INCOMPLETE' \ + || missing+=("runner_abort: does not append FIRSTMATE-INCOMPLETE to progress.txt") +fi +# Budget expiry and death-without-sentinel both route through that one path. +if [ -n "$watch" ]; then + abort_calls="$(printf '%s\n' "$watch" | grep -c 'runner_abort')" + [ "$abort_calls" -ge 2 ] \ + || missing+=("runner_watch: only $abort_calls of the two non-sentinel endings (expiry, death) route through runner_abort") +fi +# Operator abort (Ctrl-C / SIGTERM) uses the same path. +trap_fn="$(fn_body runner_install_abort_trap)" +printf '%s\n' "$trap_fn" | grep -Fq 'runner_abort' \ + || missing+=("runner_install_abort_trap: an operator abort does not route through runner_abort") +printf '%s\n' "$trap_fn" | grep -Fq 'INT TERM' \ + || missing+=("runner_install_abort_trap: INT/TERM are not trapped") + +# --- (8) --no-focus on EVERY herdr launch ---------------------------------- +# Only real invocations count. Comment lines are excluded so the documenting header cannot +# satisfy this, and printf/echo/runner_log lines are excluded because they merely NAME the +# command in a diagnostic (e.g. "herdr agent start returned no pane id") rather than run it. +start_lines="$(grep -n 'herdr agent start' "$RUNNER" \ + | grep -vE '^[0-9]+:[[:space:]]*#' \ + | grep -vE '^[0-9]+:[[:space:]]*(printf|echo|runner_log)[[:space:]]')" || start_lines="" +if [ -z "$start_lines" ]; then + missing+=("session-runner.sh: no herdr agent start invocation — the herdr rung of the ladder is gone") +else + for ln in $(printf '%s\n' "$start_lines" | cut -d: -f1); do + # The invocation may wrap; look at the line plus its two continuations. + sed -n "${ln},$((ln + 2))p" "$RUNNER" | grep -Fq -- '--no-focus' \ + || missing+=("session-runner.sh:$ln — a herdr agent start invocation is missing --no-focus (it would steal the operator's focus)") + done +fi + +# --- (9) teardown verb: pane close, and no nonexistent stop/kill verb ------- +# herdr 0.7.4 has NO agent stop / agent kill verb (agent --help lists only +# list/get/read/send/rename/focus/wait/start/attach/explain), and a nonexistent verb inside +# a teardown trap fails silently — the session would survive the teardown. +printf '%s\n' "$teardown" | code_only | grep -Fq 'herdr pane close' \ + || missing+=("runner_teardown: the herdr branch does not INVOKE the live-verified 'herdr pane close' teardown verb (a log line naming it is not a teardown)") +printf '%s\n' "$teardown" | code_only | grep -Fq 'tmux kill-session' \ + || missing+=("runner_teardown: the tmux branch does not kill the agent- session") + +# --- (10) commands that must not appear in the library --------------------- +# These would disturb a SHARED herdr server or its operator config. `herdr agent get` is +# deliberately NOT on this list — it is the liveness oracle and is required (assertion 11). +FORBIDDEN=( + 'herdr server stop' + 'herdr update' + 'herdr channel set' + 'herdr agent stop' + 'herdr agent kill' + '.config/herdr' +) +for cmd in "${FORBIDDEN[@]}"; do + grep -Fq "$cmd" "$RUNNER" \ + && missing+=("session-runner.sh contains a forbidden command or path: '$cmd'") +done + +# --- (11) the liveness oracle is present ----------------------------------- +alive="$(fn_body runner_alive | code_only)" +printf '%s\n' "$alive" | grep -Fq 'herdr agent get' \ + || missing+=("runner_alive: herdr-mode liveness no longer uses 'herdr agent get' (its exit code is the oracle)") +printf '%s\n' "$alive" | grep -Fq 'tmux has-session' \ + || missing+=("runner_alive: tmux-mode liveness no longer uses 'tmux has-session'") + +# --- (12) sourceable library: the caller owns shell options ---------------- +# A file-scope `set` in a sourced file silently rewrites the caller's option state for the +# rest of its execution, and no linter flags it. Strictness belongs inside the functions. +if grep -nE '^set ' "$RUNNER" >/dev/null 2>&1; then + missing+=("session-runner.sh sets shell options at file scope — a sourced library must not mutate the caller's options") +fi + +if [ "${#missing[@]}" -gt 0 ]; then + printf 'REGRESSION: session-runner ladder contract broken:\n' >&2 + printf ' - %s\n' "${missing[@]}" >&2 + exit 1 +fi + +echo "PASS: ladder herdr->tmux->foreground, health pinned to status: running + compatible: yes, nesting guard before the probe pane, fingerprint gate degrades+logs, resolve_timeout_ms bounds every watch at 14400000 default, exit paths teardown+unlock+FIRSTMATE-INCOMPLETE, --no-focus on every launch, teardown via pane close, no server-disturbing commands, agent get oracle intact, no file-scope set" >&2 +exit 0 diff --git a/.oh/scripts/__tests__/firstmate.test.ts b/.oh/scripts/__tests__/firstmate.test.ts new file mode 100644 index 00000000..a5e2d5bb --- /dev/null +++ b/.oh/scripts/__tests__/firstmate.test.ts @@ -0,0 +1,863 @@ +import { afterEach, describe, expect, it } from "vitest"; +import { spawnSync } from "node:child_process"; +import { fileURLToPath } from "node:url"; +import { + chmodSync, + copyFileSync, + existsSync, + mkdirSync, + mkdtempSync, + readFileSync, + rmSync, + statSync, + writeFileSync, +} from "node:fs"; +import { tmpdir } from "node:os"; +import path from "node:path"; + +const __filename = fileURLToPath(import.meta.url); +const __dirname = path.dirname(__filename); + +const REPO_ROOT = path.resolve(__dirname, "../../.."); +const SCRIPT = path.join(REPO_ROOT, ".oh", "scripts", "firstmate.sh"); +const LIB = path.join(REPO_ROOT, ".oh", "scripts", "lib", "session-runner.sh"); +const TEMPLATE_REL = path.join( + ".oh", + "skills", + "firstmate", + "templates", + "session-prompt.md", +); +const TEMPLATE = path.join(REPO_ROOT, TEMPLATE_REL); +const SCRIPT_SOURCE = readFileSync(SCRIPT, "utf-8"); + +// Absolute bash, resolved once via the parent's PATH — the child env is fully +// controlled below, so a bare "bash" would resolve against the child's PATH. +const BASH = (() => { + const r = spawnSync("bash", ["-c", "command -v bash"], { encoding: "utf-8" }); + return (r.stdout ?? "").trim() || "/usr/bin/bash"; +})(); + +const scratch: string[] = []; + +function tmpDir(prefix: string): string { + const dir = mkdtempSync(path.join(tmpdir(), prefix)); + scratch.push(dir); + return dir; +} + +afterEach(() => { + while (scratch.length) { + rmSync(scratch.pop() as string, { recursive: true, force: true }); + } +}); + +// --------------------------------------------------------------------------- +// Fixture binaries +// +// Same shape as session-runner.test.ts: one stub per tool, driven entirely by +// STUB_* env vars, appending every invocation to $STUB_CALLS. The calls file is +// also how a NEGATIVE is proven — "the short-circuit launched nothing" is +// asserted as "the calls file does not exist". +// --------------------------------------------------------------------------- + +const HERDR_STUB = `#!/usr/bin/env bash +printf '%s\\n' "herdr $*" >> "\${STUB_CALLS:-/dev/null}" +sub="\${1:-}"; shift || true +case "$sub" in + status) + printf 'client:\\n version: 0.7.4\\n\\nserver:\\n status: running\\n compatible: yes\\n' + ;; + agent) + verb="\${1:-}"; shift || true + case "$verb" in + start) + printf '{"result":{"agent":{"cwd":"%s","foreground_cwd":"%s","pane_id":"%s"}},"type":"agent_started"}\\n' \\ + "\${STUB_HERDR_FG_CWD:-/w}" "\${STUB_HERDR_FG_CWD:-/w}" "\${STUB_HERDR_PANE_ID:-w7:p3}" + ;; + get) + if [ "\${STUB_HERDR_AGENT_LIVE:-0}" = "1" ]; then + printf '{"result":{"agent":{"pane_id":"%s"}},"type":"agent_info"}\\n' "\${STUB_HERDR_PANE_ID:-w7:p3}" + exit 0 + fi + printf '{"error":{"type":"agent_not_found"}}\\n' >&2; exit 1 + ;; + *) exit 64 ;; + esac + ;; + wait) exit "\${STUB_HERDR_WAIT_RC:-0}" ;; + pane) + verb="\${1:-}"; shift || true + case "$verb" in + read) printf '%s\\n' "\${STUB_HERDR_PROBE_OUT:-}" ;; + list) + printf '{"result":{"panes":[{"pane_id":"%s","foreground_cwd":"%s","cwd":"%s"}]}}\\n' \\ + "\${STUB_HERDR_PANE_ID:-w7:p3}" "\${STUB_HERDR_FG_CWD:-/w}" "\${STUB_HERDR_FG_CWD:-/w}" + ;; + close) printf '{"result":{"type":"ok"}}\\n' ;; + *) exit 64 ;; + esac + ;; + *) exit 64 ;; +esac +`; + +// has-session is target-aware: firstmate asks about TWO different names — the +// bare slug (the ralph cross-executor guard) and agent-firstmate- (its own +// liveness oracle) — and the tests must be able to answer them differently. +const TMUX_STUB = `#!/usr/bin/env bash +printf '%s\\n' "tmux $*" >> "\${STUB_CALLS:-/dev/null}" +verb="\${1:-}" +case "$verb" in + has-session) + target="" + while [ "$#" -gt 0 ]; do + if [ "$1" = "-t" ]; then shift; target="\${1:-}"; fi + shift + done + for s in \${STUB_TMUX_SESSIONS:-}; do + if [ "$s" = "$target" ]; then exit 0; fi + done + exit 1 + ;; + new-session) exit "\${STUB_TMUX_NEW_SESSION_RC:-0}" ;; + kill-session) exit 0 ;; + *) exit 0 ;; +esac +`; + +function makeBin(opts: { herdr?: boolean; tmux?: boolean }): string { + const dir = tmpDir("fm-bin-"); + if (opts.herdr) { + const p = path.join(dir, "herdr"); + writeFileSync(p, HERDR_STUB); + chmodSync(p, 0o755); + } + if (opts.tmux) { + const p = path.join(dir, "tmux"); + writeFileSync(p, TMUX_STUB); + chmodSync(p, 0o755); + } + return dir; +} + +// --------------------------------------------------------------------------- +// Fixture repo: a task folder honouring the four-file contract, plus the real +// session-prompt template copied in so the renderer runs against the shipped +// artifact rather than a stand-in. +// --------------------------------------------------------------------------- + +interface Repo { + root: string; + slug: string; + taskDir: string; + progressFile: string; + runnerTmp: string; + lock: string; + promptFile: string; + callsFile: string; +} + +function makeRepo( + slug: string, + opts: { progress?: string; files?: string[]; branch?: string } = {}, +): Repo { + const root = tmpDir("fm-repo-"); + const taskDir = path.join(root, ".oh", "tasks", slug); + const runnerTmp = path.join(root, "tmp"); + mkdirSync(taskDir, { recursive: true }); + mkdirSync(runnerTmp, { recursive: true }); + mkdirSync(path.join(root, path.dirname(TEMPLATE_REL)), { recursive: true }); + copyFileSync(TEMPLATE, path.join(root, TEMPLATE_REL)); + + const files = opts.files ?? [ + "prd.md", + "prd.json", + "prompt.md", + "progress.txt", + ]; + for (const f of files) { + if (f === "prd.json") { + writeFileSync( + path.join(taskDir, f), + JSON.stringify( + { + branchName: opts.branch ?? `feat/4242-${slug}`, + description: `a fixture task (issue #4242)`, + userStories: [], + }, + null, + 2, + ), + ); + } else if (f === "progress.txt") { + writeFileSync(path.join(taskDir, f), opts.progress ?? "# progress\n"); + } else { + writeFileSync(path.join(taskDir, f), `# ${f}\n`); + } + } + + return { + root, + slug, + taskDir, + progressFile: path.join(taskDir, "progress.txt"), + runnerTmp, + lock: path.join(runnerTmp, `firstmate-${slug}.lock`), + promptFile: path.join(runnerTmp, `firstmate-${slug}.prompt.md`), + callsFile: path.join(root, "calls.txt"), + }; +} + +interface RunResult { + stdout: string; + stderr: string; + status: number; +} + +function run( + args: string[], + opts: { repo?: Repo; bin?: string; env?: NodeJS.ProcessEnv } = {}, +): RunResult { + const repo = opts.repo; + const env: NodeJS.ProcessEnv = { + // The stub bin is PREPENDED to the real PATH: the tests here need git, jq, + // sed and tee to be real, and only want herdr/tmux shadowed. + PATH: opts.bin ? `${opts.bin}:${process.env.PATH ?? ""}` : process.env.PATH, + ...(repo + ? { + HOME: repo.root, + RUNNER_TMPDIR: repo.runnerTmp, + STUB_CALLS: repo.callsFile, + } + : {}), + ...opts.env, + }; + const result = spawnSync(BASH, [SCRIPT, ...args], { + encoding: "utf-8", + cwd: repo?.root, + env, + }); + return { + stdout: result.stdout ?? "", + stderr: result.stderr ?? "", + status: result.status ?? -1, + }; +} + +/** Source firstmate.sh (the source guard stops before the main body) and run a snippet. */ +function sourceCall( + snippet: string, + opts: { env?: NodeJS.ProcessEnv; cwd?: string } = {}, +): RunResult { + const result = spawnSync(BASH, ["-c", `source '${SCRIPT}'\n${snippet}`], { + encoding: "utf-8", + cwd: opts.cwd, + env: { PATH: process.env.PATH, ...opts.env }, + }); + return { + stdout: result.stdout ?? "", + stderr: result.stderr ?? "", + status: result.status ?? -1, + }; +} + +function readCalls(repo: Repo): string { + try { + return readFileSync(repo.callsFile, "utf-8"); + } catch { + return ""; + } +} + +/** + * The probe-pane output that MATCHES the caller's own fingerprint, so the + * execution-context gate lets herdr through. The library strips the marker when + * it reads a fingerprint back, so the stub has to re-emit the marked line. + */ +function matchingProbeOutput(worktree: string): string { + const r = spawnSync( + BASH, + ["-c", `source '${LIB}'\nrunner_local_fingerprint '${worktree}'`], + { encoding: "utf-8" }, + ); + return `FIRSTMATE-FINGERPRINT ${(r.stdout ?? "").trim()}`; +} + +// --------------------------------------------------------------------------- +// Slug + four-file contract (the shared sourced helper) +// --------------------------------------------------------------------------- + +describe("task-folder validation", () => { + it("rejects a slug that is not kebab-case, with ralph.sh's message", () => { + const repo = makeRepo("ok-slug"); + const r = run(["Bad Slug"], { repo }); + expect(r.status).toBe(2); + expect(r.stderr).toContain("must match ^[a-z0-9-]+$"); + }); + + it("rejects a missing task folder", () => { + const repo = makeRepo("ok-slug"); + const r = run(["absent-slug"], { repo }); + expect(r.status).toBe(1); + expect(r.stderr).toContain("does not exist"); + expect(r.stderr).toContain("scaffold a task with /prd then /ralph first"); + }); + + // The four-file contract is the interface both executors share; a partial + // folder is a scaffolding bug, never something to launch a session against. + for (const missing of ["prd.md", "prd.json", "prompt.md", "progress.txt"]) { + it(`rejects a task folder missing ${missing}`, () => { + const files = ["prd.md", "prd.json", "prompt.md", "progress.txt"].filter( + (f) => f !== missing, + ); + const repo = makeRepo("partial", { files }); + const r = run(["partial"], { repo }); + expect(r.status).toBe(1); + expect(r.stderr).toContain(`${missing} is missing`); + expect(r.stderr).toContain("four-file contract"); + }); + } + + it("validates via the shared sourced helper, not a private copy", () => { + expect(SCRIPT_SOURCE).toContain("lib/task-contract.sh"); + expect(SCRIPT_SOURCE).toContain("task_contract_validate_slug"); + expect(SCRIPT_SOURCE).toContain("task_contract_validate_dir"); + }); + + it("rejects an unknown option and an unknown harness", () => { + const repo = makeRepo("ok-slug"); + const bad = run(["--nope", "ok-slug"], { repo }); + expect(bad.status).toBe(2); + expect(bad.stderr).toContain("unknown option"); + + const harness = run(["--harness", "deepagents", "ok-slug"], { repo }); + expect(harness.status).toBe(2); + expect(harness.stderr).toContain("unknown harness"); + }); + + it("requires exactly one positional slug", () => { + const repo = makeRepo("ok-slug"); + const r = run([], { repo }); + expect(r.status).toBe(2); + expect(r.stderr).toContain("Usage:"); + }); +}); + +// --------------------------------------------------------------------------- +// Sentinel short-circuit +// --------------------------------------------------------------------------- + +describe("sentinel short-circuit", () => { + it("exits 0 without launching anything when STATUS: COMPLETE is already present", () => { + const repo = makeRepo("done-slug", { + progress: "# progress\n\nSTATUS: COMPLETE\n", + }); + const bin = makeBin({ tmux: true, herdr: true }); + const r = run(["done-slug"], { repo, bin }); + + expect(r.status).toBe(0); + expect(r.stdout).toContain("already present"); + // Proof of the negative: no runner was ever consulted, and no lock claimed. + expect(readCalls(repo)).toBe(""); + expect(existsSync(repo.lock)).toBe(false); + }); + + it("is anchored to the whole line — prose about the marker does not short-circuit", () => { + const repo = makeRepo("prose-slug", { + progress: "# progress\n\nI will append STATUS: COMPLETE when done.\n", + }); + const bin = makeBin({ tmux: true }); + const r = run(["--runner", "tmux", "--no-watch", "prose-slug"], { + repo, + bin, + }); + + expect(r.status).toBe(0); + expect(r.stdout).not.toContain("already present"); + expect(readCalls(repo)).toContain("tmux new-session"); + }); +}); + +// --------------------------------------------------------------------------- +// Cross-executor guard +// --------------------------------------------------------------------------- + +describe("cross-executor guard", () => { + // ralph names its tmux session after the BARE slug; firstmate names its own + // agent-firstmate-. The two never collide on a session name, so nothing + // but this guard stops both from writing the same progress.txt. + it("refuses to launch while a ralph tmux session for the same slug is live", () => { + const repo = makeRepo("busy-slug"); + const bin = makeBin({ tmux: true }); + const r = run(["--runner", "tmux", "busy-slug"], { + repo, + bin, + env: { STUB_TMUX_SESSIONS: "busy-slug" }, + }); + + expect(r.status).not.toBe(0); + expect(r.stderr).toContain("executor conflict"); + expect(r.stderr).toContain("ralph"); + expect(r.stderr).toContain("busy-slug"); + // It refuses BEFORE claiming the lock, so a refusal leaves no debris. + expect(existsSync(repo.lock)).toBe(false); + expect(readCalls(repo)).not.toContain("new-session"); + }); + + it("does not mistake its own agent-firstmate- session for a ralph run", () => { + const repo = makeRepo("mine-slug"); + const bin = makeBin({ tmux: true }); + const r = run(["--runner", "tmux", "--no-watch", "mine-slug"], { + repo, + bin, + env: { STUB_TMUX_SESSIONS: "agent-firstmate-mine-slug" }, + }); + + expect(r.stderr).not.toContain("executor conflict"); + }); +}); + +// --------------------------------------------------------------------------- +// Idempotency: the atomic mkdir launch-claim + the stale-lock reclaim +// --------------------------------------------------------------------------- + +describe("launch-claim lock", () => { + it("claims /tmp/firstmate-.lock with an atomic mkdir on a clean launch", () => { + const repo = makeRepo("lock-slug"); + const bin = makeBin({ tmux: true }); + const r = run(["--runner", "tmux", "--no-watch", "lock-slug"], { + repo, + bin, + }); + + expect(r.status).toBe(0); + expect(existsSync(repo.lock)).toBe(true); + expect(statSync(repo.lock).isDirectory()).toBe(true); + expect(SCRIPT_SOURCE).toContain('mkdir "$lock"'); + }); + + it("refuses a second launch while the lock is held AND the session is live", () => { + const repo = makeRepo("live-slug"); + const bin = makeBin({ tmux: true }); + mkdirSync(repo.lock); + + const r = run(["--runner", "tmux", "--no-watch", "live-slug"], { + repo, + bin, + env: { STUB_TMUX_SESSIONS: "agent-firstmate-live-slug" }, + }); + + expect(r.status).not.toBe(0); + expect(r.stderr).toContain("already running"); + expect(r.stderr).toContain("--kill live-slug"); + expect(readCalls(repo)).not.toContain("new-session"); + }); + + // The herdr-mode oracle is `herdr agent get `'s EXIT CODE: 0 = + // agent_info (live), 1 = agent_not_found (gone). + it("cross-checks liveness with `herdr agent get` in herdr mode", () => { + const repo = makeRepo("held-slug"); + const bin = makeBin({ herdr: true }); + const fp = matchingProbeOutput(repo.root); + mkdirSync(repo.lock); + + const live = run(["--runner", "herdr", "--no-watch", "held-slug"], { + repo, + bin, + env: { + STUB_HERDR_PROBE_OUT: fp, + STUB_HERDR_FG_CWD: repo.root, + STUB_HERDR_AGENT_LIVE: "1", + }, + }); + expect(live.status).not.toBe(0); + expect(live.stderr).toContain("already running"); + expect(readCalls(repo)).toContain("herdr agent get firstmate-held-slug"); + + // Same lock, but the agent is gone (agent_not_found, exit 1) — reclaimable. + const gone = run(["--runner", "herdr", "--no-watch", "held-slug"], { + repo, + bin, + env: { + STUB_HERDR_PROBE_OUT: fp, + STUB_HERDR_FG_CWD: repo.root, + STUB_HERDR_AGENT_LIVE: "0", + }, + }); + expect(gone.status).toBe(0); + expect(gone.stderr).toContain("stale lock"); + }); + + // The kill -9 shape: the lock survived, the session did not. A stale lock must + // never wedge a slug permanently. + it("treats a lock with no live session as stale and reclaimable", () => { + const repo = makeRepo("stale-slug"); + const bin = makeBin({ tmux: true }); + mkdirSync(repo.lock); + // Debris inside the lock proves the reclaim really recreated the directory. + writeFileSync(path.join(repo.lock, "debris"), "from the crashed run\n"); + + const r = run(["--runner", "tmux", "--no-watch", "stale-slug"], { + repo, + bin, + env: { STUB_TMUX_SESSIONS: "" }, + }); + + expect(r.status).toBe(0); + expect(r.stderr).toContain("stale lock"); + expect(existsSync(repo.lock)).toBe(true); + expect(existsSync(path.join(repo.lock, "debris"))).toBe(false); + expect(readCalls(repo)).toContain("tmux new-session"); + }); +}); + +// --------------------------------------------------------------------------- +// Exit paths — no path may leave the lock behind +// --------------------------------------------------------------------------- + +describe("exit paths", () => { + it("removes the lock and appends FIRSTMATE-INCOMPLETE after a launch failure", () => { + const repo = makeRepo("fail-slug"); + const bin = makeBin({ tmux: true }); + const r = run(["--runner", "tmux", "--no-watch", "fail-slug"], { + repo, + bin, + env: { STUB_TMUX_NEW_SESSION_RC: "1" }, + }); + + expect(r.status).not.toBe(0); + expect(existsSync(repo.lock)).toBe(false); + const progress = readFileSync(repo.progressFile, "utf-8"); + expect(progress).toContain("FIRSTMATE-INCOMPLETE"); + expect(progress).toContain("launch failure"); + }); + + it("removes the lock when the herdr session cannot be verified in the launch cwd", () => { + const repo = makeRepo("cwd-slug"); + const bin = makeBin({ herdr: true }); + const fp = matchingProbeOutput(repo.root); + const r = run(["--runner", "herdr", "--no-watch", "cwd-slug"], { + repo, + bin, + env: { + STUB_HERDR_PROBE_OUT: fp, + // The pane reports a DIFFERENT cwd than the one we launched into: the + // cwd flag lied, which in herdr mode means a different environment. + STUB_HERDR_FG_CWD: "/somewhere/else", + }, + }); + + expect(r.status).not.toBe(0); + expect(r.stderr).toContain("different environment"); + expect(existsSync(repo.lock)).toBe(false); + expect(readFileSync(repo.progressFile, "utf-8")).toContain( + "FIRSTMATE-INCOMPLETE", + ); + }); +}); + +// --------------------------------------------------------------------------- +// render_session_prompt — the renderer lives here (US-003), the contract in US-002 +// --------------------------------------------------------------------------- + +describe("render_session_prompt", () => { + const DECLARED = ["", "", ""]; + + it("substitutes every declared placeholder and lets none survive", () => { + const r = sourceCall( + `render_session_prompt '${TEMPLATE}' demo-slug feat/9-demo 9`, + ); + expect(r.status).toBe(0); + + // Every declared token is gone … + for (const token of DECLARED) { + expect(r.stdout).not.toContain(token); + } + // … and no OTHER angle-bracket token was introduced either, which is the + // second half of the contract: the renderer adds no token of its own. + const survivors = [...r.stdout.matchAll(/<[^ <>]+>/g)].map((m) => m[0]); + expect(survivors).toEqual([]); + + // … replaced by the real values, in every position they occurred. + expect(r.stdout).toContain("# First Mate session — demo-slug"); + expect(r.stdout).toContain(".oh/tasks/demo-slug/prd.json"); + expect(r.stdout).toContain("feat/9-demo"); + expect(r.stdout).toContain("Issue: #9"); + }); + + // `{curly}` is US-002's second notation: runtime-fill text the SESSION writes. + // Substituting it would destroy the progress-entry and commit templates. + it("leaves {curly-brace} runtime-fill text untouched", () => { + const r = sourceCall( + `render_session_prompt '${TEMPLATE}' demo-slug feat/9-demo 9`, + ); + expect(r.stdout).toContain("{story title}"); + expect(r.stdout).toContain("{YYYY-MM-DD HH:MM UTC}"); + expect(r.stdout).toContain("Submitted-by: {active harness identity}"); + }); + + it("drops the authoring contract header and keeps the whole body", () => { + const r = sourceCall( + `render_session_prompt '${TEMPLATE}' demo-slug feat/9-demo 9`, + ); + expect(r.stdout).not.toContain("END CONTRACT HEADER"); + expect(r.stdout).not.toContain("ANCHOR 1:"); + // The body's first and last sections both survive the header strip. + expect(r.stdout).toContain("## 1. Load the task graph"); + expect(r.stdout).toContain("## 9. Reference"); + }); + + it("requires all four arguments and an existing template", () => { + const missingArg = sourceCall( + `render_session_prompt '${TEMPLATE}' demo-slug feat/9-demo`, + ); + expect(missingArg.status).toBe(2); + expect(missingArg.stderr).toContain("requires all four arguments"); + + const missingTemplate = sourceCall( + `render_session_prompt /nope/session-prompt.md s b 1`, + ); + expect(missingTemplate.status).toBe(1); + expect(missingTemplate.stderr).toContain("is missing"); + }); + + it("substitutes only the declared set — an undeclared token is not its business", () => { + const dir = tmpDir("fm-tpl-"); + const tpl = path.join(dir, "session-prompt.md"); + writeFileSync(tpl, "body for and \n"); + const r = sourceCall(`render_session_prompt '${tpl}' s b 1`); + expect(r.status).toBe(0); + expect(r.stdout).toContain("body for s and "); + }); + + // A surviving in a live session prompt is silent and expensive. The + // substitutions run in declaration order, so a later value carrying an earlier + // token is the one shape that can reintroduce one — and the self-check refuses + // to emit the prompt rather than shipping it half-rendered. + it("refuses to emit a prompt in which a declared placeholder survived", () => { + const dir = tmpDir("fm-tpl-"); + const tpl = path.join(dir, "session-prompt.md"); + writeFileSync(tpl, "slug= branch=\n"); + const r = sourceCall(`render_session_prompt '${tpl}' s '' 1`); + expect(r.status).toBe(1); + expect(r.stderr).toContain("placeholder survived rendering"); + expect(r.stdout).toBe(""); + }); + + it("resolves the branch and issue from prd.json", () => { + const repo = makeRepo("meta-slug", { branch: "feat/77-meta" }); + const prd = path.join(repo.taskDir, "prd.json"); + const branch = sourceCall(`firstmate_branch_name '${prd}'`); + const issue = sourceCall(`firstmate_issue_number '${prd}'`); + expect(branch.stdout.trim()).toBe("feat/77-meta"); + expect(issue.stdout.trim()).toBe("4242"); + }); + + it("renders BARE DIGITS for the issue — the template writes its own #", () => { + const repo = makeRepo("hash-slug"); + const prd = path.join(repo.taskDir, "prd.json"); + const r = sourceCall(`firstmate_issue_number '${prd}'`, { + env: { PATH: process.env.PATH, FIRSTMATE_ISSUE: "#812" }, + }); + expect(r.stdout.trim()).toBe("812"); + }); +}); + +// --------------------------------------------------------------------------- +// Launch reporting + the naming contract +// --------------------------------------------------------------------------- + +describe("launch", () => { + it("prints the resolved runner mode, the session handle, the log path and the watch command", () => { + const repo = makeRepo("report-slug"); + const bin = makeBin({ tmux: true }); + const r = run(["--runner", "tmux", "--no-watch", "report-slug"], { + repo, + bin, + }); + + expect(r.status).toBe(0); + expect(r.stdout).toContain("runner: tmux"); + expect(r.stdout).toContain("handle: agent-firstmate-report-slug"); + expect(r.stdout).toContain( + path.join(repo.runnerTmp, "agent-firstmate-report-slug.log"), + ); + expect(r.stdout).toContain("watch: tmux attach -t agent-firstmate-report-slug"); + expect(r.stdout).toContain("budget: 14400000ms"); + }); + + it("uses the herdr naming contract and verifies foreground_cwd in herdr mode", () => { + const repo = makeRepo("herdr-slug"); + const bin = makeBin({ herdr: true }); + const fp = matchingProbeOutput(repo.root); + const r = run(["--runner", "herdr", "--no-watch", "herdr-slug"], { + repo, + bin, + env: { STUB_HERDR_PROBE_OUT: fp, STUB_HERDR_FG_CWD: repo.root }, + }); + + expect(r.status).toBe(0); + expect(r.stdout).toContain("runner: herdr"); + expect(r.stdout).toContain("handle: firstmate-herdr-slug (pane w7:p3)"); + expect(r.stdout).toContain( + path.join(repo.runnerTmp, "firstmate-herdr-slug.log"), + ); + expect(r.stdout).toContain("watch: herdr agent read firstmate-herdr-slug"); + + const calls = readCalls(repo); + expect(calls).toContain("herdr agent start firstmate-herdr-slug"); + expect(calls).toContain("--no-focus"); + expect(calls).toContain("herdr pane list"); + }); + + it("writes the rendered prompt to /tmp/firstmate-.prompt.md and launches from it", () => { + const repo = makeRepo("prompt-slug"); + const bin = makeBin({ tmux: true }); + const r = run(["--runner", "tmux", "--no-watch", "prompt-slug"], { + repo, + bin, + }); + + expect(r.status).toBe(0); + expect(existsSync(repo.promptFile)).toBe(true); + const rendered = readFileSync(repo.promptFile, "utf-8"); + expect(rendered).toContain("# First Mate session — prompt-slug"); + expect(rendered).toContain("feat/4242-prompt-slug"); + expect(rendered).not.toContain(""); + + // The launch string pipes that file into the harness and tees the log. + const calls = readCalls(repo); + expect(calls).toContain(repo.promptFile); + expect(calls).toContain("| tee "); + }); + + it("honours FIRSTMATE_HARNESS_CMD and exports the session's own signals", () => { + const repo = makeRepo("env-slug"); + const bin = makeBin({ tmux: true }); + const marker = path.join(repo.root, "env-marker.txt"); + const r = run(["--runner", "foreground", "--no-watch", "env-slug"], { + repo, + bin, + env: { + FIRSTMATE_HARNESS_CMD: `printf 'session=%s slug=%s prompt=%s\\n' "$FIRSTMATE_SESSION" "$FIRSTMATE_SLUG" "$FIRSTMATE_PROMPT_FILE" > ${marker}`, + }, + }); + + expect(r.status).toBe(0); + // The foreground child is detached from this process's wait, so poll briefly. + const deadline = Date.now() + 10_000; + while (!existsSync(marker) && Date.now() < deadline) { + spawnSync(BASH, ["-c", "sleep 0.2"]); + } + const observed = readFileSync(marker, "utf-8"); + expect(observed).toContain("session=1"); + expect(observed).toContain("slug=env-slug"); + expect(observed).toContain(repo.promptFile); + }); +}); + +// --------------------------------------------------------------------------- +// End-to-end: a session that reaches the sentinel +// --------------------------------------------------------------------------- + +describe("watch to completion", () => { + it("exits 0, tears down and releases the lock once STATUS: COMPLETE lands", () => { + const repo = makeRepo("e2e-slug"); + const bin = makeBin({ tmux: true }); + const r = run(["--runner", "foreground", "e2e-slug"], { + repo, + bin, + env: { + RUNNER_POLL_INTERVAL_S: "1", + FIRSTMATE_HARNESS_CMD: `printf 'STATUS: COMPLETE\\n' >> "$FIRSTMATE_TASK_DIR/progress.txt"`, + }, + }); + + expect(r.status).toBe(0); + expect(r.stdout).toContain("STATUS: COMPLETE observed"); + expect(existsSync(repo.lock)).toBe(false); + expect(readFileSync(repo.progressFile, "utf-8")).not.toContain( + "FIRSTMATE-INCOMPLETE", + ); + }); +}); + +// --------------------------------------------------------------------------- +// --kill: the manual escape hatch +// --------------------------------------------------------------------------- + +describe("--kill", () => { + it("clears the lock, tears the session down and records the outcome", () => { + const repo = makeRepo("kill-slug"); + const bin = makeBin({ tmux: true, herdr: true }); + mkdirSync(repo.lock); + + const r = run(["--kill", "kill-slug"], { + repo, + bin, + env: { STUB_HERDR_AGENT_LIVE: "1" }, + }); + + expect(r.status).toBe(0); + expect(existsSync(repo.lock)).toBe(false); + expect(readFileSync(repo.progressFile, "utf-8")).toContain( + "FIRSTMATE-INCOMPLETE", + ); + + const calls = readCalls(repo); + // herdr 0.7.4 has NO agent stop/kill verb — `pane close` is the primitive. + expect(calls).toContain("herdr pane close w7:p3"); + expect(calls).toContain("tmux kill-session -t agent-firstmate-kill-slug"); + expect(calls).not.toContain("agent stop"); + expect(calls).not.toContain("agent kill"); + // The server is never stopped or restarted. + expect(calls).not.toContain("server stop"); + expect(r.stdout).toContain("the herdr server was not stopped or restarted"); + }); + + it("still validates the slug", () => { + const repo = makeRepo("kill-slug"); + const r = run(["--kill", "NOPE"], { repo }); + expect(r.status).toBe(2); + expect(r.stderr).toContain("must match ^[a-z0-9-]+$"); + }); +}); + +// --------------------------------------------------------------------------- +// Static contract — the things a probe will also pin +// --------------------------------------------------------------------------- + +describe("static contract", () => { + it("is executable and launches via the US-001 session-runner library", () => { + expect(statSync(SCRIPT).mode & 0o111).not.toBe(0); + expect(SCRIPT_SOURCE).toContain("lib/session-runner.sh"); + for (const fn of [ + "runner_detect", + "runner_launch", + "runner_verify_cwd", + "runner_alive", + "runner_teardown", + "runner_abort", + "resolve_timeout_ms", + ]) { + expect(SCRIPT_SOURCE).toContain(fn); + } + }); + + it("references no nonexistent herdr verb and never touches the server", () => { + expect(SCRIPT_SOURCE).not.toMatch(/herdr agent (stop|kill)/); + expect(SCRIPT_SOURCE).not.toMatch(/herdr server stop/); + expect(SCRIPT_SOURCE).not.toMatch(/herdr update/); + expect(SCRIPT_SOURCE).not.toMatch(/herdr channel set/); + }); + + it("leaves .oh/scripts/ralph.sh zero-diff", () => { + const r = spawnSync( + "git", + ["diff", "--stat", "--", ".oh/scripts/ralph.sh"], + { cwd: REPO_ROOT, encoding: "utf-8" }, + ); + expect((r.stdout ?? "").trim()).toBe(""); + }); +}); diff --git a/.oh/scripts/__tests__/session-runner.test.ts b/.oh/scripts/__tests__/session-runner.test.ts new file mode 100644 index 00000000..1531bff5 --- /dev/null +++ b/.oh/scripts/__tests__/session-runner.test.ts @@ -0,0 +1,1008 @@ +import { afterEach, describe, expect, it } from "vitest"; +import { spawnSync } from "node:child_process"; +import { fileURLToPath } from "node:url"; +import { + chmodSync, + mkdirSync, + mkdtempSync, + readFileSync, + rmSync, + symlinkSync, + writeFileSync, +} from "node:fs"; +import { tmpdir } from "node:os"; +import path from "node:path"; + +const __filename = fileURLToPath(import.meta.url); +const __dirname = path.dirname(__filename); + +const REPO_ROOT = path.resolve(__dirname, "../../.."); +const LIB = path.join(REPO_ROOT, ".oh", "scripts", "lib", "session-runner.sh"); +const LIB_SOURCE = readFileSync(LIB, "utf-8"); + +// Absolute bash, resolved once via the parent's PATH. The ladder tests hand the +// child a PATH containing ONLY a fixture bin dir (that is how "herdr absent" / +// "tmux absent" are staged), and spawnSync resolves a bare command name against +// that stripped child PATH — which would ENOENT. +const BASH = (() => { + const r = spawnSync("bash", ["-c", "command -v bash"], { encoding: "utf-8" }); + return (r.stdout ?? "").trim() || "/usr/bin/bash"; +})(); + +// Real paths of the tools the library itself shells out to. Symlinked into an +// isolated fixture bin so a test can withhold `herdr`/`tmux` specifically +// without also withholding coreutils. +const PASSTHROUGH_TOOLS = [ + "bash", + "sh", + "cat", + "date", + "dirname", + "env", + "grep", + "head", + "hostname", + "jq", + "kill", + "ls", + "mkdir", + "mktemp", + "printf", + "rm", + "rmdir", + "sed", + "sleep", + "tail", + "tee", + "touch", + "uname", +]; + +const scratch: string[] = []; + +function tmpDir(prefix: string): string { + const dir = mkdtempSync(path.join(tmpdir(), prefix)); + scratch.push(dir); + return dir; +} + +afterEach(() => { + while (scratch.length) { + rmSync(scratch.pop() as string, { recursive: true, force: true }); + } +}); + +// --------------------------------------------------------------------------- +// Fixture binaries +// +// The stubs are driven entirely by STUB_* env vars, so one stub covers every +// herdr/tmux shape the ladder has to cope with. Every invocation is appended to +// $STUB_CALLS, which is how the tests assert that (for example) the nesting +// guard never reached `herdr agent start`. +// --------------------------------------------------------------------------- + +const HERDR_STUB = `#!/usr/bin/env bash +printf '%s\\n' "herdr $*" >> "\${STUB_CALLS:-/dev/null}" +sub="\${1:-}"; shift || true +case "$sub" in + status) + if [ "\${STUB_HERDR_SERVER:-running}" != "running" ]; then + printf 'client:\\n version: 0.7.4\\n\\nserver:\\n status: unavailable\\n compatible: unknown\\n' + exit 1 + fi + printf 'client:\\n version: 0.7.4\\n channel: stable\\n protocol: 16\\n\\nserver:\\n status: running\\n version: 0.7.4\\n protocol: 16\\n compatible: yes\\n socket: /tmp/herdr.sock\\n\\nupdate:\\n restart_needed: no\\n' + ;; + agent) + verb="\${1:-}"; shift || true + case "$verb" in + start) + if [ "\${STUB_HERDR_START_FAIL:-0}" = "1" ]; then + printf '{"error":{"type":"start_failed"}}\\n'; exit 1 + fi + # Shape observed live 2026-08-12 (agent_started payload). + printf '{"id":"cli:agent:start","result":{"agent":{"cwd":"%s","foreground_cwd":"%s","pane_id":"%s"}},"type":"agent_started"}\\n' \\ + "\${STUB_HERDR_FG_CWD:-/w}" "\${STUB_HERDR_FG_CWD:-/w}" "\${STUB_HERDR_PANE_ID:-w7:p3}" + ;; + get) + if [ "\${STUB_HERDR_AGENT_LIVE:-1}" = "1" ]; then + printf '{"result":{"agent":{"pane_id":"%s"}},"type":"agent_info"}\\n' "\${STUB_HERDR_PANE_ID:-w7:p3}" + exit 0 + fi + printf '{"error":{"type":"agent_not_found"}}\\n' >&2; exit 1 + ;; + *) exit 64 ;; + esac + ;; + wait) + exit "\${STUB_HERDR_WAIT_RC:-0}" + ;; + pane) + verb="\${1:-}"; shift || true + case "$verb" in + read) printf '%s\\n' "\${STUB_HERDR_PROBE_OUT:-}" ;; + list) + printf '{"id":"cli:pane:list","result":{"panes":[{"pane_id":"%s","foreground_cwd":"%s","cwd":"%s"}],"type":"pane_list"}}\\n' \\ + "\${STUB_HERDR_PANE_ID:-w7:p3}" "\${STUB_HERDR_FG_CWD:-/w}" "\${STUB_HERDR_FG_CWD:-/w}" + ;; + close) printf '{"result":{"type":"ok"}}\\n' ;; + *) exit 64 ;; + esac + ;; + *) exit 64 ;; +esac +`; + +const TMUX_STUB = `#!/usr/bin/env bash +printf '%s\\n' "tmux $*" >> "\${STUB_CALLS:-/dev/null}" +case "\${1:-}" in + has-session) exit "\${STUB_TMUX_HAS_SESSION:-0}" ;; + new-session) exit "\${STUB_TMUX_NEW_SESSION_RC:-0}" ;; + kill-session) exit 0 ;; + *) exit 0 ;; +esac +`; + +interface BinOpts { + /** Write the herdr stub into the fixture bin. */ + herdr?: boolean; + /** Write the tmux stub into the fixture bin. */ + tmux?: boolean; + /** + * Isolated: the child PATH is ONLY the fixture bin (plus symlinked + * coreutils), so a tool with no stub is genuinely absent. Non-isolated: the + * fixture bin is prepended to the real PATH, so stubs shadow the real + * binaries and everything else stays reachable. + */ + isolated?: boolean; +} + +function makeBin(opts: BinOpts): { dir: string; pathEnv: string } { + const dir = tmpDir("sr-bin-"); + if (opts.herdr) { + const p = path.join(dir, "herdr"); + writeFileSync(p, HERDR_STUB); + chmodSync(p, 0o755); + } + if (opts.tmux) { + const p = path.join(dir, "tmux"); + writeFileSync(p, TMUX_STUB); + chmodSync(p, 0o755); + } + if (opts.isolated) { + for (const tool of PASSTHROUGH_TOOLS) { + const r = spawnSync(BASH, ["-c", `command -v ${tool}`], { + encoding: "utf-8", + }); + const real = (r.stdout ?? "").trim(); + if (!real || !real.startsWith("/")) continue; + try { + symlinkSync(real, path.join(dir, tool)); + } catch { + /* already present (e.g. a stub of the same name) */ + } + } + return { dir, pathEnv: dir }; + } + return { dir, pathEnv: `${dir}:${process.env.PATH ?? ""}` }; +} + +interface RunResult { + stdout: string; + stderr: string; + status: number; + signal: string | null; + ms: number; +} + +/** + * Source the library in a fresh bash and run `snippet`. Sourcing only defines + * functions and assigns constants — the library sets no shell options, so the + * caller's option state (and this test runner's) is never touched. + */ +function sh( + snippet: string, + opts: { env?: NodeJS.ProcessEnv; timeoutMs?: number } = {}, +): RunResult { + const started = Date.now(); + const result = spawnSync(BASH, ["-c", `source '${LIB}'\n${snippet}`], { + encoding: "utf-8", + env: { ...opts.env }, + timeout: opts.timeoutMs, + }); + return { + stdout: result.stdout ?? "", + stderr: result.stderr ?? "", + status: result.status ?? -1, + signal: result.signal ?? null, + ms: Date.now() - started, + }; +} + +/** A task folder with a progress.txt, plus an isolated RUNNER_TMPDIR. */ +function makeTask(slug: string, progress = "# progress\n") { + const root = tmpDir("sr-task-"); + const taskDir = path.join(root, "tasks", slug); + const worktree = path.join(root, "worktree"); + const runnerTmp = path.join(root, "tmp"); + mkdirSync(taskDir, { recursive: true }); + mkdirSync(worktree, { recursive: true }); + mkdirSync(runnerTmp, { recursive: true }); + writeFileSync(path.join(taskDir, "progress.txt"), progress); + return { + root, + taskDir, + worktree, + runnerTmp, + progressFile: path.join(taskDir, "progress.txt"), + lock: path.join(runnerTmp, `firstmate-${slug}.lock`), + callsFile: path.join(root, "calls.txt"), + }; +} + +/** + * Everything the stubs were asked to do. Absent file = the stub was never + * invoked at all, which is exactly what the nesting guard has to achieve. + */ +function readCalls(t: ReturnType): string { + try { + return readFileSync(t.callsFile, "utf-8"); + } catch { + return ""; + } +} + +/** The fingerprint the CALLER will compute, so a stub can match or diverge. */ +function callerFingerprint(worktree: string): string { + const r = spawnSync( + BASH, + ["-c", `source '${LIB}'\nrunner_local_fingerprint '${worktree}'`], + { encoding: "utf-8" }, + ); + return (r.stdout ?? "").trim(); +} + +// --------------------------------------------------------------------------- +// resolve_timeout_ms — the ONLY source of the session budget +// --------------------------------------------------------------------------- + +describe("resolve_timeout_ms", () => { + const DEFAULT = "14400000"; + + it("defaults to 14400000 (4h) when FIRSTMATE_TIMEOUT_MS is unset", () => { + const t = makeTask("budget"); + const r = sh(`resolve_timeout_ms budget`, { + env: { PATH: process.env.PATH, RUNNER_TMPDIR: t.runnerTmp }, + }); + expect(r.stdout.trim()).toBe(DEFAULT); + expect(r.stderr).not.toContain("rejected"); + }); + + it("honours a valid positive override", () => { + const t = makeTask("budget"); + const r = sh(`resolve_timeout_ms budget`, { + env: { + PATH: process.env.PATH, + RUNNER_TMPDIR: t.runnerTmp, + FIRSTMATE_TIMEOUT_MS: "60000", + }, + }); + expect(r.stdout.trim()).toBe("60000"); + expect(r.stderr).not.toContain("rejected"); + }); + + // 0 is the dangerous one: herdr's `wait output --timeout 0` semantics are + // unknown, and an unvalidated 0 would make the poll ceilings expire instantly + // (or never). Routing every consumer through this helper makes it unreachable. + for (const [label, value] of [ + ["zero", "0"], + ["negative", "-1"], + ["non-numeric", "abc"], + ["empty", ""], + ] as const) { + it(`rejects a ${label} FIRSTMATE_TIMEOUT_MS, falls back to the default, and logs it`, () => { + const t = makeTask("budget"); + const r = sh(`resolve_timeout_ms budget`, { + env: { + PATH: process.env.PATH, + RUNNER_TMPDIR: t.runnerTmp, + FIRSTMATE_TIMEOUT_MS: value, + }, + }); + expect(r.stdout.trim()).toBe(DEFAULT); + expect(r.stderr).toContain("rejected FIRSTMATE_TIMEOUT_MS"); + const log = readFileSync( + path.join(t.runnerTmp, "firstmate-budget.log"), + "utf-8", + ); + expect(log).toContain("rejected FIRSTMATE_TIMEOUT_MS"); + }); + } +}); + +// --------------------------------------------------------------------------- +// runner_detect — the ladder and its degrade cases +// --------------------------------------------------------------------------- + +describe("runner_detect ladder", () => { + function detectEnv( + t: ReturnType, + bin: { pathEnv: string }, + extra: NodeJS.ProcessEnv = {}, + ): NodeJS.ProcessEnv { + return { + PATH: bin.pathEnv, + HOME: t.root, + RUNNER_TMPDIR: t.runnerTmp, + STUB_CALLS: t.callsFile, + ...extra, + }; + } + + // Degrade case 1 — herdr absent. + it("degrades to tmux when herdr is not installed", () => { + const t = makeTask("ladder"); + const bin = makeBin({ tmux: true, isolated: true }); + const r = sh(`runner_detect ladder '${t.worktree}'`, { + env: detectEnv(t, bin), + }); + expect(r.stdout.trim()).toBe("tmux"); + expect(r.stderr).toContain("herdr is not installed"); + }); + + // Degrade case 2 — binary up, server down. + it("degrades to tmux when the herdr binary is up but the server is down", () => { + const t = makeTask("ladder"); + const bin = makeBin({ herdr: true, tmux: true, isolated: true }); + const r = sh(`runner_detect ladder '${t.worktree}'`, { + env: detectEnv(t, bin, { STUB_HERDR_SERVER: "down" }), + }); + expect(r.stdout.trim()).toBe("tmux"); + expect(r.stderr).toContain("status: running"); + expect(r.stderr).toContain("compatible: yes"); + // Never probed: the health predicate failed first. + expect(readFileSync(t.callsFile, "utf-8")).not.toContain("agent start"); + }); + + // Degrade case 3 — no herdr, no tmux. + it("degrades to foreground when neither herdr nor tmux is installed", () => { + const t = makeTask("ladder"); + const bin = makeBin({ isolated: true }); + const r = sh(`runner_detect ladder '${t.worktree}'`, { + env: detectEnv(t, bin), + }); + expect(r.stdout.trim()).toBe("foreground"); + expect(r.stderr).toContain("tmux is not installed"); + }); + + // Degrade case 4 — the execution-context gate, mismatch path. The stubbed + // probe pane answers with a fingerprint from a DIFFERENT machine, which is + // exactly the live topology this sandbox has (herdr panes are host + // processes). No real herdr is involved. + it("degrades to tmux when the probe pane's fingerprint does not match the caller's", () => { + const t = makeTask("ladder"); + const bin = makeBin({ herdr: true, tmux: true, isolated: true }); + const r = sh(`runner_detect ladder '${t.worktree}'`, { + env: detectEnv(t, bin, { + STUB_HERDR_PROBE_OUT: + "FIRSTMATE-FINGERPRINT host=some-other-host docker=no worktree=no", + }), + }); + expect(r.stdout.trim()).toBe("tmux"); + expect(r.stderr).toContain("fingerprint mismatch"); + // The reason names WHICH fields differed, and carries both fingerprints. + expect(r.stderr).toMatch(/fingerprint mismatch on [a-z,]*host/); + expect(r.stderr).toContain("host=some-other-host"); + // And it was written to the firstmate log, not only to stderr. + const log = readFileSync( + path.join(t.runnerTmp, "firstmate-ladder.log"), + "utf-8", + ); + expect(log).toContain("fingerprint mismatch"); + }); + + it("closes its own probe pane on the mismatch path", () => { + const t = makeTask("ladder"); + const bin = makeBin({ herdr: true, tmux: true, isolated: true }); + sh(`runner_detect ladder '${t.worktree}'`, { + env: detectEnv(t, bin, { + STUB_HERDR_PROBE_OUT: + "FIRSTMATE-FINGERPRINT host=some-other-host docker=no worktree=no", + }), + }); + const calls = readFileSync(t.callsFile, "utf-8"); + expect(calls).toContain("agent start"); + expect(calls).toMatch(/herdr pane close w7:p3/); + }); + + it("selects herdr — and still closes the probe pane — when the fingerprints match", () => { + const t = makeTask("ladder"); + const bin = makeBin({ herdr: true, tmux: true, isolated: true }); + const r = sh(`runner_detect ladder '${t.worktree}'`, { + env: detectEnv(t, bin, { + STUB_HERDR_PROBE_OUT: `FIRSTMATE-FINGERPRINT ${callerFingerprint(t.worktree)}`, + }), + }); + expect(r.stdout.trim()).toBe("herdr"); + expect(readFileSync(t.callsFile, "utf-8")).toMatch(/herdr pane close w7:p3/); + }); + + it("degrades to tmux when the probe pane yields no fingerprint at all", () => { + const t = makeTask("ladder"); + const bin = makeBin({ herdr: true, tmux: true, isolated: true }); + const r = sh(`runner_detect ladder '${t.worktree}'`, { + env: detectEnv(t, bin, { STUB_HERDR_PROBE_OUT: "" }), + }); + expect(r.stdout.trim()).toBe("tmux"); + expect(r.stderr).toContain("no probe fingerprint obtained"); + }); + + // The nesting guard is the ZEROTH check: a detection path that itself nests a + // pane would be self-defeating, so no probe may be launched at all. + it("nesting guard: HERDR_ENV=1 rules herdr out without launching a probe pane", () => { + const t = makeTask("ladder"); + const bin = makeBin({ herdr: true, tmux: true, isolated: true }); + const r = sh(`runner_detect ladder '${t.worktree}'`, { + env: detectEnv(t, bin, { HERDR_ENV: "1" }), + }); + expect(r.stdout.trim()).toBe("tmux"); + expect(r.stderr).toContain("nesting guard"); + expect(r.stderr).toContain("allow_nested=false"); + const calls = readCalls(t); + expect(calls).not.toContain("agent start"); + // Not even `herdr status` — the guard short-circuits before any herdr call. + expect(calls.trim()).toBe(""); + }); + + it("nesting guard: HERDR_PANE_ID alone also rules herdr out", () => { + const t = makeTask("ladder"); + const bin = makeBin({ herdr: true, tmux: true, isolated: true }); + const r = sh(`runner_detect ladder '${t.worktree}'`, { + env: detectEnv(t, bin, { HERDR_PANE_ID: "w1:p1" }), + }); + expect(r.stdout.trim()).toBe("tmux"); + expect(readCalls(t).trim()).toBe(""); + }); +}); + +// --------------------------------------------------------------------------- +// Explicit overrides are hard errors, never silent degrades +// --------------------------------------------------------------------------- + +describe("runner_detect overrides", () => { + it("rejects an unknown runner name", () => { + const t = makeTask("ov"); + const bin = makeBin({ herdr: true, tmux: true, isolated: true }); + const r = sh(`runner_detect ov '${t.worktree}' bogus`, { + env: { PATH: bin.pathEnv, RUNNER_TMPDIR: t.runnerTmp }, + }); + expect(r.status).not.toBe(0); + expect(r.stderr).toContain("unknown runner bogus"); + }); + + it("hard-errors when herdr is requested but not installed", () => { + const t = makeTask("ov"); + const bin = makeBin({ tmux: true, isolated: true }); + const r = sh(`runner_detect ov '${t.worktree}' herdr`, { + env: { PATH: bin.pathEnv, RUNNER_TMPDIR: t.runnerTmp }, + }); + expect(r.status).not.toBe(0); + expect(r.stdout.trim()).not.toBe("tmux"); + expect(r.stderr).toContain("herdr is not installed"); + }); + + // The override may never force a silent host-side run: an installed, healthy + // but OUT-OF-ENVIRONMENT herdr is a hard error naming the mismatch. + it("hard-errors — naming the fingerprint mismatch — when herdr is requested but out-of-environment", () => { + const t = makeTask("ov"); + const bin = makeBin({ herdr: true, tmux: true, isolated: true }); + const r = sh(`runner_detect ov '${t.worktree}' herdr`, { + env: { + PATH: bin.pathEnv, + RUNNER_TMPDIR: t.runnerTmp, + STUB_CALLS: t.callsFile, + STUB_HERDR_PROBE_OUT: + "FIRSTMATE-FINGERPRINT host=some-other-host docker=no worktree=no", + }, + }); + expect(r.status).not.toBe(0); + expect(r.stdout.trim()).not.toBe("tmux"); + expect(r.stderr).toContain("fingerprint mismatch"); + expect(r.stderr).toContain("Refusing to degrade silently"); + }); + + it("honours OH_RUNNER as the override channel", () => { + const t = makeTask("ov"); + const bin = makeBin({ tmux: true, isolated: true }); + const r = sh(`runner_detect ov '${t.worktree}'`, { + env: { + PATH: bin.pathEnv, + RUNNER_TMPDIR: t.runnerTmp, + OH_RUNNER: "herdr", + }, + }); + expect(r.status).not.toBe(0); + expect(r.stderr).toContain("requested explicitly but is unavailable"); + }); + + it("hard-errors when tmux is requested but not installed", () => { + const t = makeTask("ov"); + const bin = makeBin({ isolated: true }); + const r = sh(`runner_detect ov '${t.worktree}' tmux`, { + env: { PATH: bin.pathEnv, RUNNER_TMPDIR: t.runnerTmp }, + }); + expect(r.status).not.toBe(0); + expect(r.stderr).toContain("tmux is not installed"); + }); + + it("always honours an explicit foreground request", () => { + const t = makeTask("ov"); + const bin = makeBin({ isolated: true }); + const r = sh(`runner_detect ov '${t.worktree}' foreground`, { + env: { PATH: bin.pathEnv, RUNNER_TMPDIR: t.runnerTmp }, + }); + expect(r.status).toBe(0); + expect(r.stdout.trim()).toBe("foreground"); + }); +}); + +// --------------------------------------------------------------------------- +// runner_launch — pane id provenance, --no-focus, and the tee in every branch +// --------------------------------------------------------------------------- + +describe("runner_launch", () => { + it("parses the pane id out of the agent_started payload and exposes it via runner_pane_id", () => { + const t = makeTask("launch"); + const bin = makeBin({ herdr: true, isolated: true }); + const r = sh( + `runner_launch herdr launch '${t.worktree}' 'echo hi' >/dev/null 2>&1\nrunner_pane_id`, + { + env: { + PATH: bin.pathEnv, + RUNNER_TMPDIR: t.runnerTmp, + STUB_CALLS: t.callsFile, + STUB_HERDR_PANE_ID: "w9:p42", + }, + }, + ); + expect(r.stdout.trim()).toBe("w9:p42"); + }); + + it("passes --no-focus and tees the herdr log", () => { + const t = makeTask("launch"); + const bin = makeBin({ herdr: true, isolated: true }); + sh(`runner_launch herdr launch '${t.worktree}' 'echo hi' >/dev/null 2>&1`, { + env: { + PATH: bin.pathEnv, + RUNNER_TMPDIR: t.runnerTmp, + STUB_CALLS: t.callsFile, + }, + }); + const start = readFileSync(t.callsFile, "utf-8") + .split("\n") + .find((l) => l.includes("agent start")); + expect(start).toBeDefined(); + expect(start).toContain("--no-focus"); + expect(start).toContain("firstmate-launch"); + expect(start).toContain(`2>&1 | tee ${t.runnerTmp}/firstmate-launch.log`); + // The cd is inside the launched command, not only in the --cwd flag: + // runner flags that claim to set a cwd frequently set only metadata. + expect(start).toContain(`cd ${t.worktree} &&`); + }); + + it("tees the tmux log and uses the agent- category session name", () => { + const t = makeTask("launch"); + const bin = makeBin({ tmux: true, isolated: true }); + sh(`runner_launch tmux launch '${t.worktree}' 'echo hi' >/dev/null 2>&1`, { + env: { + PATH: bin.pathEnv, + RUNNER_TMPDIR: t.runnerTmp, + STUB_CALLS: t.callsFile, + }, + }); + const call = readFileSync(t.callsFile, "utf-8") + .split("\n") + .find((l) => l.includes("new-session")); + expect(call).toBeDefined(); + expect(call).toContain("-s agent-firstmate-launch"); + expect(call).toContain(`-c ${t.worktree}`); + expect(call).toContain( + `2>&1 | tee ${t.runnerTmp}/agent-firstmate-launch.log`, + ); + }); + + it("tees the foreground log too (new behavior — ralph's fallback does not log)", () => { + const t = makeTask("launch"); + const r = sh( + `runner_launch foreground launch '${t.worktree}' 'echo hello-foreground' >/dev/null 2>&1\nwait "$RUNNER_FG_PID"\ncat "${t.runnerTmp}/firstmate-launch.log"`, + { env: { PATH: process.env.PATH, RUNNER_TMPDIR: t.runnerTmp } }, + ); + expect(r.stdout).toContain("hello-foreground"); + }); + + it("fails loudly when herdr agent start returns no pane id", () => { + const t = makeTask("launch"); + const bin = makeBin({ herdr: true, isolated: true }); + const r = sh(`runner_launch herdr launch '${t.worktree}' 'echo hi'`, { + env: { + PATH: bin.pathEnv, + RUNNER_TMPDIR: t.runnerTmp, + STUB_CALLS: t.callsFile, + STUB_HERDR_START_FAIL: "1", + }, + }); + expect(r.status).not.toBe(0); + expect(r.stderr).toContain("returned no pane id"); + }); +}); + +// --------------------------------------------------------------------------- +// One pane id, three consumers +// --------------------------------------------------------------------------- + +describe("pane id provenance", () => { + it("feeds the SAME captured pane id to runner_verify_cwd, the watch, and teardown", () => { + const t = makeTask("pane", "# progress\nSTATUS: COMPLETE\n"); + const bin = makeBin({ herdr: true, isolated: true }); + const r = sh( + [ + `runner_launch herdr pane '${t.worktree}' 'echo hi' >/dev/null 2>&1`, + `runner_verify_cwd herdr '${t.worktree}' && echo VERIFY_OK`, + `runner_watch herdr pane '${t.taskDir}' >/dev/null 2>&1 && echo WATCH_OK`, + `runner_teardown herdr pane >/dev/null 2>&1`, + ].join("\n"), + { + env: { + PATH: bin.pathEnv, + RUNNER_TMPDIR: t.runnerTmp, + STUB_CALLS: t.callsFile, + STUB_HERDR_PANE_ID: "w9:p42", + STUB_HERDR_FG_CWD: t.worktree, + }, + }, + ); + expect(r.stdout).toContain("VERIFY_OK"); + expect(r.stdout).toContain("WATCH_OK"); + + const calls = readFileSync(t.callsFile, "utf-8"); + // The watch waits on the captured id... + expect(calls).toMatch(/herdr wait output w9:p42 --match \^STATUS: COMPLETE\$/); + // ...and teardown closes that same id. + expect(calls).toContain("herdr pane close w9:p42"); + }); + + it("rejects a pane that landed in a different cwd instead of loosening the check", () => { + const t = makeTask("pane"); + const bin = makeBin({ herdr: true, isolated: true }); + const r = sh( + `runner_launch herdr pane '${t.worktree}' 'echo hi' >/dev/null 2>&1\nrunner_verify_cwd herdr '${t.worktree}'`, + { + env: { + PATH: bin.pathEnv, + RUNNER_TMPDIR: t.runnerTmp, + STUB_CALLS: t.callsFile, + STUB_HERDR_FG_CWD: "/home/someone-else", + }, + }, + ); + expect(r.status).not.toBe(0); + expect(r.stderr).toContain("executing in a different environment"); + }); +}); + +// --------------------------------------------------------------------------- +// runner_alive — read-only oracles +// --------------------------------------------------------------------------- + +describe("runner_alive", () => { + it("uses `herdr agent get`'s exit code as the herdr-mode oracle", () => { + const t = makeTask("alive"); + const bin = makeBin({ herdr: true, isolated: true }); + const env = { + PATH: bin.pathEnv, + RUNNER_TMPDIR: t.runnerTmp, + STUB_CALLS: t.callsFile, + }; + const live = sh(`runner_alive herdr alive && echo LIVE`, { + env: { ...env, STUB_HERDR_AGENT_LIVE: "1" }, + }); + expect(live.stdout).toContain("LIVE"); + expect(readFileSync(t.callsFile, "utf-8")).toContain( + "herdr agent get firstmate-alive", + ); + + const gone = sh(`runner_alive herdr alive || echo GONE`, { + env: { ...env, STUB_HERDR_AGENT_LIVE: "0" }, + }); + expect(gone.stdout).toContain("GONE"); + }); + + it("uses `tmux has-session` as the tmux-mode oracle", () => { + const t = makeTask("alive"); + const bin = makeBin({ tmux: true, isolated: true }); + const env = { + PATH: bin.pathEnv, + RUNNER_TMPDIR: t.runnerTmp, + STUB_CALLS: t.callsFile, + }; + const live = sh(`runner_alive tmux alive && echo LIVE`, { + env: { ...env, STUB_TMUX_HAS_SESSION: "0" }, + }); + expect(live.stdout).toContain("LIVE"); + expect(readFileSync(t.callsFile, "utf-8")).toContain( + "tmux has-session -t agent-firstmate-alive", + ); + + const gone = sh(`runner_alive tmux alive || echo GONE`, { + env: { ...env, STUB_TMUX_HAS_SESSION: "1" }, + }); + expect(gone.stdout).toContain("GONE"); + }); +}); + +// --------------------------------------------------------------------------- +// runner_teardown — `pane close`, never a nonexistent stop/kill verb +// --------------------------------------------------------------------------- + +describe("runner_teardown", () => { + it("closes the pane in herdr mode, recovering the id from the server when it has none", () => { + const t = makeTask("down"); + const bin = makeBin({ herdr: true, isolated: true }); + // No prior launch in this shell: the id has to come back out of `agent get`. + sh(`runner_teardown herdr down >/dev/null 2>&1`, { + env: { + PATH: bin.pathEnv, + RUNNER_TMPDIR: t.runnerTmp, + STUB_CALLS: t.callsFile, + STUB_HERDR_PANE_ID: "w3:p8", + }, + }); + const calls = readFileSync(t.callsFile, "utf-8"); + expect(calls).toContain("herdr agent get firstmate-down"); + expect(calls).toContain("herdr pane close w3:p8"); + expect(calls).not.toContain("agent stop"); + expect(calls).not.toContain("agent kill"); + }); + + it("kills the agent- session in tmux mode", () => { + const t = makeTask("down"); + const bin = makeBin({ tmux: true, isolated: true }); + sh(`runner_teardown tmux down >/dev/null 2>&1`, { + env: { + PATH: bin.pathEnv, + RUNNER_TMPDIR: t.runnerTmp, + STUB_CALLS: t.callsFile, + }, + }); + expect(readFileSync(t.callsFile, "utf-8")).toContain( + "tmux kill-session -t agent-firstmate-down", + ); + }); +}); + +// --------------------------------------------------------------------------- +// runner_watch — bounded by resolve_timeout_ms, and the single exit path +// --------------------------------------------------------------------------- + +describe("runner_watch", () => { + it("returns success as soon as the whole-line sentinel is in progress.txt", () => { + const t = makeTask("watch", "# progress\nSTATUS: COMPLETE\n"); + const bin = makeBin({ tmux: true, isolated: true }); + const r = sh(`runner_watch tmux watch '${t.taskDir}' && echo DONE`, { + env: { + PATH: bin.pathEnv, + RUNNER_TMPDIR: t.runnerTmp, + STUB_CALLS: t.callsFile, + RUNNER_POLL_INTERVAL_S: "0.2", + }, + }); + expect(r.stdout).toContain("DONE"); + }); + + // The exit-path contract: teardown ran, the lock is gone, and the run is + // recorded as FIRSTMATE-INCOMPLETE. A retained lock would wedge the slug. + it("on budget expiry: tears down, removes the lock, and appends FIRSTMATE-INCOMPLETE", () => { + const t = makeTask("watch"); + const bin = makeBin({ tmux: true, isolated: true }); + mkdirSync(t.lock); + const r = sh( + `runner_watch tmux watch '${t.taskDir}'; echo "RC=$?"\n[ -d '${t.lock}' ] && echo LOCK_PRESENT || echo LOCK_GONE`, + { + env: { + PATH: bin.pathEnv, + RUNNER_TMPDIR: t.runnerTmp, + STUB_CALLS: t.callsFile, + STUB_TMUX_HAS_SESSION: "0", // session stays "alive" until the budget runs out + FIRSTMATE_TIMEOUT_MS: "1000", + RUNNER_POLL_INTERVAL_S: "0.2", + }, + timeoutMs: 30_000, + }, + ); + expect(r.stdout).toContain("RC=1"); + expect(r.stdout).toContain("LOCK_GONE"); + expect(readFileSync(t.progressFile, "utf-8")).toContain( + "FIRSTMATE-INCOMPLETE", + ); + expect(readFileSync(t.progressFile, "utf-8")).toContain( + "session budget of 1000ms expired", + ); + expect(readFileSync(t.callsFile, "utf-8")).toContain( + "tmux kill-session -t agent-firstmate-watch", + ); + }); + + it("treats death without the sentinel as FIRSTMATE-INCOMPLETE too", () => { + const t = makeTask("watch"); + const bin = makeBin({ tmux: true, isolated: true }); + mkdirSync(t.lock); + const r = sh(`runner_watch tmux watch '${t.taskDir}'; echo "RC=$?"`, { + env: { + PATH: bin.pathEnv, + RUNNER_TMPDIR: t.runnerTmp, + STUB_CALLS: t.callsFile, + STUB_TMUX_HAS_SESSION: "1", // gone + RUNNER_POLL_INTERVAL_S: "0.2", + }, + timeoutMs: 30_000, + }); + expect(r.stdout).toContain("RC=1"); + expect(readFileSync(t.progressFile, "utf-8")).toContain( + "session ended without STATUS: COMPLETE", + ); + }); + + // The poll ceiling comes from resolve_timeout_ms and nowhere else. Both + // halves matter: a valid budget must actually stop the loop... + it("bounds the tmux/foreground poll loop by the resolved budget", () => { + const t = makeTask("watch"); + const bin = makeBin({ tmux: true, isolated: true }); + const r = sh(`runner_watch tmux watch '${t.taskDir}'; echo "RC=$?"`, { + env: { + PATH: bin.pathEnv, + RUNNER_TMPDIR: t.runnerTmp, + STUB_CALLS: t.callsFile, + STUB_TMUX_HAS_SESSION: "0", + FIRSTMATE_TIMEOUT_MS: "2000", + RUNNER_POLL_INTERVAL_S: "0.2", + }, + timeoutMs: 60_000, + }); + expect(r.stdout).toContain("RC=1"); + expect(r.ms).toBeLessThan(30_000); + }); + + // ...and a rejected one must NOT become the ceiling. With FIRSTMATE_TIMEOUT_MS=0 + // an unvalidated budget would expire instantly; the validated default (4h) + // means this watch is still polling when the test kills it. + it("does not expire instantly when FIRSTMATE_TIMEOUT_MS=0 is rejected", () => { + const t = makeTask("watch"); + const bin = makeBin({ tmux: true, isolated: true }); + const r = sh(`runner_watch tmux watch '${t.taskDir}'; echo "RC=$?"`, { + env: { + PATH: bin.pathEnv, + RUNNER_TMPDIR: t.runnerTmp, + STUB_CALLS: t.callsFile, + STUB_TMUX_HAS_SESSION: "0", + FIRSTMATE_TIMEOUT_MS: "0", + RUNNER_POLL_INTERVAL_S: "0.2", + }, + timeoutMs: 4_000, + }); + // Killed by the test harness rather than having returned on its own. + expect(r.signal).not.toBeNull(); + expect(r.stdout).not.toContain("RC="); + expect(readFileSync(t.progressFile, "utf-8")).not.toContain( + "FIRSTMATE-INCOMPLETE", + ); + }); +}); + +// --------------------------------------------------------------------------- +// runner_abort — the single exit path, callable directly (launch failure, +// operator abort) +// --------------------------------------------------------------------------- + +describe("runner_abort", () => { + it("removes the lock even when the task folder is missing", () => { + const t = makeTask("abort"); + const bin = makeBin({ tmux: true, isolated: true }); + mkdirSync(t.lock); + const r = sh( + `runner_abort tmux abort '/nonexistent/task' 'launch failure' >/dev/null 2>&1\n[ -d '${t.lock}' ] && echo LOCK_PRESENT || echo LOCK_GONE`, + { + env: { + PATH: bin.pathEnv, + RUNNER_TMPDIR: t.runnerTmp, + STUB_CALLS: t.callsFile, + }, + }, + ); + expect(r.stdout).toContain("LOCK_GONE"); + }); +}); + +// --------------------------------------------------------------------------- +// The caller owns shell options +// --------------------------------------------------------------------------- + +describe("sourcing is option-neutral", () => { + // A file-scope `set` in a sourced library silently rewrites the caller's + // option state for the rest of its execution — and shellcheck does not flag + // it. Both callers below must come out exactly as they went in. + for (const [label, prelude] of [ + ["a strict caller", "set -euo pipefail"], + ["a non-strict caller", "set +e +u +o pipefail"], + ] as const) { + it(`leaves ${label}'s options untouched`, () => { + const r = spawnSync( + BASH, + [ + "-c", + [ + prelude, + `before="$-|$(set +o | tr '\\n' ';')"`, + `source '${LIB}'`, + `after="$-|$(set +o | tr '\\n' ';')"`, + `[ "$before" = "$after" ] && echo SAME || { echo "DIFFERENT"; echo "before=$before"; echo "after=$after"; }`, + ].join("\n"), + ], + { encoding: "utf-8" }, + ); + expect(r.stdout).toContain("SAME"); + expect(r.status).toBe(0); + }); + } +}); + +// --------------------------------------------------------------------------- +// Static contract — the things a future edit must not quietly drop +// --------------------------------------------------------------------------- + +describe("session-runner.sh static contract", () => { + it("defines the five public ladder functions", () => { + for (const fn of [ + "runner_detect", + "runner_launch", + "runner_verify_cwd", + "runner_alive", + "runner_teardown", + ]) { + expect(LIB_SOURCE).toMatch(new RegExp(`^${fn}\\(\\)`, "m")); + } + }); + + it("sets no shell options at file scope", () => { + const offenders = LIB_SOURCE.split("\n").filter((l) => /^set\s/.test(l)); + expect(offenders).toEqual([]); + }); + + it("carries the caller-owns-shell-options contract in its header", () => { + expect(LIB_SOURCE).toContain( + "THE CALLER OWNS SHELL OPTIONS; THIS LIBRARY MUST NOT MUTATE THEM", + ); + }); + + it("contains none of the forbidden herdr commands", () => { + for (const forbidden of [ + "herdr server stop", + "herdr update", + "herdr channel set", + "~/.config/herdr", + ]) { + expect(LIB_SOURCE).not.toContain(forbidden); + } + }); + + it("uses `herdr agent get` (the liveness oracle) and `pane close` (the only teardown verb)", () => { + expect(LIB_SOURCE).toContain("herdr agent get"); + expect(LIB_SOURCE).toContain("herdr pane close"); + // 0.7.4 has no stop/kill verb; either would fail silently inside a trap. + expect(LIB_SOURCE).not.toMatch(/herdr agent (stop|kill)/); + }); + + it("pins the two literal herdr health fields and the 4h budget default", () => { + expect(LIB_SOURCE).toContain("status: running"); + expect(LIB_SOURCE).toContain("compatible: yes"); + expect(LIB_SOURCE).toContain("RUNNER_DEFAULT_TIMEOUT_MS=14400000"); + }); +}); diff --git a/.oh/scripts/firstmate.sh b/.oh/scripts/firstmate.sh new file mode 100755 index 00000000..1e35f2b5 --- /dev/null +++ b/.oh/scripts/firstmate.sh @@ -0,0 +1,529 @@ +#!/usr/bin/env bash +# .oh/scripts/firstmate.sh — the `firstmate` executor entrypoint. +# +# Usage: +# .oh/scripts/firstmate.sh [--runner herdr|tmux|foreground] [--harness claude|pi|codex] +# [--no-watch] +# .oh/scripts/firstmate.sh --kill +# +# Launches ONE long-lived First-Mate session over the whole `.oh/tasks//` +# task graph, where ralph launches 50 fresh processes each holding one story. +# The session manager is resolved by the shared ladder in +# `.oh/scripts/lib/session-runner.sh` (herdr -> tmux -> foreground); the slug and +# four-file validation come from `.oh/scripts/lib/task-contract.sh`; the session +# prompt is rendered from `.oh/skills/firstmate/templates/session-prompt.md`. +# +# `.oh/scripts/ralph.sh` is UNTOUCHED and stays the default executor. firstmate +# is opt-in, reached only via `--executor=firstmate`. +# +# --------------------------------------------------------------------------- +# Naming contract (PRD section 5) +# --------------------------------------------------------------------------- +# herdr agent firstmate- +# tmux session agent-firstmate- +# herdr log /tmp/firstmate-.log +# tmux log /tmp/agent-firstmate-.log +# lock /tmp/firstmate-.lock (atomic mkdir launch-claim) +# rendered prompt /tmp/firstmate-.prompt.md +# terminal the whole line `STATUS: COMPLETE` in progress.txt +# +# --------------------------------------------------------------------------- +# Configuration (env vars) +# --------------------------------------------------------------------------- +# FIRSTMATE_TIMEOUT_MS session budget in ms (default 14400000 = 4h). +# Validated ONLY by resolve_timeout_ms in +# session-runner.sh — 0, negative, non-numeric and +# empty are rejected there and the default applies. +# FIRSTMATE_HARNESS claude | pi | codex (default: claude) +# FIRSTMATE_CLAUDE_FLAGS default: --dangerously-skip-permissions --print +# FIRSTMATE_PI_FLAGS default: --print +# FIRSTMATE_HARNESS_CMD full override of the launched command; the rendered +# prompt path is exported as $FIRSTMATE_PROMPT_FILE +# FIRSTMATE_BRANCH override the placeholder +# FIRSTMATE_ISSUE override the placeholder +# OH_RUNNER runner override; same values as --runner +# RUNNER_TMPDIR root for logs/lock/prompt (default /tmp; tests only) +# +# --------------------------------------------------------------------------- +# Deliberate deviation from the PRD section 5 launch sketch +# --------------------------------------------------------------------------- +# The sketch shows ` "/goal "`. There is no `/goal` +# skill in this repo (`.oh/skills/` has no `goal` entry), and threading a +# multi-kilobyte prompt through an argv nested inside `bash -lc '...'` is a +# quoting hazard. The rendered prompt is therefore written to +# $FIRSTMATE_PROMPT_FILE and piped in on stdin — the shape ralph.sh:191 already +# proves (`printf '%s' "$task" | claude $flags`). The rendered prompt IS the +# goal brief, so no wrapper verb is lost. + +set -euo pipefail + +FIRSTMATE_SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" + +# shellcheck source=/dev/null +. "$FIRSTMATE_SCRIPT_DIR/lib/session-runner.sh" +# shellcheck source=/dev/null +. "$FIRSTMATE_SCRIPT_DIR/lib/task-contract.sh" + +# The skill-owned session-prompt template (US-002), repo-root-relative. +FIRSTMATE_TEMPLATE_REL=".oh/skills/firstmate/templates/session-prompt.md" + +# The CLOSED placeholder set US-002's contract header declares. `{curly}` text +# is NOT a placeholder — it is runtime-fill the session writes — so the renderer +# leaves it alone. +FIRSTMATE_PLACEHOLDERS="slug branch issue" + +usage() { + cat >&2 <<'EOF' +Usage: firstmate.sh [--runner herdr|tmux|foreground] [--harness claude|pi|codex] + [--no-watch] + firstmate.sh --kill +EOF +} + +# ─── argument parsing ──────────────────────────────────────────────── + +parse_args() { + KILL_MODE=0 + WATCH=1 + REQUESTED_RUNNER="${OH_RUNNER:-}" + HARNESS="${FIRSTMATE_HARNESS:-claude}" + SLUG="" + local positional=() + + while [ "$#" -gt 0 ]; do + case "$1" in + --kill) + KILL_MODE=1 + ;; + --kill=*) + KILL_MODE=1 + positional+=("${1#--kill=}") + ;; + --no-watch) + WATCH=0 + ;; + --runner) + shift + if [ "${1-}" = "" ]; then + echo "Error: --runner requires a value (herdr, tmux, or foreground)." >&2 + exit 2 + fi + REQUESTED_RUNNER="$1" + ;; + --runner=*) + REQUESTED_RUNNER="${1#--runner=}" + ;; + --harness) + shift + if [ "${1-}" = "" ]; then + echo "Error: --harness requires a value (claude, pi, or codex)." >&2 + exit 2 + fi + HARNESS="$1" + ;; + --harness=*) + HARNESS="${1#--harness=}" + ;; + -h | --help) + usage + exit 0 + ;; + --) + shift + while [ "$#" -gt 0 ]; do + positional+=("$1") + shift + done + break + ;; + -*) + echo "Error: unknown option '$1'." >&2 + usage + exit 2 + ;; + *) + positional+=("$1") + ;; + esac + shift + done + + if [ "${#positional[@]}" -ne 1 ]; then + usage + exit 2 + fi + SLUG="${positional[0]}" +} + +normalize_harness() { # + case "${1:-}" in + claude | pi | codex) + printf '%s\n' "$1" + ;; + *) + printf "Error: unknown harness '%s' (expected: claude, pi, or codex).\n" "${1:-}" >&2 + return 2 + ;; + esac +} + +# ─── paths ─────────────────────────────────────────────────────────── + +firstmate_repo_root() { + git rev-parse --show-toplevel 2>/dev/null || pwd +} + +# Where the rendered prompt lands. Honours RUNNER_TMPDIR for the same reason the +# log and lock paths do — so a test never writes into the real /tmp namespace. +firstmate_prompt_path() { # + printf '%s\n' "${RUNNER_TMPDIR:-/tmp}/firstmate-${1:-}.prompt.md" +} + +# ─── placeholder resolution ────────────────────────────────────────── + +firstmate_json_field() { # + local prd="${1:-}" filter="${2:-}" + [ -f "$prd" ] || return 0 + command -v jq >/dev/null 2>&1 || return 0 + jq -r "$filter" "$prd" 2>/dev/null || true +} + +firstmate_branch_name() { # + local prd="${1:-}" branch="" + if [ -n "${FIRSTMATE_BRANCH:-}" ]; then + printf '%s\n' "$FIRSTMATE_BRANCH" + return 0 + fi + branch="$(firstmate_json_field "$prd" '.branchName // empty')" + if [ -z "$branch" ] && [ -f "$prd" ]; then + # jq-free fallback: branchName is a flat top-level string. The producer is + # wrapped so that `head` closing the pipe early (SIGPIPE, status 141) cannot + # fail the whole pipeline under `set -o pipefail` and discard a real match. + branch="$( { sed -n 's/.*"branchName"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' "$prd" || true; } | head -n 1)" + fi + [ -n "$branch" ] || branch="$(git rev-parse --abbrev-ref HEAD 2>/dev/null || true)" + [ -n "$branch" ] || branch="unknown" + printf '%s\n' "$branch" +} + +# BARE DIGITS, per US-002's contract: the template writes `#` wherever a +# `#`-prefixed reference is wanted, so the renderer must never prepend one. +firstmate_issue_number() { # + local prd="${1:-}" issue="" + if [ -n "${FIRSTMATE_ISSUE:-}" ]; then + printf '%s\n' "${FIRSTMATE_ISSUE#\#}" + return 0 + fi + issue="$(firstmate_json_field "$prd" '(.issue // .issueNumber // empty) | tostring')" + if [ -z "$issue" ] && [ -f "$prd" ]; then + # /ship-spec records the issue in prose ("… (issue #746)"), not as a field. + # `|| true` absorbs both no-match (status 1) and SIGPIPE from `head`; under + # `set -o pipefail` either would otherwise sink the whole substitution. + issue="$( { grep -oE '#[0-9]+' "$prd" || true; } | head -n 1 | tr -d '#')" + fi + [ -n "$issue" ] || issue="unknown" + printf '%s\n' "$issue" +} + +# ─── the renderer (US-003 owns the renderer; US-002 owns the contract) ── + +# Substitutes the closed placeholder set declared in the template's own contract +# header into the template BODY, and writes the result to stdout. +# +# It introduces no token the template does not declare, and self-checks that no +# declared token survived — a surviving `` in a live session prompt is a +# silent, expensive failure. +render_session_prompt() { #