Compare commits

..
244 Commits
Author SHA1 Message Date
Timmy 69cf867969 Bump version to 0.14.5
Publish sled artifact / publish-dev-artifact (push) Canceled after 0s
2026-07-24 17:41:05 +01:00
Huskies Agent b43def12f7 huskies: regen source-map.json 2026-07-21 20:19:29 +00:00
Huskies Agent c3cad6b3d8 huskies: merge 1244 refactor Deduplicate the worktree module cluster (cleanup/remove/sweep/create/lifecycle share ~60-line clones) 2026-07-21 20:19:29 +00:00
Huskies Agent c58e204e8b huskies: regen source-map.json 2026-07-21 17:51:35 +00:00
Huskies Agent 8b2dd21e22 huskies: merge 1247 refactor Deduplicate chat-transport LLM command handling (discord <-> whatsapp 45-line clone) 2026-07-21 17:51:35 +00:00
Huskies Agent 326b2b4a32 huskies: regen source-map.json 2026-07-21 17:29:35 +00:00
Huskies Agent 10dd239c92 huskies: merge 1246 refactor Extract shared LLM-runtime logic duplicated between gemini and openai runtimes 2026-07-21 17:29:35 +00:00
Huskies Agent 1583ade9fc huskies: merge 1245 refactor Deduplicate validation/requests.rs — five internal ~50-85 line self-clones 2026-07-21 16:55:10 +00:00
Huskies Agent 5c3b433e7d huskies: merge 1251 story script/check gates duplication and cognitive complexity 2026-07-21 16:15:10 +00:00
Huskies Agent 14331127b8 huskies: regen source-map.json 2026-07-21 16:08:03 +00:00
Huskies Agent 7de8fa1924 huskies: merge 1249 story Checkpoint the CRDT snapshot periodically, not once by accident 2026-07-21 16:08:03 +00:00
Huskies Agent 82d9e3c460 huskies: regen source-map.json 2026-07-21 15:25:35 +00:00
Huskies Agent 013a5da60b huskies: merge 1243 refactor Merge jobs stop bloating replicated CRDT state 2026-07-21 15:25:34 +00:00
Huskies Agent 2ea633a2f1 huskies: merge 1242 refactor script/release builds project images so they can't drift 2026-07-21 15:10:39 +00:00
Huskies Agent b9e0a16bf8 huskies: regen source-map.json 2026-07-21 14:55:49 +00:00
Huskies Agent 78b2e7a9a2 huskies: merge 1250 bug Bug fields steps_to_reproduce, actual_result and expected_result cannot be edited after creation 2026-07-21 14:55:49 +00:00
Huskies Agent d06f5b5410 huskies: merge 1241 bug CRDT snapshot has no schema migration; a failed load silently starts empty 2026-07-21 14:47:40 +00:00
Timmy 40fb6367c1 Bump version to 0.14.4
Publish sled artifact / publish-dev-artifact (push) Canceled after 0s
2026-07-21 14:51:39 +01:00
Huskies Agent 1859e79491 huskies: regen source-map.json 2026-07-21 13:46:43 +00:00
Huskies Agent 5eeb036875 huskies: merge 1248 bug Duplicate Working line: obsolete digging-in watcher survives alongside 1240 placeholder 2026-07-21 13:46:43 +00:00
Timmy 1de05b480b Bump version to 0.14.3
Publish sled artifact / publish-dev-artifact (push) Canceled after 0s
2026-07-21 13:34:16 +01:00
Huskies Agent c906707a4f huskies: merge 1240 story Matrix bot shows live progress by editing its placeholder message 2026-07-21 12:17:42 +00:00
Huskies Agent afde373676 huskies: regen source-map.json 2026-07-21 12:10:20 +00:00
Huskies Agent 8a32c1ffd8 huskies: merge 1238 bug worktree tests fail on macOS: tests clobber global HOME 2026-07-21 12:10:20 +00:00
Huskies Agent 22710571df huskies: regen source-map.json 2026-07-21 11:42:44 +00:00
Huskies Agent 6fbd755846 huskies: merge 1239 story Matrix bot posts \"Working...\" immediately on message receipt 2026-07-21 11:42:44 +00:00
Timmy df1e339b47 Removing hardcoded huskies session 2026-07-21 11:29:57 +01:00
Timmy b55816d0a5 Bump version to 0.15.3
Publish sled artifact / publish-dev-artifact (push) Canceled after 0s
2026-07-20 17:14:24 +01:00
Huskies Agent eac57c6757 huskies: regen source-map.json 2026-07-20 14:44:32 +00:00
Huskies Agent c9804ecfdf huskies: merge 1236 story Ask "what happened with X" and get a paged, subject-scoped history 2026-07-20 14:44:32 +00:00
Huskies Agent 30ca3ad463 huskies: merge 1235 bug Notifications fire off the filesystem watcher, not the state machine — story 995's TransitionFired subscriber was never wired into startup 2026-07-20 14:36:49 +00:00
Huskies Agent f7b7f21e88 huskies: merge 1237 bug status resolves the story number against a different project than show 2026-07-20 14:30:23 +00:00
Huskies Agent 7ac5bd196f huskies: regen source-map.json 2026-07-19 21:30:50 +00:00
Huskies Agent 30b0d31500 huskies: merge 1230 story Sleds self-upgrade on startup to the gateway's published artifact if the baked binary is behind 2026-07-19 21:30:50 +00:00
Huskies Agent 2b9fcf2878 huskies: regen source-map.json 2026-07-19 20:47:19 +00:00
Huskies Agent 7bbd34bc3a huskies: merge 1231 story project-rebuild must not silently downgrade a live sled to the stale image binary 2026-07-19 20:47:19 +00:00
Huskies Agent 3f05648d25 huskies: regen source-map.json 2026-07-19 19:23:17 +00:00
Huskies Agent f4f0981f17 huskies: merge 1228 story Render agent questions as numbered options in chat protocols without question UI 2026-07-19 19:23:16 +00:00
Huskies Agent 933fb5a54b huskies: merge 1232 bug Gateway chat bot crashes (CLI exit 1) on a sled MCP tool error instead of surfacing it 2026-07-19 18:12:33 +00:00
Huskies Agent 7b3990430e huskies: regen source-map.json 2026-07-19 15:41:49 +00:00
Huskies Agent caf9953634 huskies: merge 1229 bug Explicit project arg ignored on the SSE MCP path — 1225's routing and create-guard are bypassed 2026-07-19 15:41:49 +00:00
Timmy c7cb3172f1 Ignoring some more files 2026-07-19 15:36:30 +01:00
Timmy 3198309db5 Bump version to 0.14.2
Publish sled artifact / publish-dev-artifact (push) Canceled after 0s
2026-07-19 00:31:16 +01:00
Huskies Agent 59481515b5 huskies: regen source-map.json 2026-07-18 19:56:30 +00:00
Huskies Agent a6cce683f5 huskies: merge 1222 bug show fails with "content unavailable" on done/archived stories — content should never be evicted 2026-07-18 19:56:30 +00:00
Huskies Agent 2b0e8e6f10 huskies: merge 1225 bug Work-item tools use active project as an implicit global; project arg ignored, creates/reads misfile silently 2026-07-18 19:20:46 +00:00
Timmy 34fe84fdd9 Bump version to 0.14.1
Publish sled artifact / publish-dev-artifact (push) Canceled after 0s
2026-07-18 19:08:17 +01:00
Huskies Agent a4e70af157 huskies: merge 1214 bug find_free_port can return a reserved/occupied port (2200), flaking its test and merge gates 2026-07-18 14:18:29 +00:00
Huskies Agent 15e9aa6b34 huskies: regen source-map.json 2026-07-18 13:58:35 +00:00
Huskies Agent 7d3de2bb44 huskies: merge 1218 story Remembered/sticky permission approvals so constrained agents stop re-prompting for the same action class 2026-07-18 13:58:35 +00:00
Huskies Agent fbaf5bf959 huskies: regen source-map.json 2026-07-18 13:50:49 +00:00
Huskies Agent 9c61dfa595 huskies: merge 1219 story Add runtime artifacts to the default project scaffold .gitignore 2026-07-18 13:50:49 +00:00
Huskies Agent 90cb005019 huskies: merge 1216 story Shorten long-turn notification to "Working..." 2026-07-18 12:43:46 +00:00
Huskies Agent 12580ae9ce huskies: merge 1215 story Pipeline board shows active-agent state distinctly from idle-blocked 2026-07-18 12:37:49 +00:00
Huskies Agent aa9306912d huskies: regen source-map.json
Publish sled artifact / publish-dev-artifact (push) Canceled after 0s
2026-07-18 11:46:52 +00:00
Huskies Agent cebe9e2737 huskies: merge 1213 story Chat "stop" command that immediately aborts the in-flight LLM turn 2026-07-18 11:46:52 +00:00
TimmyandClaude Opus 4.8 b91e2d53ff docs: refresh README release section
Update the stale 0.7.1 example to 0.14.0, note the Linux arm64 target (the sleds' arch), and mention the branch+tag push added in bcac9266.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_019fHdm92yjvguPi2LiXfLB9
2026-07-18 12:28:58 +01:00
Timmy fdfa4bac08 Bump version to 0.14.0 2026-07-18 12:20:02 +01:00
TimmyandClaude Opus 4.8 bcac92669a script/release: push branch and tag together so master doesn't lag the tag
Previously the script pushed only the tag, leaving origin/master behind the release by the version-bump commit (the tag pointed at an unpushed commit). Push the current branch and the tag atomically with --atomic.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_019fHdm92yjvguPi2LiXfLB9
2026-07-18 12:18:40 +01:00
Huskies Agent a43148ee13 huskies: regen source-map.json 2026-07-18 11:00:28 +00:00
Huskies Agent 71f3fa09c4 huskies: merge 1211 story Deterministic crash notification when the agent PTY dies mid-turn 2026-07-18 11:00:28 +00:00
Huskies Agent 153333d055 huskies: regen source-map.json 2026-07-18 10:45:40 +00:00
Huskies Agent b0f19eb0c5 huskies: merge 1212 story Sled chat messages show "workspace" instead of the real project name 2026-07-18 10:45:40 +00:00
Huskies Agent 36ec6d0f93 huskies: regen source-map.json 2026-07-18 10:12:31 +00:00
Huskies Agent 3f37a6cf34 huskies: merge 1210 story Gateway-side "digging in" notification for long tool-only turns 2026-07-18 10:12:31 +00:00
TimmyandClaude Opus 4.8 98f8825701 Fix compact no-op in gateway mode: add compact to GATEWAY_LOCAL_COMMANDS
Root cause of compact still failing after 1192/1205: in gateway mode,
on_room_message proxies any command not in GATEWAY_LOCAL_COMMANDS to the
active project's sled (proxy_bot_command) and returns — and that proxy
runs BEFORE the local compact interception (try_handle_compact_command).
'compact' was missing from the allowlist, so the gateway shipped it to a
sled (which has no compact command → no-op 'Command succeeded with no
response text'), and the real handler was never reached.

'reset' was already in the list, which is why reset worked in gateway
mode and compact did not. Add 'compact' as its sibling.

Why 1192 and 1205 both missed this: their tests drive
try_handle_compact_command directly, bypassing the gateway-proxy seam
that only exists on the full on_room_message path in gateway mode.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_019fHdm92yjvguPi2LiXfLB9
2026-07-18 09:37:05 +01:00
Huskies Agent 4ab34d3be2 huskies: regen source-map.json 2026-07-18 02:21:05 +00:00
Huskies Agent 8a7bff71aa huskies: merge 1209 story Gateway lifecycle &amp; telemetry MCP: gateway_info, restart_gateway, gateway_logs, start_story, chat_telemetry 2026-07-18 02:21:05 +00:00
Huskies Agent 3b10b29ef5 huskies: regen source-map.json 2026-07-18 02:05:03 +00:00
Huskies Agent 883f15045b huskies: merge 1204 story Slim CRDT-backed pipeline_query MCP tool (project / stages / fields / include_archived) 2026-07-18 02:05:03 +00:00
Huskies Agent cd2b417962 huskies: regen source-map.json 2026-07-18 01:54:05 +00:00
Huskies Agent 405d29d933 huskies: merge 1208 story Ops/LLM sessions can reach gateway-mode + cross-project MCP (the biggest shell-fallback cause) 2026-07-18 01:54:05 +00:00
Huskies Agent bb16f915f3 huskies: regen source-map.json 2026-07-18 01:43:24 +00:00
Huskies Agent 25bc952dff huskies: merge 1207 story fleet_resources MCP tool: on-demand host + per-container disk, load, and CPU/mem 2026-07-18 01:43:24 +00:00
Huskies Agent 5243854129 huskies: merge 1205 bug compact STILL swallowed after 1192 — interception is placed AFTER the registry dispatch, not before 2026-07-18 01:23:52 +00:00
Huskies Agent 486146ab34 huskies: regen source-map.json 2026-07-18 01:14:14 +00:00
Huskies Agent fba9b09d3e huskies: merge 1206 story fleet_identity MCP tool: read sled pins vs live signed identity, and re-pin via TOFU 2026-07-18 01:14:14 +00:00
Huskies Agent 8d2ad6424b huskies: regen source-map.json 2026-07-17 23:23:52 +00:00
Huskies Agent 75c4a8de33 huskies: merge 1203 bug Gateway status gives no response — 1187 made it proxy-only with no local resolution or error surfacing 2026-07-17 23:23:52 +00:00
Huskies Agent fd83ff2f53 huskies: merge 1202 bug Flaky test: matrix pull pull_rejects_tampered_artifact_sha256_mismatch fails nondeterministically in merge gates 2026-07-17 21:45:53 +00:00
Huskies Agent 38bdfa2ab4 huskies: regen source-map.json 2026-07-17 20:57:21 +00:00
Huskies Agent 5b340e7b20 huskies: merge 1199 story Orphaned build-dir GC: reclaim dead worktree targets without touching warm caches 2026-07-17 20:57:21 +00:00
Huskies Agent 2bd41d980c huskies: regen source-map.json 2026-07-17 19:43:27 +00:00
Huskies Agent 80efb7fcfc huskies: merge 1201 story Chat notification when a new work item is filed 2026-07-17 19:43:27 +00:00
Huskies Agent b9af302baf huskies: regen source-map.json 2026-07-17 19:25:52 +00:00
Huskies Agent c1523e8acf huskies: merge 1200 story Low-disk warning: the fleet tells the operator before the disk takes it down 2026-07-17 19:25:52 +00:00
Huskies Agent 82865956d2 huskies: merge 1195 bug Chat show renders metadata from stale content-text YAML instead of CRDT registers 2026-07-17 19:04:56 +00:00
Huskies Agent ecbed641bb huskies: regen source-map.json 2026-07-17 18:31:36 +00:00
Huskies Agent db27d0dbf3 huskies: merge 1198 bug Failed agents are never reaped — a dead pool entry blocks respawn indefinitely 2026-07-17 18:31:36 +00:00
Huskies Agent 2335fc0bbb huskies: merge 1194 bug db shadow-table tests flake when SHADOW_DB_PATH is not initialized 2026-07-17 18:06:34 +00:00
Huskies Agent ba1617934a huskies: merge 1190 bug Startup version announcement only fires on the trampoline path — normal restarts still say just "Timmy is online." 2026-07-17 17:42:23 +00:00
Huskies Agent c91ebb810c huskies: regen source-map.json 2026-07-17 17:33:55 +00:00
Huskies Agent b9d130bf64 huskies: merge 1193 story overview chat command: active work across all connected sleds 2026-07-17 17:33:55 +00:00
Huskies Agent 9db8a5006a huskies: merge 1196 bug Inactivity watchdog kills agents mid tool-call: awaited MCP calls produce no PTY output 2026-07-17 16:21:51 +00:00
Huskies Agent d4dde5d436 huskies: regen source-map.json 2026-07-17 14:24:56 +00:00
Huskies Agent a2b7b62960 huskies: merge 1191 bug /identity node_id field reports the CRDT id, not the node_identity.key the signature uses 2026-07-17 14:24:55 +00:00
Huskies Agent d77399aced huskies: regen source-map.json 2026-07-17 13:46:11 +00:00
Huskies Agent 6174828e18 huskies: merge 1192 bug compact command is swallowed by the registry placeholder before its real handler runs 2026-07-17 13:46:11 +00:00
Huskies Agent e991c6ac63 huskies: regen source-map.json 2026-07-17 13:32:30 +00:00
Huskies Agent bb5d7879ff huskies: merge 1189 story Gitea Actions workflow: build the sled artifact on master merges and publish to the dev channel 2026-07-17 13:32:30 +00:00
TimmyandClaude Fable 5 3ed3fdd6b0 Fold GatesFailed auto-retry into the block subscriber's shared budget
Code review of 1185 (merge 0f4b0c95) found the retry subscriber's central
invariant did not hold: the MergeFailure->Merge bounce caused by its own
retry reset both its attempt counter and the block subscriber's counter,
so the shared merge_failure_block_threshold budget was unreachable and a
deterministic gates failure retried forever.

One subscriber now owns one counter driving both policies:

- Counter survives PipelineEvent::MergeRetryStarted bounces (finding 1);
  a third consecutive failure blocks even with retries in between.
- Mixed failure kinds share the single budget (finding 5).
- Retries respect recovery: no counting or scheduling while a mergemaster
  is active, and perform_auto_retry re-checks before firing (finding 2).
- perform_auto_retry applies the same eligibility gates as
  assign_merge_stage (review hold, frozen, blocked, unmet deps) so freeze
  now stops a retry loop (finding 4).
- Per-story scheduling generations invalidate stale sleeping timers
  (finding 6).
- One-shot startup scan schedules a catch-up retry for stories already
  parked in GatesFailed, so restarts no longer strand them (finding 3);
  kept out of the periodic reconciler to avoid re-retrying exhausted
  stories every tick.
- Chat is notified only after the merge actually starts; a failed trigger
  logs instead of claiming a retry ran (finding 7).
- Config reads moved onto spawn_blocking (finding 8, bug 1170 class).

Deletes merge_failure_retry_subscriber.rs; notification plumbing
(WatcherEvent::MergeAutoRetry et al) is unchanged.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_019fHdm92yjvguPi2LiXfLB9
2026-07-17 13:36:45 +01:00
Huskies Agent c717ae7041 huskies: regen source-map.json 2026-07-17 12:11:27 +00:00
Huskies Agent 0f4b0c9536 huskies: merge 1185 story Bounded auto-retry for GatesFailed merges before requiring human intervention 2026-07-17 12:11:27 +00:00
Huskies Agent 5d672f5bf6 huskies: regen source-map.json 2026-07-17 12:05:04 +00:00
Huskies Agent 043c77f077 huskies: merge 1186 story compact chat command: distill session context deterministically, then reset with a seed 2026-07-17 12:05:04 +00:00
Huskies Agent b241661941 huskies: regen source-map.json 2026-07-17 11:58:37 +00:00
Huskies Agent dac0278218 huskies: merge 1188 story Merge tool results return a summary, not the full gate log (35KB per call) 2026-07-17 11:58:37 +00:00
Huskies Agent ea8e6edc94 huskies: regen source-map.json 2026-07-17 11:49:43 +00:00
Huskies Agent eda14976d0 huskies: merge 1187 story status chat command shows that project's status 2026-07-17 11:49:43 +00:00
Huskies Agent 0d26ac5a2a huskies: regen source-map.json 2026-07-17 11:39:47 +00:00
Huskies Agent 39dd6e0151 huskies: merge 1184 bug Stage transitions never reach the gateway relay: StatusEvent::StageTransition is published only in tests 2026-07-17 11:39:47 +00:00
Huskies Agent 1d71877af6 huskies: merge 1182 bug Sled uplinks all register as 'workspace' — containers never get HUSKIES_PROJECT_NAME 2026-07-17 10:46:22 +00:00
Huskies Agent ae47dd29e8 huskies: merge 1183 story Config parse error hints at misplaced top-level keys after [[component]] 2026-07-17 10:35:40 +00:00
Huskies Agent faa7825799 huskies: merge 1181 refactor Dependency freshness sweep: bump manifest minimums, refresh lockfile, drop dead serde_yaml 2026-07-16 19:07:01 +00:00
Huskies Agent 97ed97885a huskies: regen source-map.json 2026-07-16 18:10:14 +00:00
Huskies Agent de7d22cb9f huskies: merge 1180 story Sled↔gateway goes WS-only: remove the deprecated HTTP fallback paths 2026-07-16 18:10:14 +00:00
Timmy 0e684fb06f Upgrading Docker to include newer node, so front-end builds 2026-07-16 19:03:48 +01:00
Huskies Agent e0ed35adf1 huskies: regen source-map.json 2026-07-16 17:20:20 +00:00
Huskies Agent aebe75cd04 huskies: merge 1179 story Token-authenticated sled→gateway WS uplink (remote-gateway ready) 2026-07-16 17:20:20 +00:00
Huskies Agent 0738005863 huskies: regen source-map.json 2026-07-16 16:22:46 +00:00
Huskies Agent efd0097f76 huskies: merge 1176 bug base_branch fallback hardcodes master instead of auto-detecting 2026-07-16 16:22:46 +00:00
Huskies Agent a4e2d4bc25 huskies: merge 1177 bug project.toml scaffold puts top-level keys after [[component]] so uncommenting them silently no-ops 2026-07-16 16:13:17 +00:00
TimmyandClaude Fable 5 0eacffa08d Install Node 22 from NodeSource in project base image
Bookworm's apt nodejs is 18.x; frontend toolchains after the 2026-07-15
dependency upgrades (vite 7) require Node >= 20, so any sled building a
frontend via build.rs failed. NodeSource nodejs bundles npm, so the
separate apt npm package is dropped.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_019fHdm92yjvguPi2LiXfLB9
2026-07-16 16:39:51 +01:00
Huskies Agent e5df6232cf huskies: merge 1178 bug MergeFailureFinal is a trap state: successful re-merge cannot mark the story done 2026-07-16 14:19:55 +00:00
Huskies Agent 61acf98909 huskies: regen source-map.json 2026-07-16 14:03:31 +00:00
Huskies Agent 77e0394195 huskies: merge 1169 story Gateway pulls signed artifacts from a release channel into its local store 2026-07-16 14:03:31 +00:00
Huskies Agent 1e0e581bd7 huskies: regen source-map.json 2026-07-16 13:26:03 +00:00
Huskies Agent 0ac68afa4c huskies: merge 1173 story Identity-aware fleet checks: cryptographic node identity in upgrade and health probes 2026-07-16 13:26:03 +00:00
Huskies Agent 489c415fd9 huskies: regen source-map.json 2026-07-16 13:19:20 +00:00
Huskies Agent 18b065f77a huskies: merge 1163 story Replace perm_rx lock-as-presence-signal with a permission router 2026-07-16 13:19:20 +00:00
Huskies Agent 6f8a8ffd87 huskies: merge 1175 bug Flaky tests intermittently fail merge gates 2026-07-16 13:03:05 +00:00
TimmyandClaude Fable 5 6a1ee8377d docs: update stale libsqlite3-sys pin comment for sqlx 0.9 stable
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-16 11:46:26 +01:00
Huskies Agent a464febdcb huskies: merge 1174 story Gateway startup announcement reports version and model 2026-07-16 10:11:55 +00:00
Huskies Agent 8ae2eaaad3 huskies: merge 1160 bug Chat bot crash-loops on poisoned Claude Code session resume 2026-07-16 09:51:44 +00:00
Huskies Agent 23f2934e1f huskies: merge 1172 story Add Docker log rotation to project container launch args 2026-07-16 09:44:48 +00:00
Huskies Agent 69df921856 huskies: merge 1170 bug Full tokio runtime stall after unblock → merge auto-assign 2026-07-16 09:10:46 +00:00
TimmyandClaude Fable 5 f73689cfd8 Fix CRDT self-deadlock: read_llm_session re-locked the state mutex
read_llm_session acquired the CRDT_STATE mutex, then called
extract_llm_session_view while holding the guard — which called
our_node_id(), which locks the same non-reentrant std::sync::Mutex.
The thread deadlocks itself and parks forever HOLDING the lock; every
other CRDT user then queues behind it. With light traffic that's a
partial wedge (MCP `show`/content reads hang while /health stays
green); during a CRDT-write burst (unblock → auto-assign) enough
tasks pile up to pin every tokio worker: liveness heartbeat stops,
/health dies, full sled freeze. Root cause of bug 1170's repeated
sled freezes, confirmed by live gdb capture: thread parked in
lock_contended at presence::our_node_id ← read_llm_session ←
event_matches_persona, with all other threads queued on CRDT reads.

Fix: extract_llm_session_view now takes local_sled_id as a parameter;
read_llm_session computes it from the guard it already holds. The
trigger path (event_matches_persona on persona-subscribed WS events)
explains the raciness — it needs a chat/persona event racing a
pipeline transition.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_019fHdm92yjvguPi2LiXfLB9
2026-07-16 09:46:36 +01:00
Huskies Agent 1a15347b02 huskies: merge 1165 refactor Upgrade matrix-sdk 0.17 to 0.18 2026-07-16 05:14:46 +00:00
Huskies Agent a4bf11dbc1 huskies: merge 1166 refactor Upgrade ed25519-dalek 2 to 3 2026-07-16 01:08:34 +00:00
Huskies Agent 182771192b huskies: merge 1168 refactor Upgrade tokio-tungstenite 0.29 to 0.30 2026-07-16 00:53:47 +00:00
Huskies Agent 0a53c1ff0a huskies: merge 1167 refactor Upgrade garde 0.22 to 0.23 2026-07-16 00:38:34 +00:00
Huskies Agent ca05bd7224 huskies: merge 1164 refactor Lockfile refresh + sqlx alpha-to-stable 2026-07-16 00:13:38 +00:00
Huskies Agent aa977482db huskies: regen source-map.json 2026-07-16 00:03:44 +00:00
Huskies Agent e67eff17ad huskies: merge 1158 story Reject boilerplate user stories at save time 2026-07-16 00:03:44 +00:00
TimmyandClaude Fable 5 6cceec9c26 Disable CRDT debug logging in default features — fixes runtime stalls
logging-list/logging-json were in bft-json-crdt's default feature set,
so every production build printed multi-KB debug dumps on every CRDT
op — executed INSIDE the global CRDT_STATE mutex. A stdout write that
stalls while holding that lock blocks every task touching the CRDT
(tick loop, watchers, MCP, RPC), each one pinning an OS worker thread
until the tokio pool is exhausted: liveness heartbeat stops, /health
dies, zero CPU. This is the mechanism behind bug 1170 (two full-sled
freezes on huskies-server, both seconds after a CRDT write burst, the
second insert's dump truncated mid-print in the log).

The features remain available for CRDT debugging via explicit opt-in.

Also fixes a latent race in persist_tx_send_success_emits_no_warn:
it counted [crdt_persist] warns in the process-global log buffer,
which parallel tests also write to; the debug prints had been acting
as an accidental serializer. Now filters for its own story id.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_019fHdm92yjvguPi2LiXfLB9
2026-07-15 18:37:47 +01:00
TimmyandClaude Fable 5 be0c88c801 Pre-check binary writability before accepting an upgrade request
/api/upgrade now verifies the target can be replaced (create + remove
the swap's tmp file) before returning 202. A sled that cannot write
its own binary — e.g. a container predating the /opt/huskies/bin
layout — fails phase 1 of `upgrade all` loudly instead of returning
202, staying healthy, and silently remaining on the old version, which
is exactly what happened on the first fleet deploy.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_019fHdm92yjvguPi2LiXfLB9
2026-07-15 17:53:56 +01:00
TimmyandClaude Fable 5 ba0a38d403 Fix WS deadlock: never block on perm_rx in the chat handler
Since story 884 the Matrix permission listener (and the sled uplink,
when configured) hold services.perm_rx for the process lifetime. The
WS chat handler's blocking `lock().await` on that same mutex therefore
parked the entire WS connection loop forever on StartChat: chat_fut was
never polled, RPC frames on the socket were never answered, and
everything queued behind the dispatcher's serial session lock —
wedging /mcp and /rpc while /health stayed green.

Use try_lock instead: if another task already owns permission routing,
run the chat without the local permission-forwarding select arm (a
pending future keeps the select shape unchanged).

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_019fHdm92yjvguPi2LiXfLB9
2026-07-15 17:17:24 +01:00
TimmyandClaude Fable 5 01b24ff2ae Drop the drain check from upgrade — agent death is routine
Agents die all the time; the pipeline's retry machinery re-queues
their work. Skipping busy sleds just created version skew and manual
retries for no real protection. `upgrade all` now sweeps every sled
unconditionally.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_019fHdm92yjvguPi2LiXfLB9
2026-07-15 16:54:55 +01:00
TimmyandClaude Fable 5 f39c4b7c4b Remove all alternate update paths — fleet redeploy is release + upgrade all
Killed:
- rebuild_and_restart (in-container cargo self-compile): the MCP tool,
  the `rebuild` chat command in all four transports, the web-ui bot
  command, and the underlying function. This was the path that caused
  the exec() deadlocks.
- upgrade_sled gateway MCP tool: second entry point to sled upgrades,
  defaulted to serving the gateway's own macOS binary to Linux sleds.
- GET /api/huskies-binary (both sled and gateway route trees): served
  current_exe(), wrong platform when the gateway is macOS. Superseded
  by /api/artifacts/ which now also serves on the gateway route tree.
- `huskies upgrade` CLI subcommand and --source flag: third way of
  doing the same download-and-replace. Escape hatch for a bricked sled
  is `docker cp` + restart.

Kept, distinct jobs: `project-rebuild` (container/image updates),
`rebuild gateway` + script/local-release (gateway self-update).

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_019fHdm92yjvguPi2LiXfLB9
2026-07-15 16:46:11 +01:00
TimmyandClaude Fable 5 83f941b77e Revert /app/target binary fallback in entrypoint
Superseded: the running fleet was created by `new project` (no /app
mount), and the one-way upgrade path now replaces the binary at its
canonical install location (/opt/huskies/bin/huskies) directly, so the
entrypoint needs no fallback logic.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_019fHdm92yjvguPi2LiXfLB9
2026-07-15 16:28:32 +01:00
TimmyandClaude Fable 5 e8253a06b7 Add release chat command — build sled binary and publish artifact
Finds the registered project carrying the huskies source tree, runs
`cargo build --release` inside its container (dedicated
CARGO_TARGET_DIR=target/sled-release so container builds stop
clobbering host target/release), then atomically publishes the binary
to ~/.huskies/artifacts/ with a .hash sidecar for convergence checks.

Full fleet redeploy is now chat-only: `release` then `upgrade all` —
no laptop access needed.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_019fHdm92yjvguPi2LiXfLB9
2026-07-15 16:27:32 +01:00
TimmyandClaude Fable 5 18ba57a7b0 Add upgrade all chat command with drain check and convergence verify
- `upgrade all` sweeps every registered sled in sequence, streaming
  per-sled phase markers and reporting a summary.
- Binary source is now the gateway's own artifact store via
  host.docker.internal (was: unresolvable `gateway` hostname serving
  the gateway's macOS binary to Linux sleds — would have bricked them).
- Sleds with active claude processes are skipped, never killed.
- After reconnect, /api/version git_hash is compared against the
  published artifact's .hash sidecar; divergence is reported loudly.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_019fHdm92yjvguPi2LiXfLB9
2026-07-15 16:22:11 +01:00
TimmyandClaude Fable 5 07b9e1605d Serve sled binary artifacts from ~/.huskies/artifacts/
GET /api/artifacts/:filename with filename validation (no path
components, no dotfiles). Sleds only ever download binaries from their
own gateway; this endpoint is where the gateway serves them from,
replacing the current_exe()-based /api/huskies-binary which serves the
gateway's own (macOS) binary — wrong platform for Linux sleds.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_019fHdm92yjvguPi2LiXfLB9
2026-07-15 16:12:00 +01:00
TimmyandClaude Fable 5 4d22171d16 Install sled binary in huskies-owned dir so upgrades work without root
/opt/huskies/bin/huskies (chowned to the huskies user) with a symlink
from /usr/local/bin/huskies. Atomic replace needs write permission on
the directory for the tmp-write + rename, which root-owned
/usr/local/bin can't provide to the server process.

resolve_target_path() now prefers /opt/huskies/bin/huskies over
current_exe(), which can point at a stale location (e.g.
/workspace/target/release/huskies after a historical in-container
rebuild) that the entrypoint would never launch after a restart.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_019fHdm92yjvguPi2LiXfLB9
2026-07-15 16:06:31 +01:00
TimmyandClaude Fable 5 a1ae532c6e Add /api/version endpoint for upgrade convergence checks
Reports crate version + compile-time BUILD_GIT_HASH as JSON. The
gateway will poll this after `upgrade all` to verify each sled is
actually running the published artifact.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_019fHdm92yjvguPi2LiXfLB9
2026-07-15 16:03:49 +01:00
Timmy 2a7db16a77 Fixed exec bug 2026-07-15 14:55:27 +01:00
TimmyandClaude Opus 4.6 549e0349d7 Fix rebuild_and_restart in Docker project containers
CARGO_MANIFEST_DIR is baked at image build time as /app/server, but
project containers (Dockerfile.base) don't copy /app — the source is
bind-mounted at /workspace instead. Fall back to project_root when
the compile-time path doesn't exist.

Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
2026-07-14 18:42:55 +01:00
Huskies Agent 16d4294f08 huskies: regen source-map.json 2026-06-29 19:38:13 +00:00
Huskies Agent c733fb2bf9 huskies: merge 1162 refactor Update workspace Cargo.toml dependencies to latest compatible versions 2026-06-29 19:38:13 +00:00
Timmy 4c965c73b2 Added CRDT snapshotting 2026-06-29 20:11:52 +01:00
Timmy 3342d129c2 Unreachable container times out gracefully 2026-06-29 19:22:47 +01:00
Timmy 32646c6256 Catching a docker rm problem 2026-06-29 18:27:50 +01:00
Timmy 0a0ab65908 Adding doc comments. 2026-06-29 17:02:17 +01:00
Timmy feb35ddd10 Converted all external tool calling to async 2026-06-29 16:59:54 +01:00
Timmy 75f41088b1 Dealing with bot failures 2026-06-29 15:34:14 +01:00
Timmy b662a7da95 Stack detection fix 2026-06-29 13:14:06 +01:00
Timmy 146205c83b Added docs comment 2026-06-29 12:45:25 +01:00
Timmy 705f5bcc89 Adding show story mcp 2026-06-29 12:42:45 +01:00
dave 8285a98f80 huskies: regen source-map.json 2026-05-20 01:18:59 +00:00
dave 2fb935e726 huskies: merge 1156 story Periodic liveness tick so runtime freezes have a precise timestamp 2026-05-20 01:18:59 +00:00
dave 7be3bf1dbf huskies: regen source-map.json 2026-05-20 00:42:25 +00:00
dave 2a5359051e huskies: merge 1154 story Extend gateway_health with a relay-working signal — is each sled actually delivering events? 2026-05-20 00:42:24 +00:00
dave a3ac09f8a3 huskies: regen source-map.json 2026-05-20 00:23:44 +00:00
dave 846b3e1b4c huskies: merge 1153 story huskies projects chat command — list every registered project with port and status 2026-05-20 00:23:44 +00:00
dave 0c207981e9 huskies: regen source-map.json 2026-05-19 23:54:05 +00:00
dave e7456d3391 huskies: merge 1155 story Bracket logging around install_pre_commit_hook to diagnose bug 1151 freezes 2026-05-19 23:54:05 +00:00
Timmy 5bca1f6cec Bump version to 0.13.0 2026-05-20 00:00:16 +01:00
TimmyandClaude Opus 4.7 86b9d069b1 script/local-release: restore build + hot-restart workflow
1145 narrowed local-release to install-only (binary + codesign-heal
wrapper) and removed the cargo build + gateway hot-restart steps that
the script used to do. That broke the "rebuild the gateway" muscle
memory: running script/local-release no longer rebuilt or restarted
anything, just re-installed the same binary.

Restore the build + restart logic while keeping 1145's wrapper:

- `cargo build --release --bin huskies` before install
- Snapshot the prior binary to ~/bin/huskies-bin.prev for rollback
- Print PREV → NEW version delta after install
- Detect a running `huskies .*--gateway` process and SSH-safe-restart
  it (kill descendants depth-first, then nohup the wrapper from the
  detached subshell)
- Wait up to 10s for the new gateway PID to appear; on timeout, roll
  back to the previous binary and try to relaunch it
- Refuse to restart when more than one --gateway process matches, so
  we don't kill the wrong tree
- `--skip-check` bypasses script/check for already-verified changes

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-05-19 22:46:28 +01:00
dave f6ee90e169 huskies: regen source-map.json 2026-05-19 20:11:55 +00:00
dave 9a286315a3 huskies: merge 1149 story huskies health chat command — surface gateway, sled, matrix, creds, and build-hash status 2026-05-19 20:11:55 +00:00
dave 5d0801854c huskies: merge 1146 story Matrix bot auto-recovers from M_UNKNOWN_TOKEN by re-logging in from bot.toml password 2026-05-19 19:40:53 +00:00
dave 343473bc01 huskies: regen source-map.json 2026-05-19 18:39:40 +00:00
dave 2593b36072 huskies: merge 1148 story Per-sled upgrade chat command using huskies upgrade (1138), serial-locked 2026-05-19 18:39:40 +00:00
dave 34af2f1820 huskies: regen source-map.json 2026-05-19 18:34:41 +00:00
dave be7bdf8304 huskies: merge 1147 story One-active-gateway invariant via pidfile+flock — prevent double-gateway during restarts 2026-05-19 18:34:41 +00:00
dave 918f18c200 huskies: merge 1151 bug install_pre_commit_hook blocks the tokio executor — sync std::process::Command::output() in an async path stalls worktree-create-sub 2026-05-19 18:19:58 +00:00
dave 1db5473f50 huskies: regen source-map.json 2026-05-19 18:13:26 +00:00
dave de638603cd huskies: merge 1144 story Gateway trampoline-restart: detached helper survives the gateway's own death 2026-05-19 18:13:26 +00:00
dave 20ec690e22 huskies: regen source-map.json 2026-05-19 17:55:38 +00:00
dave 9a5b6f4d92 huskies: merge 1152 story Set HUSKIES_GATEWAY_URL on every sled container so 1136's relay actually spawns 2026-05-19 17:55:37 +00:00
dave 398726a14a huskies: merge 1145 story Codesign self-heal at exec time so a missed re-sign doesn't silently SIGKILL the binary 2026-05-19 17:49:57 +00:00
dave c8be24f833 huskies: regen source-map.json 2026-05-18 16:57:58 +00:00
dave f8ff63af0e huskies: merge 1142 story Force coder agents through MCP-validated Edit/Write/Bash to prevent writes to master worktree 2026-05-18 16:57:58 +00:00
dave 34e78bdbd5 huskies: regen source-map.json 2026-05-18 16:52:45 +00:00
dave fb4e52dd09 huskies: merge 1143 story Decouple LLM environmental awareness from chat transport — persona-keyed sessions and a real-time event subscription 2026-05-18 16:52:45 +00:00
dave e58ff4465a huskies: regen source-map.json 2026-05-18 14:55:31 +00:00
dave b1dec36e1c huskies: merge 1140 story One-shot project-rebuild chat command: rebuild image, swap container, reconnect, preserve state 2026-05-18 14:55:31 +00:00
dave 4aaf7dbdc6 huskies: regen source-map.json 2026-05-18 14:50:00 +00:00
dave 95c0aafb68 huskies: merge 1141 story Convert work-item type between spike/story/bug/refactor (or at least spike→story) 2026-05-18 14:50:00 +00:00
dave 5062e008c6 huskies: regen source-map.json 2026-05-18 13:54:44 +00:00
dave 55badc1e08 huskies: merge 1139 story Per-project Dockerfile fragment so agents can extend their own sled image 2026-05-18 13:54:44 +00:00
dave bdc621fb36 huskies: regen source-map.json 2026-05-18 13:33:50 +00:00
dave 0ec5c05de8 huskies: merge 1138 story In-container huskies self-update — huskies upgrade pulls a fresh binary without docker rebuild 2026-05-18 13:33:50 +00:00
dave d10634c7d6 huskies: regen source-map.json 2026-05-18 12:59:11 +00:00
dave a7bad217eb huskies: merge 1137 story First-run project init flow — walk through config instead of leaving defaults silently 2026-05-18 12:59:11 +00:00
dave f2c13c7d29 huskies: merge 1136 story Sled → gateway WebSocket back-channel so project pipeline events reach Timmy 2026-05-18 12:29:50 +00:00
dave 3444ff4e29 huskies: merge 1135 story Bootstrap Claude credentials into newly-launched project sleds 2026-05-18 12:06:32 +00:00
dave 26f4da7ba5 huskies: merge 1134 story mkdir -p ~/.huskies/&lt;name&gt;/ before ssh-keygen in adopt 2026-05-18 11:53:31 +00:00
TimmyandClaude Opus 4.7 4c6b4f5d4d fix: project sleds need claude CLI + extensions.worktreeConfig
Two issues that surfaced when story 1 ran in the adopted huskies-server
sled:

1. Dockerfile.base: the base image had no nodejs / claude CLI, so every
   coder agent spawn in an adopted project sled failed with
   `Unable to spawn claude: No viable candidates found in PATH`.  Install
   nodejs + @anthropic-ai/claude-code in the base image so every sled
   built from it can spawn agents out of the box.

2. worktree/create.rs::install_pre_commit_hook: `git config --worktree`
   requires `extensions.worktreeConfig = true` to be set on the repo
   config; without it, every worktree creation logged a noisy
   `Pre-commit hook install failed` warning.  Enable the extension
   idempotently before the per-worktree hooks-path set so the hook
   install succeeds cleanly.

After this, rebuild huskies-project-base and recreate any adopted
project containers to pick up the CLI.

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-05-18 08:40:21 +01:00
dave 70797753df huskies: merge 1132 story Chat-bot proxy reads stale gateway_project_urls BTreeMap instead of live store (1122 missed this seam) 2026-05-18 00:02:37 +00:00
Timmy ec3216072d Revert "fix: bind project container host ports to 0.0.0.0"
This reverts commit 810c8d4d72.
2026-05-18 00:28:34 +01:00
TimmyandClaude Opus 4.7 810c8d4d72 fix: bind project container host ports to 0.0.0.0
Story 1130 added HUSKIES_HOST=0.0.0.0 so the server INSIDE a project
container binds to all interfaces, but the host-side `docker -p`
mapping was still `127.0.0.1:{port}:3001` and `127.0.0.1:{ssh_port}:22`
— reachable from the docker host only, blocking remote MCP clients
and out-of-host SSH onto the project container.

Switch host-side mapping to 0.0.0.0 for both the MCP and SSH ports so
project containers spawned via `new project` are reachable from
anywhere that can route to the docker host. Existing containers
created before this commit retain their localhost-only mapping and
need to be recreated to pick up the change.

Add a regression test asserting both -p arguments use 0.0.0.0 and
reject any 127.0.0.1 restriction in the mapping.

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-05-18 00:04:32 +01:00
TimmyandClaude Opus 4.7 ce688fc0bf fix: drop package-lock.json + node_modules before npm install in Dockerfile
Previous attempt (c1318964) used npm ci + npm install --include=optional
--no-save, which still missed rolldown's platform-specific native
binding (@rolldown/binding-linux-arm64-gnu) — the runtime build still
fails with `Cannot find native binding`.

Wipe both the lockfile and node_modules so npm install resolves the
dependency tree fresh for the build platform.  The lockfile mutation
stays inside the container image.

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-05-17 23:47:43 +01:00
TimmyandClaude Opus 4.7 c131896432 fix: work around npm optional-deps bug in frontend npm install
`npm ci` alone hits npm/cli#4828: optional platform-specific bindings
(e.g. @rolldown/binding-linux-arm64-gnu introduced by 1119's vite 5→8
upgrade) listed in package-lock.json for the lockfile author's
platform are not fetched for the build platform.  The sled rebuild
fails with `Cannot find native binding`.

Follow `npm ci` with `npm install --include=optional --no-save` so the
build platform's native binding is fetched without mutating the
lockfile.

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-05-17 23:46:55 +01:00
Timmy 42e6eec9e9 Bump version to 0.12.1 2026-05-17 23:46:50 +01:00
dave fe00fe6a25 huskies: merge 1127 story Migrate all LLM-invoking transports onto assemble_prompt_context; delete legacy Vec 2026-05-17 22:28:01 +00:00
dave c97b7c841f huskies: regen source-map.json 2026-05-17 21:02:08 +00:00
dave 2d0387fe63 huskies: merge 1126 story Gateway event aggregator with per-session scope filters (Timmy=All, Sally=single sled) 2026-05-17 21:02:08 +00:00
dave 71d3047ef0 huskies: regen source-map.json 2026-05-17 20:30:02 +00:00
dave d86cc38b2a huskies: merge 1128 story Bounded event queues + EventStreamGap sentinel + observability for context assembly 2026-05-17 20:30:02 +00:00
dave 21b2efd268 huskies: regen source-map.json 2026-05-17 20:09:33 +00:00
dave badd522d60 huskies: merge 1125 story LLM session entity + assemble_prompt_context helper, wired into Matrix bot 2026-05-17 20:09:33 +00:00
dave ecd3f600d9 huskies: merge 1130 story Adopted/launched project containers bind huskies to 127.0.0.1, unreachable from host MCP 2026-05-17 20:02:22 +00:00
TimmyandClaude Opus 4.7 099df17e77 chore: gitignore /pipeline.db at repo root (phantom stale file)
A 0-byte pipeline.db sometimes appears at the repo root, left over
from old code paths. Current master correctly opens it at
.huskies/pipeline.db via project_root.join() in
server/src/startup/project.rs:280 — no relative-path opener exists.
This is purely defensive so any future regression doesn't sneak into
commits. Stops 1123 from being a coder task.

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-05-17 20:51:48 +01:00
dave c88e42eba2 huskies: regen source-map.json 2026-05-17 19:37:50 +00:00
dave 89058ebd49 huskies: merge 1124 story Persist TransitionFired into a per-sled CRDT event log 2026-05-17 19:37:50 +00:00
dave d8204ab7ed huskies: merge 1129 story find_free_port fallback returns unbindable port silently when range is exhausted 2026-05-17 19:24:29 +00:00
dave e2ea1af4c8 huskies: merge 1120 story Silence intentional-error stderr in frontend tests so failures stand out 2026-05-17 19:19:08 +00:00
dave 08780475d0 huskies: merge 1119 story Address npm audit moderate+ vulnerabilities in frontend/ 2026-05-17 19:00:55 +00:00
dave 6eb2742e7d huskies: regen source-map.json 2026-05-17 18:49:58 +00:00
dave c1b7e12b0b huskies: merge 1122 story Chat-bot switch command reads stale gateway_projects Vec instead of live gateway_projects_store 2026-05-17 18:49:58 +00:00
dave 53d44ff42a huskies: regen source-map.json 2026-05-17 18:43:43 +00:00
dave 6331dea8b0 huskies: merge 1121 story Remove the marketing website from the huskies OSS repo (now lives in huskies-server) 2026-05-17 18:43:43 +00:00
dave 240beec7de huskies: regen source-map.json 2026-05-17 17:48:44 +00:00
dave 7de167b21b huskies: merge 1116 story rebuild_and_restart loses pending CRDT ops by calling exec() before persistence channel drains 2026-05-17 17:48:44 +00:00
TimmyandClaude Opus 4.7 49af014a84 fix: build frontend before cargo in script/test (merge gate self-heal)
Story 1113 added `#[derive(RustEmbed)] #[folder = "../frontend/dist"]`
plus a unit test that calls `EmbeddedAssets::iter()`.  The macro only
generates `iter()` when the folder exists at compile time, so the Rust
build now has a hard compile-time dependency on `frontend/dist/`.

`script/test` ran `cargo clippy` (line 48) before the frontend build
(line 53+).  In a fresh merge worktree with no `frontend/dist/`, clippy
failed immediately on the `iter()` call and the script exited before
`npm run build` ever ran — the gate could never self-heal.  Blocked
1116's merge today; would block every future merge.

Move the frontend build above all cargo invocations.  Verified by
running script/test in a fresh worktree with `node_modules` and
`frontend/dist` removed: 385/385 frontend tests + cargo tests pass.

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-05-17 18:40:24 +01:00
dave 73cf1c6ff9 huskies: merge 1117 story MCP tool for adopt: expose new project --adopt as an MCP call 2026-05-17 16:42:06 +00:00
dave f8b1e14b74 huskies: merge 1118 story Automate per-project docker image builds (huskies-project-base + per-stack overlays) 2026-05-17 16:30:08 +00:00
TimmyandClaude Opus 4.7 265e6f9a15 fix(1101): strip passing-test lines before classify() lint check; remove diagnostic
The merge gate classifier was matching trigger keywords like
`missing_doc_comments` inside passing-test name lines
(e.g. `test agents::gates::tests::classify_lint_from_missing_doc_comments ... ok`),
causing every gate failure to be mis-classified as Lint and bounced
back to a fixup coder. Strip `test … … ok` lines before scanning for
lint triggers. Also removes the temporary diagnostic block in
runner.rs that confirmed the bug.

Applied directly to master because the 1101 feature branch carried
stale work from an earlier incarnation of the story that semantically
conflicted with master's later diagnostic commit (`is_fixup` deleted
on the branch, referenced on master).

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-05-17 16:52:26 +01:00
dave 40e995da88 huskies: regen source-map.json 2026-05-17 15:51:38 +00:00
dave 6e4fb7fd4b huskies: merge 1113 story [huskies-server repo] Convert static website to Next.js with static rendering 2026-05-17 15:51:37 +00:00
dave 0695ad7ae6 huskies: merge 1115 story new project: --adopt flow to wrap a container around an existing checkout 2026-05-17 15:17:12 +00:00
dave eb6b07531a huskies: merge 1114 story new project: --path flag to override default host directory 2026-05-17 14:48:49 +00:00
dave 2d6846fe03 huskies: merge 1112 story Remove static website from huskies OSS repo (moved to huskies-server) 2026-05-17 14:43:46 +00:00
330 changed files with 37349 additions and 11937 deletions
+53
View File
@@ -0,0 +1,53 @@
# Gitea Actions workflows
## `release-artifact.yml`
Triggers on every push to `master`. Builds the `linux-arm64` sled binary and
publishes it to the "dev" release channel via `script/ci-publish-artifact`.
### Runner registration
The job targets the `arm64-mac` runner label. Register an `act_runner` on an
Apple Silicon macOS host that has `cargo`/`rustc` and `curl` on `PATH`:
```sh
act_runner register \
--instance https://code.crashlabs.io \
--token <runner-registration-token> \
--labels arm64-mac
act_runner daemon
```
The registration token comes from the repo's **Settings → Actions →
Runners → Create new Runner** page in Gitea. Without a runner carrying the
`arm64-mac` label, jobs from this workflow queue indefinitely.
### Secrets
Configure these under the repo's **Settings → Actions → Secrets**. Never
commit credentials — the workflow only ever references them via
`${{ secrets.* }}`.
| Secret | Purpose |
| --- | --- |
| `HUSKIES_CHANNEL_URL` | Base URL of the dev release channel host. |
| `HUSKIES_CHANNEL_TOKEN` | Bearer token authorised to publish artifacts to that channel. |
### Channel host contract
`script/ci-publish-artifact` expects the channel host at
`HUSKIES_CHANNEL_URL` to implement:
- `POST {HUSKIES_CHANNEL_URL}/<artifact-name>` — accepts the raw artifact
bytes as the request body. Requires `Authorization: Bearer <token>` and
`X-Git-Hash: <short-git-hash>` headers. Non-2xx responses in the 4xx range
(including 401/403) are treated as permanent failures; 5xx responses and
network errors are retried with backoff.
- `GET {HUSKIES_CHANNEL_URL}/manifest.json` — returns a JSON object with a
`git_hash` field reflecting the most recently published artifact.
Requires `Authorization: Bearer <token>`.
This is a separate, unsigned channel distinct from the Ed25519-signed
release channels the `pull <channel>` gateway command consumes (see
`server/src/service/gateway/release_manifest.rs`) — CI has no safe place to
hold a channel signing key, so the dev channel trusts the bearer token alone.
+33
View File
@@ -0,0 +1,33 @@
name: Publish sled artifact
on:
push:
branches:
- master
jobs:
publish-dev-artifact:
# Registered on an arm64 macOS act_runner host — see
# .gitea/workflows/README.md for registration instructions.
runs-on: [arm64-mac]
steps:
- name: Checkout
uses: actions/checkout@v4
- name: Build linux-arm64 sled binary
env:
# Dedicated target dir so this CI build never clobbers a developer's
# incremental target/release build on the shared runner host, mirroring
# SLED_TARGET_DIR in server/src/chat/transport/matrix/release.rs.
CARGO_TARGET_DIR: target/ci-release
run: cargo build --release -p huskies
- name: Stage artifact
run: cp target/ci-release/release/huskies target/ci-release/release/huskies-linux-arm64
- name: Publish to dev channel
env:
HUSKIES_CHANNEL_URL: ${{ secrets.HUSKIES_CHANNEL_URL }}
HUSKIES_CHANNEL_TOKEN: ${{ secrets.HUSKIES_CHANNEL_TOKEN }}
run: script/ci-publish-artifact target/ci-release/release/huskies-linux-arm64 "$(git rev-parse --short HEAD)"
+2 -3
View File
@@ -6,15 +6,14 @@
# Local environment (secrets)
.env
# Local-only scripts
script/local-release
# App specific (root-level; huskies subdirectory patterns live in .huskies/.gitignore)
store.json
_merge_parsed.json
.huskies_port
.huskies/bot.toml.bak
.huskies/build_hash
# Phantom 0-byte pipeline.db sometimes appears at repo root from old code; canonical DB lives at .huskies/pipeline.db
/pipeline.db
# Per-worktree planning file (written by coder agents, must never reach squash commits)
PLAN.md
+7
View File
@@ -35,3 +35,10 @@ double_timmy_log.md
pipeline.db
pipeline.db.bak*
session_store.json
# Full untruncated merge attempt reports (one file per attempt, pruned automatically)
merge_reports/
logs/
wizard_state.json
store.json
*.db
+1 -1
View File
@@ -56,7 +56,7 @@ There are no exceptions. The merge gate runs `source-map-check` and rejects the
Before committing, run `cargo run -p source-map-gen --bin source-map-check -- --worktree . --base master` and address every missing-docs direction it prints. If you added a new module file (e.g. `foo.rs` or `foo/mod.rs`), the FIRST line of that file MUST be a `//! What this module is for` doc comment.
## Documentation
Docs live in `website/docs/*.html` (static HTML), **not** Markdown files. When a story asks you to document something, edit the relevant `.html` file in `website/docs/`.
Docs live in `website/app/docs/*.tsx` (Next.js pages), **not** Markdown files. When a story asks you to document something, edit the relevant `.tsx` file under `website/app/docs/`. Run `npm run build` in `website/` to verify your changes render correctly.
## Configuration files
- Agent config: `.huskies/agents.toml` (preferred) or `[[agent]]` blocks in `.huskies/project.toml`
+478 -56
View File
File diff suppressed because it is too large Load Diff
@@ -113,9 +113,27 @@ Layered:
- **`huskies-project-base`**: debian-slim + git + huskies binary + sshd
+ sudo + a `huskies` user with the SSH pubkey installed.
- **`huskies-stack-<stack>`**: per-stack additions. E.g. rust gets
`rustup` + `rust-analyzer` + `cargo-nextest`; node gets `node@22` +
`typescript-language-server`; etc.
- **`huskies-project-<stack>`**: per-stack additions, pre-built by
`script/build-project-images`. E.g. rust gets `rustup` +
`rust-analyzer` + `cargo-nextest`; node gets `node@22` +
`typescript-language-server`; etc. Stack fragments live in
`docker/stacks/<stack>/Dockerfile.fragment`.
- **`huskies-project-local-<name>`** *(optional)*: built on the fly at
container launch time when the project contains
`.huskies/Dockerfile.fragment`. This file is appended after the
stack overlay (`FROM huskies-project-<stack>`) so agents can extend
their own image without editing shared stack files. Because the
fragment lives inside the bind-mounted `/workspace/.huskies/`, changes
survive container recreation and are committed alongside the project
source. The `project-rebuild` command picks up the fragment
automatically when rebuilding.
Example `.huskies/Dockerfile.fragment` that adds `jq`:
```dockerfile
RUN apt-get update && apt-get install -y jq
```
- **Project layer**: the bind-mounted `/workspace` is the project source,
written by the host's editor, read by the in-container tooling.
@@ -215,7 +233,74 @@ The work breaks naturally into:
- **Phase 4:** git integration — `--git <url>` clones, host SSH key
mount, push verification.
- **Phase 5:** per-project resource limits + cleanup chat commands.
- **Phase 6:** `--adopt <dir>` wraps a container around an existing
checkout. No clone or init — bind-mount only.
- **Phase 7 (story 1137):** First-run init flow — config summary and
chat-driven overrides (see below).
Each phase ships independently and is usable on its own. Phase 1 alone
gives chat-only users a working project; later phases add the editor
and git polish.
## First-Run Init Flow (Story 1137)
After a successful `new project ... --adopt` (or any new-project
bootstrap), the bot appends a **Default configuration** block to the
adoption success reply. This block lists every scaffolded agent with
its model, budget cap, and turn limit, and provides ready-to-send
override commands.
### Example reply tail
```
**Default configuration** (3 agents):
- coder-1 (coder): model=`sonnet`, budget=$5.00, max_turns=50
- qa (qa): model=`sonnet`, budget=$4.00, max_turns=40
- mergemaster (mergemaster): model=`sonnet`, budget=$5.00, max_turns=30
Override via chat: `huskies config myapp coder.model=opus`
Project settings: `huskies config myapp default_qa=human`
Accept all defaults silently: add `--skip-config` to the bootstrap command.
```
### Config override command
```
huskies config <project> <key>=<value>
```
The gateway resolves the project's `host_path` from `projects.toml`,
then writes the setting to `.huskies/agents.toml` or
`.huskies/project.toml` on the host.
**Agent fields** (`<stage_or_name>.<field>=<value>`):
| Key | Target | Supported values |
|-----|--------|-----------------|
| `coder.model` | agents.toml, coder stage | `sonnet`, `opus`, any model string |
| `qa.model` | agents.toml, qa stage | same |
| `mergemaster.model` | agents.toml, mergemaster stage | same |
| `coder.max_turns` | agents.toml, coder stage | integer |
| `coder.max_budget` | agents.toml, coder stage | decimal (USD) |
**Project keys** (bare `<key>=<value>`):
| Key | Notes |
|-----|-------|
| `default_qa` | `"server"`, `"agent"`, or `"human"` |
| `max_retries` | integer |
| `max_coders` | integer |
| `base_branch` | branch name string |
| `timezone` | IANA timezone (e.g. `"Europe/London"`) |
| `default_coder_model` | model string |
### Skip path
Pass `--skip-config` to suppress the config block entirely:
```
new project myapp --adopt /path/to/checkout --skip-config
```
The success reply is identical to pre-1137 output — only the SSH
command and registration summary, no agent listing.
@@ -0,0 +1,46 @@
# CRDT Snapshot Compaction
## Problem
The huskies project CRDT has grown to 55K ops / 276MB in `pipeline.db`.
Every container restart replays all operations in a tight synchronous loop
on the tokio runtime (`crdt_state/state/init.rs:68-78`), taking 27+ minutes
and freezing the runtime so the HTTP server never becomes ready.
### Op bloat
55K ops across only 4,154 sequence numbers (~13 ops per seq on average).
Many zeroed-out `MergeJobCrdt` entries appear to be tombstones never cleaned
up. Some individual ops are up to 523KB. Average op size is 4.6KB.
## Proposed Fix
### 1. Snapshot (checkpoint)
After replaying all ops, serialize the materialized CRDT state to a
checkpoint blob (e.g. a `crdt_snapshot` table or a separate file). On next
startup, load the snapshot and only replay ops with `rowid > snapshot_rowid`.
At snapshot time, back up the database file so corruption is recoverable.
### 2. Op pruning / compaction
Delete ops that are superseded by the snapshot. Tombstoned/deleted items with
all-zero fields contribute nothing to materialized state and can be dropped
from the log once snapshotted.
### 3. Immediate fix: spawn_blocking
Move the replay loop to `tokio::task::spawn_blocking` so the HTTP server and
liveness ticks are not starved during replay. This does not reduce replay
time but prevents the runtime freeze.
## Acceptance Criteria
- Startup loads a snapshot when available and only replays ops newer than the snapshot sequence
- A snapshot is written after full CRDT replay completes (or periodically in background)
- DB backup is created at each snapshot time
- Startup time for 55K ops drops from 27+ minutes to under 30 seconds
- Dead/zeroed CRDT ops are pruned during compaction
- CRDT sync protocol continues to work correctly across nodes after compaction
- The replay loop runs in spawn_blocking so it does not freeze the tokio runtime
@@ -0,0 +1,127 @@
# Story 1208: Cross-Project MCP for Ops/LLM Sessions
## 1. Problem Statement
An ops/LLM session connects to a `huskies --gateway` instance's `/mcp`
endpoint. Before this story, the *only* way to act on a specific registered
project was:
1. Call `switch_project` (mutates the gateway's shared, global
`GatewayState.active_project`), then
2. Call the ordinary project-level tool (`create_story`, `get_story_todos`,
`show`, …), which the gateway silently proxies to whichever project is
currently active.
This has two problems:
- **Race condition**: `active_project` is one value shared by every
connected client. Two concurrent ops sessions targeting different projects
will step on each other's `switch_project` calls.
- **No true "read a named project once" path**: for a single lookup against
a project that isn't the current default, a caller had to mutate shared
state just to read something, then (optionally) switch back.
The practical consequence (and the reason this story exists) is that
operators and LLM agents fall back to hand-crafting raw JSON-RPC requests
directly against a project's own container port, bypassing the gateway
entirely — the "shell-fallback" this story is named for.
## 2. Chosen Mechanism: Per-Call `project` Argument
Any `tools/call` request for a non-gateway (proxied) tool may now include an
optional top-level `project` field inside `arguments`:
```json
{
"jsonrpc": "2.0",
"id": 1,
"method": "tools/call",
"params": {
"name": "create_story",
"arguments": {
"name": "Fix login bug",
"acceptance_criteria": ["..."],
"origin": "...",
"project": "robot-studio"
}
}
}
```
- If `project` is present and non-empty, the gateway looks it up in
`projects.toml` (`GatewayState.projects`) and proxies the call directly to
that project's live sled-uplink WebSocket connection —
`GatewayState::proxy_mcp_for_project` in
`server/src/service/gateway/mod.rs`. `GatewayState.active_project` is
**not read or mutated** by this path.
- If `project` is absent (the common case, and all pre-existing behavior),
the call proxies to whichever project is currently active, exactly as
before — full backward compatibility with existing sessions and
`switch_project`-based workflows.
- An unknown project name returns a JSON-RPC `-32602` (invalid params)
error listing the registered project names. A known project with no live
WS-uplink connection returns `-32603` naming the sled, matching the
existing `active_project` proxy error shape.
Implementation: `server/src/http/gateway/mcp.rs`
(`gateway_mcp_post_handler`'s `tools/call` branch,
`proxy_and_respond_for_project`) and
`server/src/service/gateway/mod.rs` (`GatewayState::sled_connection_for`,
`GatewayState::proxy_mcp_for_project`, generalized from the existing
`active_sled_connection` / `proxy_active_mcp`).
### Schema discoverability
`tools/list` merges gateway tools with the active project's own tool list.
Every merged (proxied) tool's `inputSchema.properties` gets a `project`
property injected (`inject_project_arg_schema` in `http/gateway/mcp.rs`) so
MCP clients that validate call arguments against the declared schema before
sending don't strip or reject the extra field. This is additive only — no
existing property, and no `required` list, is touched.
### Why not mirror every tool at the gateway level?
Rejected alternative: define a `project_create_story`, `project_show`, etc.
for every project-level tool at the gateway. This was rejected because it
duplicates ~15+ tool schemas and dispatch arms and drifts out of sync every
time a project-level tool's schema changes. A single per-call argument that
every proxied tool call can carry scales to new project-level tools for
free.
## 3. Fleet-Wide Reads (AC 2)
These already existed as gateway-level tools before this story and needed
no code change — listed here for completeness of the "how an ops session
connects" picture:
| Tool | Purpose |
|------|---------|
| `list_projects` | Every registered project: name, url, ssh_port, host_path, adopted/built-in marker, active marker. No liveness check. |
| `gateway_health` | Per-project health (WS heartbeat or HTTP poll) plus CRDT event-relay staleness. |
| `aggregate_pipeline_status` | Pipeline stage counts and blocked/failing items across every registered project, fetched in parallel. |
| `fleet_identity` | (Story 1206) Per-sled identity pin vs. live signed identity, and TOFU re-pin. |
## 4. How an Ops Session Should Connect
1. Point the MCP client at the gateway's `/mcp` endpoint
(`http://<gateway-host>:<port>/mcp`), the same endpoint local agents use
— there is no separate "ops" endpoint.
2. Call `tools/list` to see the merged tool surface (gateway tools + the
active project's tools, each carrying the optional `project` schema
property).
3. For a one-off call against a specific project, pass `project: "<name>"`
inside `arguments` on that call — no `switch_project` required, and no
risk of racing another session's active-project selection.
4. For fleet-wide questions (is anything down, what's blocked everywhere),
use `list_projects`, `gateway_health`, or `aggregate_pipeline_status`
directly; they already scan every registered project.
5. `switch_project` remains available for sessions that want a persistent
default (e.g. an interactive chat session working one project at a
time) — it is unaffected by this change.
## 5. Design Review Note (AC 4)
This document captures the chosen approach (per-call `project` argument,
generalized proxy functions, additive schema injection) as required by AC 4.
No new gateway-level tool surface was added for AC 1 — the existing proxy
path was extended instead, minimizing new schema/dispatch surface area and
keeping every future project-level tool automatically cross-project-capable.
@@ -0,0 +1,245 @@
# LLM Context From Events
Design overview for making any LLM-driven chat persona (Timmy at the
gateway, Sally at a single sled, future personas) aware of huskies
events without the user having to re-narrate them.
## Goal
**Update the LLM's context non-intrusively when a state transition
happens.** No new LLM turn is fired; events are simply visible to the
LLM the next time the user (or anything else) does cause it to run.
The LLM should never need to be told what already happened inside
huskies.
## Guiding Principle
**Transports have nothing to do with LLMs.** A transport (Matrix bot,
web UI, CLI, future TUIs) is a pure courier — it relays user text in,
LLM text out, and never owns LLM-facing state. Anything the LLM needs
to know lives in huskies, behind a single `assemble_prompt_context`
helper that the transport calls. Adding a new transport must require
zero changes to the event-awareness path.
## Three things this doc is NOT
1. **Triggers**`on StoryMerged{1122} do Rebuild`. These are
deterministic subscribers; they should never invoke the LLM. Covered
in a separate design.
2. **Proactive wake** — running an LLM turn *because* an event fired,
without the user typing. Costs tokens, risks ramble. Explicitly out
of scope here; a separate decision to make later.
3. **A transport feature** — this design assumes any transport that
invokes the LLM uses the same context-assembly helper. Matrix bot,
web UI, CLI all funnel through it.
## Why Past Attempts Have Failed
- **Buffer lived on the transport**, not on huskies. The current
`BotContext.pending_pipeline_events` (`server/src/chat/transport/matrix/bot/context.rs:103-116`)
is Matrix-only; web UI users see nothing of the kind, and the buffer
dies with the bot process.
- **Process-local, RAM-only**. Server rebuild → buffer empty. Any
events between the old binary's last user turn and the new binary's
first are silently lost.
- **Unbounded `mpsc` channels drop under lag.** The server logs
routinely show `[xxx-sub] Subscriber lagged, skipped N event(s)`.
When the subscriber feeding the buffer falls behind, events vanish
without being recorded.
- **No end-to-end test.** Nothing asserts "fire event E, send user
message M, the LLM's prompt contains E."
- **No cross-process aggregation.** Events in a sled have no path to
the gateway-side LLM context without bespoke plumbing per event type.
## Architecture at a Glance
```
┌────────────┐ ┌────────────┐ ┌────────────┐
│ Sled A │ │ Sled B │ │ Sled C │
│ event_log/ │ │ event_log/ │ │ event_log/ │ ◄── source of truth
└─────┬──────┘ └─────┬──────┘ └─────┬──────┘ (CRDT-backed)
│ │ │
└───────────────┼───────────────┘
┌────────────────────┐
│ Gateway aggregator │ ◄── tail-merges all sled logs
│ event_view/ │ into a single ordered stream
└─────────┬──────────┘
┌────────────────────────┐
│ Per-LLM-session state │ ◄── scope filter +
│ sessions/<id>/ │ high-water mark per stream
└─────────┬──────────────┘
┌────────────────────────┐
│ assemble_prompt_context│ ◄── single helper used by
│ (session_id) -> Str │ every transport before
└─────────┬──────────────┘ each LLM turn
┌──────────────┼──────────────┐
▼ ▼ ▼
Matrix bot Web UI CLI / TUI
```
## Event Model — Reuse What Already Exists
There is no need to invent a parallel event taxonomy. Huskies already
has a complete typed enum and a single broadcast bus:
- `server/src/pipeline_state/transition.rs` defines `PipelineEvent`
with **30 variants** covering every state-machine transition
(`DepsMet`, `GatesStarted/Passed/Failed`, `QaSkipped`,
`MergeSucceeded/Failed/FailedFinal`, `Accepted`, `Block/Unblock`,
`Abandon`, `Supersede`, `ReviewHold/Cleared`, `Reject`, `Triage`,
`Close`, `Demote`, `Freeze/Unfreeze`, `MergemasterAttempted`,
`FixupRequested`, `ReQueuedForQa`, `MergeAborted`,
`HotfixRequested`, `MergeRetryStarted`).
- The same module defines `ExecutionEvent` with 7 variants for agent
lifecycle (`SpawnRequested`, `SpawnedSuccessfully`, `Heartbeat`,
`HitRateLimit`, `Exited`, `Stopped`, `Reset`).
- Every transition fires a `TransitionFired` event on a single internal
bus. Ten subscribers already consume it (audit-log,
worktree-create-sub, worktree-cleanup-sub, merge-failure-sub,
merge-block-sub, done-archive-sub, content-gc, cost-rollup-sub,
stage-notification-sub, event-triggers).
**The LLM context injector is just the 11th subscriber on the same
bus.** It writes typed events into the per-sled CRDT event log
described below; everything downstream reuses the existing taxonomy.
Each persisted entry carries:
```
struct LoggedEvent {
id: EventId, // monotonic per sled
sled_id: SledId,
timestamp: UnixSeconds,
transition: TransitionFired, // story_id + from + to + PipelineEvent
// (or ExecutionEvent — see open question)
}
```
The few events that genuinely don't fit the pipeline state machine
(e.g. `ProjectAdopted`, `Rebuilt`, `GatewayHealthChanged`) live in a
small, separately-enumerated `InfraEvent` enum, but the same log and
the same subscriber pattern still apply.
## Session Model
An LLM session is a first-class CRDT entity:
```
struct LlmSession {
id: SessionId,
persona: Persona, // "Timmy", "Sally", ...
scope: ScopeFilter, // { sleds: All } | { sleds: Set<SledId> }
high_water: BTreeMap<SledId, EventId>, // per-stream
created: UnixSeconds,
}
```
The session id is what the transport carries; it's not the Matrix room,
not the web socket id. A given Matrix room may map to one session; a
web UI tab may map to another. Multiple transports for the same human
can share a session if you want — that's a separate UX call.
## Prompt Assembly Contract
Every transport calls one helper before invoking the LLM:
```
fn assemble_prompt_context(session_id: SessionId) -> String
```
Behavior:
1. Read the session's scope filter and high-water marks.
2. Fetch events from the gateway aggregator that match the scope and
are newer than the high-water marks.
3. Render them as a single `<system-reminder>` block, ordered by sled
then timestamp.
4. Advance the high-water marks to the latest event seen, atomically
with the LLM-turn-start CRDT op (so a crash mid-turn doesn't double-
inject).
5. Return the rendered block (empty string if no new events).
The transport prepends the result to the user's prompt and invokes the
LLM as usual.
## Persistence & Reliability Rules
- **Event log is CRDT-backed.** Survives sled restart.
- **High-water marks are CRDT-backed.** Survives gateway restart.
- **Aggregator uses bounded queues with drop-oldest semantics**, and
every drop logs `[event-agg] dropped N events for session <id>; client
must re-fetch from <high-water>`. The aggregator never silently
swallows events — if the queue is full, the session gets a sentinel
event `EventStreamGap { from, to }` so the LLM can see it missed
context.
- **End-to-end test required**: `fire(Event::StoryMerged{1122}) → user
sends "what's going on?" → assembled prompt contains "1122 merged"`.
## Multi-Persona Scoping
The same machinery serves both Timmy and Sally:
| Persona | Scope filter | Notes |
|---------|------------------------------------|--------------------------------|
| Timmy | `{ sleds: All }` | Gateway-wide; aware of every sled |
| Sally | `{ sleds: { huskies-server } }` | Single-sled; sled-local events only |
| Manny | `{ sleds: { huskies, ketflix } }` | Hypothetical; subset |
Sally never has to know Timmy exists, and vice versa. Their sessions
advance their own high-water marks against the same underlying log.
## Decisions
| Decision | Choice | Alternative |
|------------------------|-------------------------------------|----------------------------------------------|
| Event publication | Each sled owns its log | Single global log: cross-sled bottleneck |
| Aggregation | Gateway tail-merges | Each session pulls from each sled directly: N×M fanout |
| Buffer location | CRDT-persisted | In-process: lost on rebuild (current bug) |
| Event identity | Typed enum | Strings: structured-log creep, no compile-time safety |
| Drop semantics | Drop-oldest + `EventStreamGap` | Silent drop (current bug): LLM lies confidently |
| Session ↔ transport | Session is separate from transport | One per transport: web tab + Matrix get different views |
| Proactive LLM wake | OUT OF SCOPE | Wake on every event: cost + ramble |
## Open Questions
1. **Session lifecycle**. How are sessions created and garbage-
collected? Created on first transport message? GC'd after N days
idle?
2. **Event retention**. How long are events kept in the log? Forever
feels wrong; "since last terminal session turn" feels right but
needs care for multi-session readers.
3. **Multi-transport same session**. Should one human's Matrix and web
UI share a session by default, or always be separate?
4. **Render budget**. If 500 events accumulated between turns, do we
render all 500 or summarize? A `summarize_events` fallback path is
probably worth designing in from the start.
5. **Aggregator placement when there is no gateway**. A standalone
single-sled install has no gateway — does the sled itself host the
aggregator? (Probably yes; trivially "aggregates" its own log.)
## Phasing
- **Phase 0 (now):** this design doc.
- **Phase 1:** typed `Event` enum + per-sled CRDT-backed event log;
one publisher subscribes to existing pipeline transitions and writes
`StoryStaged` / `StoryMerged` / `StoryMergeFailed`.
- **Phase 2:** `LlmSession` CRDT entity + `assemble_prompt_context`
helper, wired into the Matrix bot's `handle_message` (replaces the
existing `pending_pipeline_events` Vec). End-to-end test covering the
fire-event → user-turn → prompt-contains-event contract.
- **Phase 3:** Gateway aggregator over multiple sleds; Timmy's session
scoped to `All`. Sally's session scoped to a single sled.
- **Phase 4:** Web UI and any other transports migrated onto
`assemble_prompt_context`; the Matrix-specific Vec deleted.
- **Phase 5:** Bounded queues + `EventStreamGap` sentinel; observability
for `assemble_prompt_context` runs (events injected, gaps observed).
Each phase ships independently. Phase 2 alone delivers the user-facing
fix: Timmy sees what merged when you next say anything, without you
needing to re-narrate.
+12
View File
@@ -0,0 +1,12 @@
{
"threshold": 10,
"minLines": 10,
"minTokens": 50,
"ignore": [
"**/target/**",
"**/node_modules/**",
"**/dist/**",
"**/*.svg",
"**/flamegraphs/**"
]
}
Generated
+625 -1018
View File
File diff suppressed because it is too large Load Diff
+46 -36
View File
@@ -1,58 +1,68 @@
[workspace]
members = ["server", "crates/bft-json-crdt", "crates/source-map-gen"]
members = [
"server",
"crates/bft-json-crdt",
"crates/source-map-gen",
"crates/release-manifest",
"crates/release-tool",
]
resolver = "3"
[workspace.dependencies]
async-stream = "0.3"
async-stream = "0.3.6"
async-trait = "0.1.89"
bytes = "1"
chrono = { version = "0.4.44", features = ["serde"] }
chrono-tz = "0.10"
bytes = "1.12.1"
chrono = { version = "0.4.45", features = ["serde"] }
chrono-tz = "0.10.4"
eventsource-stream = "0.2.3"
futures = "0.3"
futures = "0.3.32"
homedir = "0.3.6"
ignore = "0.4.25"
mime_guess = "2"
ignore = "0.4.29"
mime_guess = "2.0.5"
notify = "8.2.0"
poem = { version = "3", features = ["websocket", "test"] }
poem = { version = "3.1.12", features = ["websocket", "test"] }
portable-pty = "0.9.0"
reqwest = { version = "0.13.3", features = ["json", "stream"] }
rust-embed = "8"
ed25519-dalek = { version = "2", default-features = false, features = ["rand_core"] }
reqwest = { version = "0.13.4", features = ["json", "stream"] }
rust-embed = "8.12.0"
ed25519-dalek = { version = "3.0.0", default-features = false, features = ["rand_core"] }
indexmap = { version = "2.14.0", features = ["serde"] }
rand = "0.10"
serde = { version = "1", features = ["derive"] }
serde_json = "1"
serde_urlencoded = "0.7"
sha1 = "0.11"
rand = "0.10.2"
serde = { version = "1.0.228", features = ["derive"] }
serde_json = "1.0.150"
serde_urlencoded = "0.7.1"
sha1 = "0.11.0"
sha2 = "0.11.0"
hmac = "0.13"
subtle = "2"
base64 = "0.22"
serde_yaml = "0.9"
strip-ansi-escapes = "0.2"
tempfile = "3"
tokio = { version = "1", features = ["rt-multi-thread", "macros", "sync"] }
toml = "1.1.2"
uuid = { version = "1.23.1", features = ["v4", "serde"] }
tokio-tungstenite = { version = "0.29.0", features = ["connect", "rustls-tls-native-roots"] }
hmac = "0.13.0"
subtle = "2.6.1"
base64 = "0.22.1"
strip-ansi-escapes = "0.2.1"
tempfile = "3.27.0"
tokio = { version = "1.52.4", features = ["rt-multi-thread", "macros", "sync"] }
toml = "1.1.3"
uuid = { version = "1.24.0", features = ["v4", "serde"] }
tokio-tungstenite = { version = "0.30.0", features = ["connect", "rustls-tls-native-roots"] }
walkdir = "2.5.0"
filetime = "0.2"
matrix-sdk = { version = "0.17", default-features = false, features = [
filetime = "0.2.29"
# 0.18 is current as of this pin (verified via `cargo search matrix-sdk`).
# matrix-sdk-sqlite 0.18 pulls in rusqlite 0.37, which requires libsqlite3-sys
# ==0.35.x — see the pin rationale in server/Cargo.toml. Bumping matrix-sdk
# past 0.18 may require re-checking that ceiling.
matrix-sdk = { version = "0.18", default-features = false, features = [
"sqlite",
"e2e-encryption",
] }
pulldown-cmark = { version = "0.13.3", default-features = false, features = [
pulldown-cmark = { version = "0.13.4", default-features = false, features = [
"html",
] }
regex = "1"
libc = "0.2"
nutype = { version = "0.7", features = ["serde"] }
garde = { version = "0.22", features = ["derive"] }
ammonia = "4.1"
sqlx = { version = "=0.9.0-alpha.1", default-features = false, features = [
regex = "1.13.1"
libc = "0.2.186"
nutype = { version = "0.7.0", features = ["serde"] }
garde = { version = "0.23", features = ["derive"] }
ammonia = "4.1.3"
sqlx = { version = "0.9.0", default-features = false, features = [
"runtime-tokio",
"sqlite",
"macros",
"migrate",
] }
serde_yaml = "0.9.34"
+6 -2
View File
@@ -79,6 +79,10 @@ cd frontend && npm install && npm run dev
Configuration lives in `.huskies/project.toml`. See `.huskies/bot.toml.*.example` for transport setup.
## Website
The huskies.dev website source has moved to [crashlabs/huskies-server](https://code.crashlabs.io/crashlabs/huskies-server).
## Architecture
Internal architecture documentation lives in [`docs/architecture/`](docs/architecture/):
@@ -91,10 +95,10 @@ Internal architecture documentation lives in [`docs/architecture/`](docs/archite
Requires a Gitea API token in `.env` (`GITEA_TOKEN=your_token`).
```bash
script/release 0.7.1
script/release 0.14.0
```
This bumps version in `Cargo.toml` and `package.json`, builds macOS arm64 and Linux amd64 binaries, tags the repo, and publishes a Gitea release with changelog and binaries attached.
This bumps version in `Cargo.toml` and `package.json`, builds macOS arm64, Linux amd64, and Linux arm64 binaries, tags the repo, pushes the branch and tag, and publishes a Gitea release with changelog and binaries attached.
## Multi-node CRDT sync (rendezvous)
+7
View File
@@ -0,0 +1,7 @@
# cognitive_complexity is allow-by-default in clippy; script/check enables it
# with `-W clippy::cognitive_complexity`. This threshold is set well above
# clippy's own default (25) to accommodate existing large dispatch functions
# (e.g. Matrix bot command routing) without requiring an unrelated refactor;
# it still gates against genuinely runaway complexity introduced going
# forward.
cognitive-complexity-threshold = 200
+11 -5
View File
@@ -6,8 +6,14 @@ edition = "2021"
[lib]
crate-type = ["lib"]
# The logging-* features print multi-KB debug dumps on every CRDT op — and
# they execute inside the global CRDT state mutex in the server, so a stalled
# stdout write while holding that lock can pin every tokio worker and freeze
# the whole process (bug 1170). They are development tools: opt in explicitly
# with `--features bft-json-crdt/logging-list` when debugging CRDT internals.
# Never enable them in a production build.
[features]
default = ["bft", "logging-list", "logging-json"]
default = ["bft"]
logging-list = ["logging-base"]
logging-json = ["logging-base"]
logging-base = []
@@ -15,18 +21,18 @@ bft = []
[dependencies]
bft-crdt-derive = { path = "bft-crdt-derive" }
colored = "3"
colored = "3.1.1"
ed25519-dalek = { workspace = true }
indexmap = { workspace = true, features = ["serde"] }
rand = { workspace = true }
random_color = "1"
random_color = "1.1.0"
serde = { workspace = true, features = ["derive"] }
serde_json = { workspace = true, features = ["preserve_order"] }
serde_with = "3"
serde_with = "3.21.0"
sha2 = { workspace = true }
[dev-dependencies]
criterion = { version = "0.8", features = ["html_reports"] }
criterion = { version = "0.8.2", features = ["html_reports"] }
serde = { workspace = true, features = ["derive"] }
serde_json = { workspace = true, features = ["preserve_order"] }
@@ -8,8 +8,8 @@ publish = false
proc-macro = true
[dependencies]
indexmap = { version = "2.2.6", features = ["serde"] }
proc-macro2 = "1.0.47"
proc-macro-crate = "3"
quote = "1.0.21"
syn = { version = "2", features = ["full"] }
indexmap = { version = "2.14.0", features = ["serde"] }
proc-macro2 = "1.0.106"
proc-macro-crate = "3.5.0"
quote = "1.0.46"
syn = { version = "2.0.119", features = ["full"] }
+5
View File
@@ -20,6 +20,10 @@ use std::{
/// An RGA-like list CRDT that can store a CRDT-like datatype
#[derive(Clone, Serialize, Deserialize)]
#[serde(bound(
serialize = "T: serde::Serialize",
deserialize = "T: serde::de::DeserializeOwned"
))]
pub struct ListCrdt<T>
where
T: CrdtNode,
@@ -32,6 +36,7 @@ where
pub ops: Vec<Op<T>>,
/// Queue of messages where K is the ID of the message yet to arrive
/// and V is the list of operations depending on it
#[serde(skip)]
message_q: HashMap<OpId, Vec<Op<T>>>,
/// The sequence number of this node
our_seq: SequenceNumber,
+6 -1
View File
@@ -7,6 +7,7 @@
use crate::debug::DebugView;
use crate::json_crdt::{CrdtNode, JsonValue, OpState};
use crate::op::{join_path, print_path, Op, PathSegment, SequenceNumber};
use serde::{Deserialize, Serialize};
use std::cmp::{max, Ordering};
use std::fmt::Debug;
@@ -14,7 +15,11 @@ use crate::keypair::AuthorId;
/// A simple delete-wins, last-writer-wins (LWW) register CRDT.
/// Basically only for adding support for primitives within a more complex CRDT
#[derive(Clone)]
#[derive(Clone, Serialize, Deserialize)]
#[serde(bound(
serialize = "T: serde::Serialize",
deserialize = "T: serde::de::DeserializeOwned"
))]
pub struct LwwRegisterCrdt<T>
where
T: CrdtNode,
+8
View File
@@ -0,0 +1,8 @@
[package]
name = "release-manifest"
version = "0.1.0"
edition = "2024"
[dependencies]
serde = { workspace = true, features = ["derive"] }
serde_json = { workspace = true }
+126
View File
@@ -0,0 +1,126 @@
//! Shared release-manifest type for the signed release-channel pull pipeline.
//!
//! The publisher tool (`crates/release-tool`) builds a [`ReleaseManifest`],
//! serializes it to canonical bytes, signs those bytes with the channel's
//! Ed25519 private key, and publishes the resulting [`SignedManifest`] as
//! `manifest.json` on the release channel. The gateway (`huskies-server`)
//! fetches that file, re-serializes the embedded manifest with
//! [`ReleaseManifest::canonical_bytes`], and verifies the signature against
//! its pinned public key before trusting anything in it.
//!
//! Keeping the type in its own dependency-light crate lets both sides agree
//! on the exact byte representation to sign/verify without the server crate
//! ever linking signing code, and without the publisher tool depending on
//! the full `huskies` server crate.
use serde::{Deserialize, Serialize};
/// The signed payload describing one published release artifact.
///
/// Field order is significant: [`ReleaseManifest::canonical_bytes`] relies on
/// `serde_json`'s struct serialization preserving declaration order, so the
/// signer and verifier always agree on the exact bytes being signed.
#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
pub struct ReleaseManifest {
/// Filename of the artifact within the channel (e.g. `huskies-linux-arm64`).
pub artifact: String,
/// Lowercase hex sha256 digest of the artifact's bytes.
pub sha256: String,
/// Version identifier — the short git commit hash the artifact was built from.
pub version: String,
/// Release channel name this manifest was signed for (e.g. `stable`).
pub channel: String,
/// Unix timestamp (seconds) the manifest was signed at.
///
/// Used for rollback/replay detection: a pull refuses any manifest whose
/// timestamp is not strictly newer than the currently installed one.
pub timestamp: i64,
}
impl ReleaseManifest {
/// Serialize this manifest deterministically for signing and verification.
///
/// Both the publisher and the gateway construct this independently from
/// their own in-memory `ReleaseManifest` value — the manifest.json file's
/// exact on-disk byte layout is never itself the signed payload.
pub fn canonical_bytes(&self) -> Vec<u8> {
serde_json::to_vec(self).expect("ReleaseManifest serialization cannot fail")
}
}
/// A [`ReleaseManifest`] plus its Ed25519 signature (lowercase hex), as
/// published to a release channel's `manifest.json`.
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct SignedManifest {
/// The manifest describing the published artifact.
pub manifest: ReleaseManifest,
/// Hex-encoded Ed25519 signature over `manifest.canonical_bytes()`.
pub signature: String,
}
#[cfg(test)]
mod tests {
use super::*;
fn sample() -> ReleaseManifest {
ReleaseManifest {
artifact: "huskies-linux-arm64".to_string(),
sha256: "a".repeat(64),
version: "abc1234".to_string(),
channel: "stable".to_string(),
timestamp: 1_700_000_000,
}
}
#[test]
fn canonical_bytes_is_deterministic() {
let m = sample();
assert_eq!(m.canonical_bytes(), m.canonical_bytes());
}
#[test]
fn canonical_bytes_changes_with_any_field() {
let m1 = sample();
let mut m2 = sample();
m2.timestamp += 1;
assert_ne!(m1.canonical_bytes(), m2.canonical_bytes());
}
#[test]
fn signed_manifest_roundtrips_through_json() {
let signed = SignedManifest {
manifest: sample(),
signature: "deadbeef".to_string(),
};
let json = serde_json::to_string(&signed).unwrap();
let parsed: SignedManifest = serde_json::from_str(&json).unwrap();
assert_eq!(parsed.manifest, signed.manifest);
assert_eq!(parsed.signature, signed.signature);
}
#[test]
fn manifest_missing_field_fails_to_parse() {
let bad = serde_json::json!({
"artifact": "huskies-linux-arm64",
"sha256": "a".repeat(64),
"version": "abc1234",
"channel": "stable"
// timestamp missing
});
let result: Result<ReleaseManifest, _> = serde_json::from_value(bad);
assert!(
result.is_err(),
"manifest missing a field must fail to parse"
);
}
#[test]
fn signed_manifest_missing_signature_fails_to_parse() {
let bad = serde_json::json!({ "manifest": sample() });
let result: Result<SignedManifest, _> = serde_json::from_value(bad);
assert!(
result.is_err(),
"signed manifest missing signature must fail to parse"
);
}
}
+18
View File
@@ -0,0 +1,18 @@
[package]
name = "release-tool"
version = "0.1.0"
edition = "2024"
[[bin]]
name = "release-tool"
path = "src/main.rs"
[dependencies]
release-manifest = { path = "../release-manifest" }
ed25519-dalek = { workspace = true }
sha2 = { workspace = true }
rand = { workspace = true }
serde_json = { workspace = true }
[dev-dependencies]
tempfile = { workspace = true }
+311
View File
@@ -0,0 +1,311 @@
//! `release-tool` — offline publisher CLI for signed release channels.
//!
//! Generates a release-channel Ed25519 keypair and signs release manifests
//! for a channel's `manifest.json`. This binary is intentionally its own
//! crate, depending only on [`release_manifest`] and `ed25519-dalek` — it
//! never links against the `huskies` server crate, so the running gateway
//! has no code path that can read a channel's private signing key. Run this
//! tool offline (or in a separate publish pipeline) and copy only the
//! resulting public key hex into the gateway's `projects.toml`.
//!
//! Usage:
//! ```text
//! release-tool keygen <key-out-path>
//! release-tool sign --key <path> --artifact <path> --version <str> --channel <str> --out <path> [--timestamp <unix-secs>]
//! ```
use ed25519_dalek::{Signer, SigningKey};
use rand::Rng;
use release_manifest::{ReleaseManifest, SignedManifest};
use sha2::{Digest, Sha256};
use std::path::{Path, PathBuf};
fn main() {
let args: Vec<String> = std::env::args().collect();
let result = match args.get(1).map(String::as_str) {
Some("keygen") => run_keygen(&args[2..]),
Some("sign") => run_sign(&args[2..]),
_ => Err(
"usage: release-tool keygen <key-out-path> | release-tool sign --key <path> \
--artifact <path> --version <str> --channel <str> --out <path> [--timestamp <unix-secs>]"
.to_string(),
),
};
if let Err(e) = result {
eprintln!("error: {e}");
std::process::exit(1);
}
}
// ── keygen ───────────────────────────────────────────────────────────────────
fn run_keygen(args: &[String]) -> Result<(), String> {
let key_path = args.first().ok_or("keygen requires a key-out-path")?;
let signing_key = generate_signing_key();
write_seed_file(Path::new(key_path), &signing_key)?;
let pubkey_hex = hex_encode(signing_key.verifying_key().as_bytes());
println!("Wrote private key seed to {key_path}");
println!("Pinned release public key (paste into projects.toml as `pubkey`):");
println!("{pubkey_hex}");
Ok(())
}
fn generate_signing_key() -> SigningKey {
let mut seed = [0u8; 32];
rand::rng().fill_bytes(&mut seed);
SigningKey::from_bytes(&seed)
}
fn write_seed_file(path: &Path, signing_key: &SigningKey) -> Result<(), String> {
if let Some(parent) = path.parent()
&& !parent.as_os_str().is_empty()
{
std::fs::create_dir_all(parent)
.map_err(|e| format!("cannot create {}: {e}", parent.display()))?;
}
#[cfg(unix)]
{
use std::io::Write;
use std::os::unix::fs::OpenOptionsExt;
let mut file = std::fs::OpenOptions::new()
.write(true)
.create(true)
.truncate(true)
.mode(0o600)
.open(path)
.map_err(|e| format!("cannot create {}: {e}", path.display()))?;
file.write_all(&signing_key.to_bytes())
.map_err(|e| format!("cannot write {}: {e}", path.display()))
}
#[cfg(not(unix))]
{
std::fs::write(path, signing_key.to_bytes())
.map_err(|e| format!("cannot write {}: {e}", path.display()))
}
}
fn load_seed_file(path: &Path) -> Result<SigningKey, String> {
let bytes = std::fs::read(path).map_err(|e| format!("cannot read {}: {e}", path.display()))?;
let seed: [u8; 32] = bytes
.try_into()
.map_err(|_| format!("{} must contain exactly 32 bytes", path.display()))?;
Ok(SigningKey::from_bytes(&seed))
}
// ── sign ─────────────────────────────────────────────────────────────────────
/// Parsed `sign` subcommand arguments.
struct SignArgs {
key: PathBuf,
artifact: PathBuf,
version: String,
channel: String,
out: PathBuf,
timestamp: Option<i64>,
}
fn parse_sign_args(args: &[String]) -> Result<SignArgs, String> {
let mut key = None;
let mut artifact = None;
let mut version = None;
let mut channel = None;
let mut out = None;
let mut timestamp = None;
let mut i = 0;
while i < args.len() {
let flag = args[i].as_str();
let value = args
.get(i + 1)
.ok_or_else(|| format!("missing value for {flag}"))?;
match flag {
"--key" => key = Some(PathBuf::from(value)),
"--artifact" => artifact = Some(PathBuf::from(value)),
"--version" => version = Some(value.clone()),
"--channel" => channel = Some(value.clone()),
"--out" => out = Some(PathBuf::from(value)),
"--timestamp" => {
timestamp = Some(
value
.parse::<i64>()
.map_err(|_| format!("--timestamp must be an integer, got `{value}`"))?,
)
}
other => return Err(format!("unknown flag `{other}`")),
}
i += 2;
}
Ok(SignArgs {
key: key.ok_or("--key is required")?,
artifact: artifact.ok_or("--artifact is required")?,
version: version.ok_or("--version is required")?,
channel: channel.ok_or("--channel is required")?,
out: out.ok_or("--out is required")?,
timestamp,
})
}
fn run_sign(args: &[String]) -> Result<(), String> {
let parsed = parse_sign_args(args)?;
let signing_key = load_seed_file(&parsed.key)?;
let artifact_bytes = std::fs::read(&parsed.artifact)
.map_err(|e| format!("cannot read {}: {e}", parsed.artifact.display()))?;
let artifact_name = parsed
.artifact
.file_name()
.and_then(|n| n.to_str())
.ok_or("--artifact path has no filename")?
.to_string();
let timestamp = match parsed.timestamp {
Some(t) => t,
None => std::time::SystemTime::now()
.duration_since(std::time::UNIX_EPOCH)
.map_err(|e| format!("system clock before epoch: {e}"))?
.as_secs() as i64,
};
let signed = sign_manifest(
&signing_key,
artifact_name,
&artifact_bytes,
parsed.version,
parsed.channel,
timestamp,
);
let json =
serde_json::to_string_pretty(&signed).map_err(|e| format!("serialise manifest: {e}"))?;
std::fs::write(&parsed.out, json)
.map_err(|e| format!("cannot write {}: {e}", parsed.out.display()))?;
println!("Signed manifest written to {}", parsed.out.display());
Ok(())
}
/// Build and sign a [`SignedManifest`] for the given artifact bytes.
///
/// Pure aside from the signature computation — split out from `run_sign` so
/// tests can exercise it without touching the filesystem.
fn sign_manifest(
signing_key: &SigningKey,
artifact: String,
artifact_bytes: &[u8],
version: String,
channel: String,
timestamp: i64,
) -> SignedManifest {
let mut hasher = Sha256::new();
hasher.update(artifact_bytes);
let sha256 = hex_encode(&hasher.finalize());
let manifest = ReleaseManifest {
artifact,
sha256,
version,
channel,
timestamp,
};
let signature = hex_encode(&signing_key.sign(&manifest.canonical_bytes()).to_bytes());
SignedManifest {
manifest,
signature,
}
}
// ── helpers ──────────────────────────────────────────────────────────────────
fn hex_encode(bytes: &[u8]) -> String {
bytes.iter().map(|b| format!("{b:02x}")).collect()
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn keygen_then_sign_produces_verifiable_signature() {
let tmp = tempfile::tempdir().unwrap();
let key_path = tmp.path().join("channel.key");
let signing_key = generate_signing_key();
write_seed_file(&key_path, &signing_key).unwrap();
let loaded = load_seed_file(&key_path).unwrap();
assert_eq!(loaded.verifying_key(), signing_key.verifying_key());
let signed = sign_manifest(
&loaded,
"huskies-linux-arm64".to_string(),
b"fake binary contents",
"abc1234".to_string(),
"stable".to_string(),
1_700_000_000,
);
// Verify with ed25519-dalek directly, mirroring how the gateway verifies.
use ed25519_dalek::Verifier;
let sig_bytes: [u8; 64] = hex_bytes(&signed.signature).try_into().unwrap();
let sig = ed25519_dalek::Signature::from_bytes(&sig_bytes);
assert!(
signing_key
.verifying_key()
.verify(&signed.manifest.canonical_bytes(), &sig)
.is_ok(),
"signature produced by sign_manifest must verify against the signing key's pubkey"
);
}
#[test]
fn sign_manifest_hashes_artifact_bytes() {
let signing_key = generate_signing_key();
let signed = sign_manifest(
&signing_key,
"art".to_string(),
b"hello world",
"v1".to_string(),
"stable".to_string(),
1,
);
let mut hasher = Sha256::new();
hasher.update(b"hello world");
let expected = hex_encode(&hasher.finalize());
assert_eq!(signed.manifest.sha256, expected);
}
#[test]
fn parse_sign_args_rejects_missing_required_flag() {
let args: Vec<String> = vec!["--key".into(), "k".into()];
assert!(parse_sign_args(&args).is_err());
}
#[test]
fn parse_sign_args_accepts_all_flags() {
let args: Vec<String> = vec![
"--key".into(),
"k".into(),
"--artifact".into(),
"a".into(),
"--version".into(),
"v1".into(),
"--channel".into(),
"stable".into(),
"--out".into(),
"o".into(),
"--timestamp".into(),
"42".into(),
];
let parsed = parse_sign_args(&args).unwrap();
assert_eq!(parsed.timestamp, Some(42));
assert_eq!(parsed.channel, "stable");
}
fn hex_bytes(s: &str) -> Vec<u8> {
(0..s.len())
.step_by(2)
.map(|i| u8::from_str_radix(&s[i..i + 2], 16).unwrap())
.collect()
}
}
+4 -6
View File
@@ -111,10 +111,9 @@ fn parse_pub_item(line: &str) -> Option<(String, String)> {
let rest = if let Some(r) = trimmed.strip_prefix("pub(") {
let end = r.find(')')?;
r[end + 1..].trim_start()
} else if let Some(r) = trimmed.strip_prefix("pub ") {
r.trim_start()
} else {
return None;
let r = trimmed.strip_prefix("pub ")?;
r.trim_start()
};
// Handle "async fn"
@@ -139,10 +138,9 @@ fn parse_pub_item(line: &str) -> Option<(String, String)> {
("const", r.trim_start())
} else if let Some(r) = rest.strip_prefix("static ") {
("static", r.trim_start())
} else if let Some(r) = rest.strip_prefix("mod ") {
("mod", r.trim_start())
} else {
return None;
let r = rest.strip_prefix("mod ")?;
("mod", r.trim_start())
};
let name: String = name_part
+4 -6
View File
@@ -123,10 +123,9 @@ fn parse_exported_item(line: &str) -> Option<(String, String)> {
// Strip "export default" or "export"
let rest = if let Some(r) = trimmed.strip_prefix("export default ") {
r.trim_start()
} else if let Some(r) = trimmed.strip_prefix("export ") {
r.trim_start()
} else {
return None;
let r = trimmed.strip_prefix("export ")?;
r.trim_start()
};
// Strip optional "async"
@@ -148,10 +147,9 @@ fn parse_exported_item(line: &str) -> Option<(String, String)> {
("const", r.trim_start())
} else if let Some(r) = rest.strip_prefix("let ") {
("let", r.trim_start())
} else if let Some(r) = rest.strip_prefix("enum ") {
("enum", r.trim_start())
} else {
return None;
let r = rest.strip_prefix("enum ")?;
("enum", r.trim_start())
};
let name: String = name_part
+17 -3
View File
@@ -7,7 +7,7 @@
#
# Tested with: OrbStack (recommended on macOS), Docker Desktop (slower bind mounts)
FROM rust:1.93-bookworm AS base
FROM rust:1.94-bookworm AS base
# Clippy and rustfmt are needed at runtime for acceptance gates
RUN rustup component add clippy rustfmt
@@ -46,8 +46,17 @@ WORKDIR /app
# build.rs) can produce the release binary with embedded frontend assets.
COPY . .
# Build frontend deps first (better layer caching)
RUN cd frontend && npm ci
# Build frontend deps first (better layer caching).
# Cannot use `npm ci` because of npm's optional-dependencies bug
# (npm/cli#4828): platform-specific bindings (e.g. rolldown's
# linux-arm64-gnu native binary, introduced by 1119's vite 5→8 upgrade)
# get listed in package-lock.json for the lockfile author's platform
# only, so `npm ci` skips them on every other platform — the build
# then fails at runtime with `Cannot find native binding`. Wipe the
# lockfile + node_modules and let `npm install` resolve fresh for the
# build platform. The lockfile mutation stays inside the container
# image and never reaches the host repo.
RUN cd frontend && rm -rf node_modules package-lock.json && npm install
# Build the release binary (build.rs runs npm run build for the frontend)
RUN cargo build --release \
@@ -79,6 +88,11 @@ RUN curl -fsSL https://deb.nodesource.com/setup_22.x | bash - \
# Claude Code CLI in runtime
RUN npm install -g @anthropic-ai/claude-code
# jscpd — duplication detector used by script/check. Installed in the
# runtime stage (not just the base build stage) so it's available to agents
# running script/check inside the sled, not only on a developer machine.
RUN npm install -g jscpd
# Cargo and Rust toolchain needed at runtime for:
# - rebuild_and_restart (cargo build inside the container)
# - Agent-driven cargo commands (cargo clippy, cargo test, etc.)
+16 -1
View File
@@ -29,8 +29,21 @@ RUN apt-get update && apt-get install -y --no-install-recommends \
sudo \
&& rm -rf /var/lib/apt/lists/*
# Node.js 22.x from NodeSource — bookworm's apt nodejs is 18.x, which modern
# frontend toolchains (vite ≥ 7) refuse to run under. Project builds that
# shell out to npm (e.g. build.rs frontend steps) need ≥ 20.
RUN curl -fsSL https://deb.nodesource.com/setup_22.x | bash - \
&& apt-get install -y --no-install-recommends nodejs \
&& npm install -g @anthropic-ai/claude-code \
&& rm -rf /var/lib/apt/lists/*
# Copy the huskies binary and entrypoint from the main image.
COPY --from=huskies-src /usr/local/bin/huskies /usr/local/bin/huskies
# The binary lives in /opt/huskies/bin (chowned to the huskies user below) so
# the in-process upgrade path (`POST /api/upgrade`) can atomically replace it
# without root: rename() needs write permission on the *directory*, not just
# the file. /usr/local/bin/huskies stays a symlink so PATH resolution and the
# entrypoint CMD are unchanged.
COPY --from=huskies-src /usr/local/bin/huskies /opt/huskies/bin/huskies
COPY --from=huskies-src /usr/local/bin/entrypoint.sh /usr/local/bin/entrypoint.sh
# Non-root user — Claude Code refuses --dangerously-skip-permissions as root.
@@ -41,6 +54,8 @@ RUN groupadd -r huskies \
&& mkdir -p /home/huskies/.ssh \
&& chmod 700 /home/huskies/.ssh \
&& chown -R huskies:huskies /home/huskies \
&& chown -R huskies:huskies /opt/huskies \
&& ln -s /opt/huskies/bin/huskies /usr/local/bin/huskies \
&& mkdir -p /workspace \
&& chown huskies:huskies /workspace \
&& git config --global init.defaultBranch master \
+3
View File
@@ -29,6 +29,9 @@ services:
- HUSKIES_PORT=3001
# Bind to all interfaces so Docker port forwarding works.
- HUSKIES_HOST=0.0.0.0
# Gateway URL so this sled's relay task forwards CRDT events to the gateway.
# Uses host.docker.internal so the container can reach the gateway on the host.
- HUSKIES_GATEWAY_URL=http://host.docker.internal:3000
# Optional: Matrix bot credentials (if using Matrix integration)
- MATRIX_HOMESERVER=${MATRIX_HOMESERVER:-}
- MATRIX_USER=${MATRIX_USER:-}
+10
View File
@@ -1,6 +1,16 @@
#!/bin/sh
set -e
# ── Claude credentials ────────────────────────────────────────────────
# The `new project` command bind-mounts the host ~/.claude/.credentials.json
# at /run/claude-credentials-src:ro. We copy it here so the huskies user
# owns the file and mode 0600 is enforced regardless of host uid/gid.
if [ -f /run/claude-credentials-src ]; then
mkdir -p /home/huskies/.claude
cp /run/claude-credentials-src /home/huskies/.claude/.credentials.json
chmod 600 /home/huskies/.claude/.credentials.json
fi
# ── SSH authorized key ────────────────────────────────────────────────
# HUSKIES_SSH_PUBKEY is set by `new project` when it generates a keypair.
# Write it to authorized_keys so the user can connect with the matching
+6 -5
View File
@@ -13,19 +13,20 @@
USER root
# OpenJDK 21 (current LTS) and Maven for build support.
# OpenJDK 17 (Bookworm's default LTS) and Maven for build support.
RUN apt-get update && apt-get install -y --no-install-recommends \
openjdk-21-jdk-headless \
openjdk-17-jdk-headless \
maven \
&& rm -rf /var/lib/apt/lists/*
ENV JAVA_HOME="/usr/lib/jvm/java-21-openjdk-amd64"
RUN ln -s /usr/lib/jvm/java-17-openjdk-* /usr/lib/jvm/java-17-openjdk
ENV JAVA_HOME="/usr/lib/jvm/java-17-openjdk"
# Eclipse JDT Language Server — canonical LSP for Java/JVM (Java, Kotlin, Groovy).
# Pin to a specific release; update JDTLS_VERSION + JDTLS_BUILD for upgrades.
# All releases: https://github.com/eclipse-jdtls/eclipse.jdt.ls/releases
ENV JDTLS_VERSION="1.38.0" \
JDTLS_BUILD="202503271418"
ENV JDTLS_VERSION="1.60.0" \
JDTLS_BUILD="202606262232"
RUN mkdir -p /opt/jdtls \
&& curl -fsSL \
"https://download.eclipse.org/jdtls/milestones/${JDTLS_VERSION}/jdt-language-server-${JDTLS_VERSION}-${JDTLS_BUILD}.tar.gz" \
+4
View File
@@ -14,10 +14,14 @@
USER root
# Build tools required by rustup and many Rust crates.
# libudev-dev: serial/USB device crates (libudev-sys, serialport).
# libclang-dev: bindgen-based crates; cargo test is skipped when absent.
RUN apt-get update && apt-get install -y --no-install-recommends \
build-essential \
pkg-config \
libssl-dev \
libudev-dev \
libclang-dev \
&& rm -rf /var/lib/apt/lists/*
ENV RUSTUP_HOME="/home/huskies/.rustup" \
+945 -1215
View File
File diff suppressed because it is too large Load Diff
+5 -5
View File
@@ -1,7 +1,7 @@
{
"name": "huskies",
"private": true,
"version": "0.12.0",
"version": "0.14.5",
"type": "module",
"scripts": {
"dev": "vite",
@@ -32,11 +32,11 @@
"@types/node": "^25.0.0",
"@types/react": "^19.1.8",
"@types/react-dom": "^19.1.6",
"@vitejs/plugin-react": "^4.6.0",
"@vitest/coverage-v8": "^2.1.9",
"@vitejs/plugin-react": "^5.2.0",
"@vitest/coverage-v8": "^4.1.6",
"jsdom": "^28.1.0",
"typescript": "~5.8.3",
"vite": "^5.4.21",
"vitest": "^2.1.4"
"vite": "^8.0.13",
"vitest": "^4.1.6"
}
}
+2
View File
@@ -160,6 +160,7 @@ describe("App", () => {
});
it("shows error when openProject fails", async () => {
const errorSpy = vi.spyOn(console, "error").mockImplementation(() => {});
mockedApi.openProject.mockRejectedValue(new Error("Path does not exist"));
await renderApp();
@@ -182,6 +183,7 @@ describe("App", () => {
await waitFor(() => {
expect(screen.getByText(/Path does not exist/)).toBeInTheDocument();
});
errorSpy.mockRestore();
});
it("shows known projects list", async () => {
+2
View File
@@ -266,6 +266,8 @@ describe("subscribeAgentStream", () => {
});
it("handles malformed JSON without throwing", () => {
vi.spyOn(console, "error").mockImplementation(() => {});
subscribeAgentStream("42_story_test", "coder", vi.fn());
expect(() => {
@@ -472,9 +472,16 @@ describe("Slash command handling (Story 374)", () => {
});
describe("Story 1058: WebSocket errors do not appear in chat", () => {
let consoleSpy: ReturnType<typeof vi.spyOn>;
beforeEach(() => {
capturedWsHandlers = null;
setupMocks();
consoleSpy = vi.spyOn(console, "error").mockImplementation(() => {});
});
afterEach(() => {
consoleSpy.mockRestore();
});
it("does not add a chat message when onError is called", async () => {
+1 -1
View File
@@ -135,7 +135,7 @@ export function PermissionDialog({
fontSize: "0.9em",
}}
>
Always Allow
Don't ask again this session
</button>
</div>
</div>
@@ -453,6 +453,67 @@ describe("StagePanel", () => {
expect(badge).toHaveTextContent("BLOCKED");
});
// Story 1215 — confirms the board shows an active-agent indicator instead
// of the plain blocked icon whenever a running/pending agent is present,
// even though blocked=true, and falls back to the plain icon otherwise.
it("shows active-agent indicator instead of blocked icon when a running/pending agent is present, even if blocked=true", () => {
const items: PipelineStageItem[] = [
{
story_id: "54_story_blocked_active_agent",
name: "Blocked With Active Agent",
error: null,
merge_failure: null,
agent: { agent_name: "coder", model: "claude", status: "running" },
review_hold: null,
qa: null,
depends_on: null,
blocked: true,
},
];
render(<StagePanel title="Current" items={items} />);
const badge = screen.getByTestId("blocked-badge-54_story_blocked_active_agent");
expect(badge).not.toHaveTextContent("BLOCKED");
expect(badge).toHaveTextContent("RECOVERING");
});
it("drives the distinct indicator off the agent's running/pending status, not the blocked flag alone", () => {
const items: PipelineStageItem[] = [
{
story_id: "55_story_blocked_stale_agent",
name: "Blocked With Completed Agent",
error: null,
merge_failure: null,
agent: { agent_name: "coder", model: "claude", status: "completed" },
review_hold: null,
qa: null,
depends_on: null,
blocked: true,
},
];
render(<StagePanel title="Current" items={items} />);
const badge = screen.getByTestId("blocked-badge-55_story_blocked_stale_agent");
expect(badge).toHaveTextContent("BLOCKED");
});
it("shows the plain blocked icon for a story that is blocked with no live agent", () => {
const items: PipelineStageItem[] = [
{
story_id: "56_story_blocked_no_agent",
name: "Blocked No Agent",
error: null,
merge_failure: null,
agent: null,
review_hold: null,
qa: null,
depends_on: null,
blocked: true,
},
];
render(<StagePanel title="Current" items={items} />);
const badge = screen.getByTestId("blocked-badge-56_story_blocked_no_agent");
expect(badge).toHaveTextContent("BLOCKED");
});
it("shows spinning icon for merge_failure item with running mergemaster", () => {
const items: PipelineStageItem[] = [
{
@@ -227,6 +227,7 @@ describe("usePathCompletion hook", () => {
});
it("sets completionError when listDirectoryAbsolute throws an Error", async () => {
const errorSpy = vi.spyOn(console, "error").mockImplementation(() => {});
mockListDir.mockRejectedValue(new Error("Permission denied"));
const { result } = renderHook(() =>
@@ -242,9 +243,13 @@ describe("usePathCompletion hook", () => {
await waitFor(() => {
expect(result.current.completionError).toBe("Permission denied");
});
expect(errorSpy).toHaveBeenCalledWith(new Error("Permission denied"));
errorSpy.mockRestore();
});
it("sets generic completionError when listDirectoryAbsolute throws a non-Error", async () => {
const errorSpy = vi.spyOn(console, "error").mockImplementation(() => {});
mockListDir.mockRejectedValue("some string error");
const { result } = renderHook(() =>
@@ -262,6 +267,9 @@ describe("usePathCompletion hook", () => {
"Failed to compute suggestion.",
);
});
expect(errorSpy).toHaveBeenCalledWith("some string error");
errorSpy.mockRestore();
});
it("clears suggestionTail when selected match path does not start with input", async () => {
+39
View File
@@ -0,0 +1,39 @@
#!/usr/bin/env bash
set -euo pipefail
# Build all project images in dependency order:
# huskies → huskies-project-base → huskies-project-<stack> (one per stack fragment)
#
# Called automatically by `script/release` (story 1242) so the huskies-project-*
# images never drift from the version being published. Also safe to run
# standalone after `script/docker_rebuild` or whenever you add a new stack —
# each step re-tags the image with the latest layers.
cd "$(dirname "$0")/.."
if [[ -f .env ]]; then
set -a
source .env
set +a
fi
CACHE_FLAG=""
if [[ "${1:-}" == "--no-cache" ]]; then
CACHE_FLAG="--no-cache"
fi
echo "==> Building huskies"
docker build $CACHE_FLAG -t huskies -f docker/Dockerfile .
echo "==> Building huskies-project-base"
docker build $CACHE_FLAG -t huskies-project-base -f docker/Dockerfile.base .
for fragment in docker/stacks/*/Dockerfile.fragment; do
stack=$(basename "$(dirname "$fragment")")
image="huskies-project-${stack}"
echo "==> Building ${image}"
(printf 'FROM huskies-project-base\n'; cat "$fragment") \
| docker build $CACHE_FLAG -t "$image" -
done
echo "All project images built."
+18 -4
View File
@@ -1,7 +1,8 @@
#!/usr/bin/env bash
# Pre-commit quality gate: fmt-check, clippy, cargo check, and doc-coverage.
# Run this before committing to catch fmt drift, clippy warnings, compile
# errors, and missing doc comments without waiting for the full test suite.
# Pre-commit quality gate: fmt-check, clippy, duplication, cargo check, and
# doc-coverage. Run this before committing to catch fmt drift, clippy
# warnings, duplicate code, compile errors, and missing doc comments without
# waiting for the full test suite.
set -euo pipefail
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
@@ -11,7 +12,20 @@ echo "=== Checking Rust formatting ==="
cargo fmt --manifest-path "$PROJECT_ROOT/Cargo.toml" --all --check
echo "=== Running cargo clippy ==="
cargo clippy --manifest-path "$PROJECT_ROOT/Cargo.toml" --workspace --all-targets -- -D warnings
# clippy::cognitive_complexity is allow-by-default; -W activates it so that
# -D warnings turns violations into a hard failure. The threshold it's
# measured against lives in clippy.toml (committed, not passed ad hoc).
cargo clippy --manifest-path "$PROJECT_ROOT/Cargo.toml" --workspace --all-targets -- -W clippy::cognitive_complexity -D warnings
echo "=== Checking code duplication (jscpd) ==="
if ! command -v jscpd &>/dev/null; then
echo "FAIL: jscpd is not installed. Install it with 'npm install -g jscpd' before running script/check." >&2
exit 1
fi
# Duplication threshold lives in .jscpd.json (committed, not passed ad hoc).
# jscpd exits non-zero automatically when duplication exceeds that threshold.
jscpd --config "$PROJECT_ROOT/.jscpd.json" \
"$PROJECT_ROOT/server/src" "$PROJECT_ROOT/frontend/src" "$PROJECT_ROOT/crates"
echo "=== Checking doc coverage on changed files ==="
cargo run --manifest-path "$PROJECT_ROOT/Cargo.toml" -p source-map-gen --bin source-map-check --quiet -- --worktree "$PROJECT_ROOT" --base master
+121
View File
@@ -0,0 +1,121 @@
#!/usr/bin/env bash
# script/ci-publish-artifact — upload a built sled artifact to a release
# channel and verify the channel's manifest reflects the published commit.
#
# Usage: script/ci-publish-artifact <artifact-path> [git-hash]
#
# git-hash defaults to `git rev-parse --short HEAD` when omitted.
#
# Required env:
# HUSKIES_CHANNEL_URL Base URL of the release channel (Gitea secret).
# HUSKIES_CHANNEL_TOKEN Bearer token authorised to publish (Gitea secret).
#
# Protocol against the channel host:
# POST {HUSKIES_CHANNEL_URL}/<artifact-name>
# Headers: Authorization: Bearer <token>, X-Git-Hash: <hash>
# Body: raw artifact bytes
# GET {HUSKIES_CHANNEL_URL}/manifest.json
# Headers: Authorization: Bearer <token>
# Body: JSON object with a "git_hash" field
#
# 5xx responses and network errors are retried with backoff; 4xx responses
# (including auth failures) fail immediately since retrying won't fix them.
set -euo pipefail
ARTIFACT_PATH="${1:?Usage: script/ci-publish-artifact <artifact-path> [git-hash]}"
GIT_HASH="${2:-$(git rev-parse --short HEAD)}"
if [ -z "${HUSKIES_CHANNEL_URL:-}" ]; then
echo "Error: HUSKIES_CHANNEL_URL is not set." >&2
exit 1
fi
if [ -z "${HUSKIES_CHANNEL_TOKEN:-}" ]; then
echo "Error: HUSKIES_CHANNEL_TOKEN is not set." >&2
exit 1
fi
if [ ! -f "$ARTIFACT_PATH" ]; then
echo "Error: artifact not found at $ARTIFACT_PATH" >&2
exit 1
fi
ARTIFACT_NAME="$(basename "$ARTIFACT_PATH")"
CHANNEL_URL="${HUSKIES_CHANNEL_URL%/}"
UPLOAD_URL="${CHANNEL_URL}/${ARTIFACT_NAME}"
MANIFEST_URL="${CHANNEL_URL}/manifest.json"
UPLOAD_MAX_ATTEMPTS="${HUSKIES_CI_PUBLISH_MAX_ATTEMPTS:-3}"
UPLOAD_BACKOFF_SECS="${HUSKIES_CI_PUBLISH_BACKOFF_SECS:-1}"
MANIFEST_MAX_ATTEMPTS=3
MANIFEST_BACKOFF_SECS=1
RESPONSE_FILE="$(mktemp)"
trap 'rm -f "$RESPONSE_FILE"' EXIT
# ── Upload ────────────────────────────────────────────────────────────────
attempt=1
while :; do
echo "==> Uploading ${ARTIFACT_NAME} (${GIT_HASH}), attempt ${attempt}/${UPLOAD_MAX_ATTEMPTS}..."
HTTP_CODE=$(curl -sS --connect-timeout 10 --max-time 60 \
-o "$RESPONSE_FILE" -w "%{http_code}" \
-X POST \
-H "Authorization: Bearer ${HUSKIES_CHANNEL_TOKEN}" \
-H "X-Git-Hash: ${GIT_HASH}" \
--data-binary "@${ARTIFACT_PATH}" \
"${UPLOAD_URL}") || HTTP_CODE="000"
RESPONSE_BODY="$(cat "$RESPONSE_FILE" 2>/dev/null || true)"
case "$HTTP_CODE" in
2??)
echo "==> Upload succeeded (HTTP ${HTTP_CODE})."
break
;;
401|403)
echo "Error: upload rejected — authentication failed (HTTP ${HTTP_CODE})." >&2
echo "Response: ${RESPONSE_BODY}" >&2
exit 1
;;
4??)
echo "Error: upload rejected by the channel (HTTP ${HTTP_CODE}); not retrying a client error." >&2
echo "Response: ${RESPONSE_BODY}" >&2
exit 1
;;
esac
if [ "$attempt" -ge "$UPLOAD_MAX_ATTEMPTS" ]; then
echo "Error: upload failed after ${UPLOAD_MAX_ATTEMPTS} attempts (last HTTP ${HTTP_CODE})." >&2
echo "Response: ${RESPONSE_BODY}" >&2
exit 1
fi
echo "==> Transient failure (HTTP ${HTTP_CODE}); retrying in ${UPLOAD_BACKOFF_SECS}s..."
sleep "$UPLOAD_BACKOFF_SECS"
attempt=$((attempt + 1))
UPLOAD_BACKOFF_SECS=$((UPLOAD_BACKOFF_SECS * 2))
done
# ── Verify manifest ──────────────────────────────────────────────────────
attempt=1
while :; do
echo "==> Verifying channel manifest reflects ${GIT_HASH} (attempt ${attempt}/${MANIFEST_MAX_ATTEMPTS})..."
MANIFEST_JSON=$(curl -sS --connect-timeout 10 --max-time 30 \
-H "Authorization: Bearer ${HUSKIES_CHANNEL_TOKEN}" \
"${MANIFEST_URL}") || MANIFEST_JSON=""
MANIFEST_HASH=$(printf '%s' "$MANIFEST_JSON" \
| python3 -c "import sys,json; print(json.load(sys.stdin).get('git_hash',''))" 2>/dev/null || echo "")
if [ "$MANIFEST_HASH" = "$GIT_HASH" ]; then
echo "==> Published ${ARTIFACT_NAME} (${GIT_HASH}) to ${CHANNEL_URL}; manifest verified."
exit 0
fi
if [ "$attempt" -ge "$MANIFEST_MAX_ATTEMPTS" ]; then
echo "Error: manifest mismatch — channel reports git_hash '${MANIFEST_HASH}', expected '${GIT_HASH}'." >&2
echo "Manifest: ${MANIFEST_JSON}" >&2
exit 1
fi
sleep "$MANIFEST_BACKOFF_SECS"
attempt=$((attempt + 1))
MANIFEST_BACKOFF_SECS=$((MANIFEST_BACKOFF_SECS * 2))
done
+2
View File
@@ -24,4 +24,6 @@ docker compose -f docker/docker-compose.yml down
docker compose -f docker/docker-compose.yml build $CACHE_FLAG
docker compose -f docker/docker-compose.yml up -d
script/build-project-images $CACHE_FLAG
echo "Rebuild complete. Logs: docker compose -f docker/docker-compose.yml logs -f"
+165
View File
@@ -0,0 +1,165 @@
#!/usr/bin/env bash
# Build huskies, install (codesign-heal wrapper + underlying binary), and if a
# gateway is running on this host, hot-restart it detached from the current shell
# so SSH disconnect — e.g. when redeploying from a phone — doesn't kill it.
#
# Skips the restart silently if no gateway is running. Errors loudly if more
# than one matches, so we don't restart the wrong one.
#
# Pass --skip-check to bypass `script/check` (useful for docs / build-script
# changes you've already verified).
#
# On relaunch failure the previous binary is restored from
# ~/bin/huskies-bin.prev and re-launched, so a bad deploy doesn't leave the
# host without a working gateway.
#
# After a `cp` or download the binary loses its ad-hoc signature and macOS
# SIGKILLs it silently on Apple Silicon. The wrapper at ~/bin/huskies re-signs
# the underlying binary at ~/bin/huskies-bin whenever codesign validation
# fails, then execs it. Normal launches (already signed) are zero-overhead.
set -euo pipefail
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
PROJECT_ROOT="$(cd "$SCRIPT_DIR/.." && pwd)"
LOG_DIR="${HUSKIES_LOG_DIR:-$PROJECT_ROOT/logs}"
GATEWAY_PATTERN='huskies .*--gateway'
BIN_DIR="${HOME}/bin"
UNDERLYING="${BIN_DIR}/huskies-bin"
WRAPPER="${BIN_DIR}/huskies"
PREV_BIN="${BIN_DIR}/huskies-bin.prev"
NEW_BIN="${PROJECT_ROOT}/target/release/huskies"
SKIP_CHECK=0
for arg in "$@"; do
case "$arg" in
--skip-check) SKIP_CHECK=1 ;;
-h|--help) sed -n '2,17p' "$0"; exit 0 ;;
*) echo "Unknown arg: $arg (use --help)" >&2; exit 2 ;;
esac
done
if [ "$SKIP_CHECK" -eq 0 ] && [ -x "$SCRIPT_DIR/check" ]; then
echo "=== Running script/check ==="
"$SCRIPT_DIR/check"
fi
echo "=== Building release binary ==="
cd "$PROJECT_ROOT"
cargo build --release --bin huskies
mkdir -p "$BIN_DIR"
# Snapshot current binary so we can roll back if the relaunch fails.
PREV_VERSION=""
if [ -x "$UNDERLYING" ]; then
PREV_VERSION="$("$UNDERLYING" --version 2>/dev/null || echo unknown)"
cp "$UNDERLYING" "$PREV_BIN"
fi
cp "$NEW_BIN" "$UNDERLYING"
chmod +x "$UNDERLYING"
codesign -s - -f "$UNDERLYING" 2>/dev/null
NEW_VERSION="$("$UNDERLYING" --version 2>/dev/null || echo unknown)"
echo "==> Installed binary: ${UNDERLYING}"
if [ -n "$PREV_VERSION" ]; then
echo " version: $PREV_VERSION → $NEW_VERSION"
else
echo " version: $NEW_VERSION (no prior install)"
fi
cat > "${WRAPPER}" << 'WRAPPER_EOF'
#!/usr/bin/env bash
# Codesign-heal wrapper — re-signs ~/bin/huskies-bin if the signature is
# missing or invalid, then execs the binary. Logs only when it re-signs.
BIN="${HOME}/bin/huskies-bin"
if ! codesign --verify --quiet "${BIN}" 2>/dev/null; then
codesign -s - "${BIN}"
echo "[codesign-heal] re-signed ~/bin/huskies-bin" >&2
fi
exec "${BIN}" "$@"
WRAPPER_EOF
chmod +x "${WRAPPER}"
echo "==> Installed wrapper: ${WRAPPER}"
# ── Hot-restart gateway if one is running ─────────────────────────────
collect_descendants() {
local pid="$1" kid
for kid in $(pgrep -P "$pid" 2>/dev/null); do
collect_descendants "$kid"
printf '%s\n' "$kid"
done
}
GATEWAY_PIDS="$(pgrep -f "$GATEWAY_PATTERN" || true)"
if [ -z "$GATEWAY_PIDS" ]; then
echo "==> No running gateway found; install complete."
exit 0
fi
if [ "$(echo "$GATEWAY_PIDS" | wc -l)" -gt 1 ]; then
echo "Error: multiple gateway processes match '${GATEWAY_PATTERN}':" >&2
ps -p $GATEWAY_PIDS -o pid,args >&2 || true
echo "Refusing to guess which to restart." >&2
exit 3
fi
GATEWAY_PID="$GATEWAY_PIDS"
GATEWAY_ARGS="$(ps -p "$GATEWAY_PID" -o args= | sed -E 's@^[^ ]*huskies[^ ]* @@')"
GATEWAY_CWD="$(lsof -p "$GATEWAY_PID" 2>/dev/null | awk '$4=="cwd"{print $9; exit}')"
if [ -z "$GATEWAY_CWD" ]; then GATEWAY_CWD="$PWD"; fi
LOG_FILE="$LOG_DIR/gateway-$(date +%Y%m%d-%H%M%S).log"
mkdir -p "$LOG_DIR"
DESCENDANTS="$(collect_descendants "$GATEWAY_PID" | tr '\n' ' ')"
echo "==> Stopping gateway tree (pids: $GATEWAY_PID $DESCENDANTS)"
# Kill descendants depth-first so PTY children die before the gateway, then the gateway.
for pid in $DESCENDANTS $GATEWAY_PID; do
kill "$pid" 2>/dev/null || true
done
sleep 2
echo "==> Restarting gateway"
echo " log: $LOG_FILE"
(
cd "$GATEWAY_CWD"
nohup "$WRAPPER" $GATEWAY_ARGS >> "$LOG_FILE" 2>&1 < /dev/null &
disown
)
# Wait up to 10s for the new gateway to appear AND be a different PID.
NEW_PID=""
for _ in 1 2 3 4 5 6 7 8 9 10; do
sleep 1
candidate="$(pgrep -f "$GATEWAY_PATTERN" 2>/dev/null || true)"
if [ -n "$candidate" ] && [ "$candidate" != "$GATEWAY_PID" ]; then
NEW_PID="$candidate"
break
fi
done
if [ -n "$NEW_PID" ]; then
echo "==> Gateway restarted as pid $NEW_PID"
exit 0
fi
# ── Rollback ──────────────────────────────────────────────────────────
echo "Error: new gateway failed to come up within 10s; rolling back" >&2
if [ -x "$PREV_BIN" ]; then
cp "$PREV_BIN" "$UNDERLYING"
chmod +x "$UNDERLYING"
codesign -s - -f "$UNDERLYING" 2>/dev/null
echo "==> Restored previous binary"
(
cd "$GATEWAY_CWD"
nohup "$WRAPPER" $GATEWAY_ARGS >> "$LOG_FILE" 2>&1 < /dev/null &
disown
)
sleep 2
if pgrep -f "$GATEWAY_PATTERN" >/dev/null 2>&1; then
echo "==> Gateway restored to previous version"
exit 1
fi
fi
echo "Error: rollback failed; gateway is DOWN. Inspect $LOG_FILE." >&2
exit 1
+20 -1
View File
@@ -87,6 +87,19 @@ cross build --release --target x86_64-unknown-linux-musl
echo "==> Building Linux arm64 (static musl via cross)..."
cross build --release --target aarch64-unknown-linux-musl
# ── Build project images ─────────────────────────────────────────
# Rebuild the huskies-project-* Docker images from this exact source tree
# (the version-bump commit above already landed, so build.rs's `git
# rev-parse HEAD` embeds the matching git hash) so they never drift from
# the binary being published below. A release that can't produce these
# images fails loudly here, before anything is tagged, pushed, or published.
echo "==> Building project images..."
if ! "${SCRIPT_DIR}/script/build-project-images"; then
echo "Error: failed to build huskies-project-* images at ${VERSION}."
echo "Release aborted — nothing was tagged, pushed, or published."
exit 1
fi
# ── Package ────────────────────────────────────────────────────
DIST="target/dist"
rm -rf "$DIST"
@@ -281,7 +294,13 @@ echo "$RELEASE_BODY"
# ── Tag & Push ─────────────────────────────────────────────────
echo "==> Tagging ${TAG}..."
git tag -a "$TAG" -m "Release ${TAG}"
git push origin "$TAG"
# Push the branch (with the version-bump commit) and the tag together, so
# the remote branch never lags the release tag. --atomic means both refs
# land or neither does, avoiding a pushed tag pointing at an unpushed commit.
BRANCH="$(git rev-parse --abbrev-ref HEAD)"
echo "==> Pushing ${BRANCH} and ${TAG}..."
git push --atomic origin "$BRANCH" "$TAG"
# ── Create Gitea Release ──────────────────────────────────────
echo "==> Creating release on Gitea..."
+12 -10
View File
@@ -11,10 +11,12 @@ export GIT_CONFIG_VALUE_0=master
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
PROJECT_ROOT="$(cd "$SCRIPT_DIR/.." && pwd)"
# Ordered fail-fast: cheapest deterministic checks first, slowest builds and
# test suites last. `set -euo pipefail` aborts at the first failure, so a fmt
# or clippy drift never wastes time on a frontend build or a multi-minute
# test run.
# Ordered fail-fast: cheapest deterministic checks first. The frontend build
# must run *before* anything that compiles Rust, because story 1113 introduced
# a compile-time dependency on `frontend/dist/` via `rust-embed` — a fresh
# merge worktree without that directory will fail `cargo clippy` on
# `EmbeddedAssets::iter()` before the frontend build has a chance to populate
# it. `set -euo pipefail` aborts at the first failure.
echo "=== Checking Rust formatting ==="
if cargo fmt --version &>/dev/null; then
@@ -44,12 +46,6 @@ if [ "$_dup_found" -eq 1 ]; then
exit 1
fi
echo "=== Running cargo clippy ==="
cargo clippy --manifest-path "$PROJECT_ROOT/Cargo.toml" --all-targets --all-features -- -D warnings
echo "=== Checking doc coverage on changed files ==="
cargo run --manifest-path "$PROJECT_ROOT/Cargo.toml" -p source-map-gen --bin source-map-check --quiet -- --worktree "$PROJECT_ROOT" --base master
echo "=== Building frontend ==="
if [ -d "$PROJECT_ROOT/frontend" ]; then
cd "$PROJECT_ROOT/frontend"
@@ -75,6 +71,12 @@ else
echo "Skipping frontend build (no frontend directory)"
fi
echo "=== Running cargo clippy ==="
cargo clippy --manifest-path "$PROJECT_ROOT/Cargo.toml" --all-targets --all-features -- -D warnings
echo "=== Checking doc coverage on changed files ==="
cargo run --manifest-path "$PROJECT_ROOT/Cargo.toml" -p source-map-gen --bin source-map-check --quiet -- --worktree "$PROJECT_ROOT" --base master
echo "=== Running Rust tests ==="
cargo test --manifest-path "$PROJECT_ROOT/Cargo.toml" --bin huskies
cargo test --manifest-path "$PROJECT_ROOT/Cargo.toml" -p source-map-gen
+7 -6
View File
@@ -1,6 +1,6 @@
[package]
name = "huskies"
version = "0.12.0"
version = "0.14.5"
edition = "2024"
build = "build.rs"
@@ -13,12 +13,10 @@ chrono-tz = { workspace = true }
futures = { workspace = true }
homedir = { workspace = true }
ignore = { workspace = true }
mime_guess = { workspace = true }
notify = { workspace = true }
poem = { workspace = true, features = ["websocket"] }
portable-pty = { workspace = true }
reqwest = { workspace = true, features = ["json", "stream", "form"] }
rust-embed = { workspace = true }
serde = { workspace = true, features = ["derive"] }
serde_json = { workspace = true }
serde_urlencoded = { workspace = true }
@@ -42,13 +40,15 @@ tokio-tungstenite = { workspace = true }
# against a missing system libsqlite3.
#
# The 0.35 pin is the ceiling: rusqlite 0.37 (matrix-sdk-sqlite) requires
# 0.35.x exactly, and sqlx-sqlite 0.9.0-alpha.1 requires >=0.30, <0.36. Bumping
# this needs one of those upstreams to widen their range first.
# 0.35.x exactly, and sqlx-sqlite 0.9.0 requires >=0.30.1, <0.38. Bumping this
# needs matrix-sdk to move to a newer rusqlite first; sqlx then allows up to
# 0.37.x.
libsqlite3-sys = { version = "0.35.0", features = ["bundled"] }
sqlx = { workspace = true }
wait-timeout = "0.2.1"
bft-json-crdt = { path = "../crates/bft-json-crdt", default-features = false, features = ["bft"] }
source-map-gen = { path = "../crates/source-map-gen" }
release-manifest = { path = "../crates/release-manifest" }
ed25519-dalek = { workspace = true }
rand = { workspace = true }
nutype = { workspace = true }
@@ -64,5 +64,6 @@ check-cfg = ["cfg(feature, values(\"logging-base\"))"]
[dev-dependencies]
tempfile = { workspace = true }
mockito = "1"
mockito = "1.7.2"
filetime = { workspace = true }
serde_yaml = { workspace = true }
@@ -0,0 +1,9 @@
-- Stores a serialized CRDT state snapshot so startup can skip replaying
-- the full op log. Only the single most recent snapshot row is kept.
CREATE TABLE IF NOT EXISTS crdt_snapshot (
id INTEGER PRIMARY KEY CHECK (id = 1),
at_seq INTEGER NOT NULL,
max_rowid INTEGER NOT NULL,
state_json TEXT NOT NULL,
created_at TEXT NOT NULL
);
+18 -2
View File
@@ -61,6 +61,17 @@ pub(super) fn build_agent_app_context(
);
let (reconciliation_tx, _) = broadcast::channel(64);
let (perm_tx, perm_rx) = tokio::sync::mpsc::unbounded_channel();
let permission_registry = crate::service::permission_router::ResponderRegistry::new();
crate::service::permission_router::spawn_permission_router(
perm_rx,
Arc::clone(&permission_registry),
);
let (question_tx, question_rx) = tokio::sync::mpsc::unbounded_channel();
let question_registry = crate::service::question_router::QuestionResponderRegistry::new();
crate::service::question_router::spawn_question_router(
question_rx,
Arc::clone(&question_registry),
);
let timer_store = Arc::new(crate::service::timer::TimerStore::load(
project_root.join(".huskies").join("timers.json"),
));
@@ -74,9 +85,13 @@ pub(super) fn build_agent_app_context(
bot_name: "Agent".to_string(),
bot_user_id: String::new(),
ambient_rooms: Arc::new(std::sync::Mutex::new(std::collections::HashSet::new())),
perm_rx: Arc::new(tokio::sync::Mutex::new(perm_rx)),
pending_perm_replies: Arc::new(tokio::sync::Mutex::new(std::collections::HashMap::new())),
permission_registry,
pending_perm_replies: crate::service::permission_router::PendingPermReplies::new(),
permission_timeout_secs: 120,
remembered_permissions: crate::service::permission_router::RememberedPermissions::new(),
question_registry,
pending_question_replies: crate::service::question_router::PendingQuestionReplies::new(),
question_timeout_secs: 120,
status: agents.status_broadcaster(),
chat_dispatcher: Arc::new(crate::chat::dispatcher::ChatDispatcher::new(1_500)),
});
@@ -90,6 +105,7 @@ pub(super) fn build_agent_app_context(
watcher_tx,
reconciliation_tx,
perm_tx,
question_tx,
qa_app_process: Arc::new(std::sync::Mutex::new(None)),
bot_shutdown: None,
matrix_shutdown_tx: None,
+2 -2
View File
@@ -159,7 +159,7 @@ pub(super) async fn detect_conflicts(
our_claims.remove(&story_id);
// Stop any local agent for this story by looking up its name.
if let Ok(agent_list) = agents.list_agents() {
if let Ok(agent_list) = agents.list_agents().await {
for info in agent_list {
if info.story_id == story_id {
let _ = agents
@@ -219,7 +219,7 @@ pub(super) fn reclaim_timed_out_work(_project_root: &Path) {
/// Check for completed agents, push their feature branches to the remote,
/// and report completion via CRDT.
pub(super) async fn check_completions_and_push(agents: &AgentPool, _project_root: &Path) {
let Ok(agent_list) = agents.list_agents() else {
let Ok(agent_list) = agents.list_agents().await else {
return;
};
+19
View File
@@ -213,6 +213,15 @@ pub async fn run(
// Track which stories we've claimed so we can detect conflicts.
let mut our_claims: HashMap<String, f64> = HashMap::new();
// Low-disk-space watchdog (story 1200 AC1): tracks rate-limit/recovery
// state across loop iterations. Thresholds come from the config loaded
// at startup; host_id identifies this sled in chat messages and the
// gateway dedupe key.
let mut disk_watch_state = crate::service::disk_watch::DiskWatchState::default();
let disk_watch_host_id =
crdt_state::our_node_id().unwrap_or_else(|| "unknown-host".to_string());
let disk_watch_status = agents.status_broadcaster();
// Main loop: heartbeat, scan, claim, detect conflicts.
let mut interval = tokio::time::interval(std::time::Duration::from_secs(SCAN_INTERVAL_SECS));
loop {
@@ -221,6 +230,16 @@ pub async fn run(
// Write heartbeat.
write_heartbeat(&rendezvous_url, port);
// Low-disk-space check (story 1200 AC1): every tick period.
crate::service::disk_watch::io::check_and_notify(
&project_root,
&config.disk_watch,
&mut disk_watch_state,
&watcher_tx,
&disk_watch_status,
&disk_watch_host_id,
);
// Scan CRDT for claimable work.
scan_and_claim(&agents, &project_root, &mut our_claims).await;
+58 -24
View File
@@ -33,16 +33,28 @@ impl GateFailureKind {
/// Called once when a gate fails to produce a typed kind. Downstream code
/// matches on the variant and must not call this on subsequent reads.
pub fn classify(output: &str) -> Self {
// Strip `test <name> ... ok` lines before checking lint-trigger keywords so
// a passing test whose name contains e.g. `missing_doc_comments` or `clippy::`
// does not produce a false-positive Lint classification (story 1101).
let stripped_for_lint: String = output
.lines()
.filter(|l| {
let t = l.trim();
!(t.starts_with("test ") && t.ends_with("... ok"))
})
.collect::<Vec<_>>()
.join("\n");
let is_lint = stripped_for_lint.contains("error[clippy::")
|| stripped_for_lint.contains("warning[clippy::")
|| stripped_for_lint.contains("missing_doc_comments");
if output.contains("CONFLICT (content):") || output.contains("Merge conflict:") {
GateFailureKind::ContentConflict
} else if output.contains("Diff in ") || output.contains("would reformat") {
GateFailureKind::Fmt
} else if output.contains("missing-docs direction") {
GateFailureKind::SourceMapCheck
} else if output.contains("error[clippy::")
|| output.contains("warning[clippy::")
|| output.contains("missing_doc_comments")
{
} else if is_lint {
GateFailureKind::Lint
} else if output.contains("error[E") {
// rustc compile errors (e.g. `error[E0063]: missing field`).
@@ -435,26 +447,35 @@ mod tests {
use super::*;
fn init_git_repo(repo: &std::path::Path) {
Command::new("git")
.args(["init"])
.current_dir(repo)
.output()
.unwrap();
Command::new("git")
.args(["config", "user.email", "test@test.com"])
.current_dir(repo)
.output()
.unwrap();
Command::new("git")
.args(["config", "user.name", "Test"])
.current_dir(repo)
.output()
.unwrap();
Command::new("git")
.args(["commit", "--allow-empty", "-m", "init"])
.current_dir(repo)
.output()
.unwrap();
use crate::git_test_support::git_ok;
git_ok(
Command::new("git")
.args(["init"])
.current_dir(repo)
.output(),
"git init",
);
git_ok(
Command::new("git")
.args(["config", "user.email", "test@test.com"])
.current_dir(repo)
.output(),
"git config user.email",
);
git_ok(
Command::new("git")
.args(["config", "user.name", "Test"])
.current_dir(repo)
.output(),
"git config user.name",
);
git_ok(
Command::new("git")
.args(["commit", "--allow-empty", "-m", "init"])
.current_dir(repo)
.output(),
"git commit",
);
}
// ── find_duplicate_module_files tests ────────────────────────
@@ -871,6 +892,19 @@ mod tests {
);
}
/// Story 1101: a passing test whose name contains a lint trigger keyword
/// must NOT produce a Lint classification.
#[test]
fn classify_does_not_false_positive_on_test_name_substring() {
let output = "test agents::gates::tests::classify_lint_from_missing_doc_comments ... ok\n\
test result: ok. 1 passed; 0 failed";
assert_ne!(
GateFailureKind::classify(output),
GateFailureKind::Lint,
"passing test name containing 'missing_doc_comments' must not classify as Lint"
);
}
#[test]
fn classify_source_map_check_from_missing_docs_direction() {
assert_eq!(
+58 -21
View File
@@ -121,7 +121,7 @@ pub fn move_story_to_done(story_id: &str) -> Result<(), String> {
Stage::Merge { .. } => PipelineEvent::MergeSucceeded {
merge_commit: GitSha("accepted".to_string()),
},
Stage::MergeFailure { .. } => PipelineEvent::Accepted,
Stage::MergeFailure { .. } | Stage::MergeFailureFinal { .. } => PipelineEvent::Accepted,
Stage::Coding { .. } | Stage::Qa | Stage::Backlog => PipelineEvent::Close,
_ => {
return Err(format!(
@@ -627,6 +627,34 @@ mod tests {
);
}
/// Regression test (story 1178): a story in `Stage::MergeFailureFinal`
/// whose merge is later retried and succeeds must be movable to Done.
/// Before this fix, `move_story_to_done` had no arm for
/// `MergeFailureFinal`, so it always returned an error even after a
/// real, successful re-merge — the exact "trap state" this story fixes.
#[test]
fn move_story_to_done_from_merge_failure_final_succeeds() {
crate::db::ensure_content_store();
crate::db::write_item_with_content(
"99952_story_merge_failure_final",
"merge_failure_final",
"---\nname: Merge Failure Final Test\n---\n# Story\n",
crate::db::ItemMeta::named("Merge Failure Final Test"),
);
move_story_to_done("99952_story_merge_failure_final")
.expect("move_story_to_done should succeed from MergeFailureFinal");
let item = crate::pipeline_state::read_typed("99952_story_merge_failure_final")
.expect("CRDT read should succeed")
.expect("item should exist in CRDT");
assert_eq!(
item.stage.dir_name(),
"done",
"item should be in done after move from MergeFailureFinal"
);
}
// ── item_type_from_id tests ────────────────────────────────────────────────
#[test]
@@ -793,26 +821,35 @@ mod tests {
// ── feature_branch_has_unmerged_changes tests ────────────────────────────
fn init_git_repo(repo: &std::path::Path) {
Command::new("git")
.args(["init"])
.current_dir(repo)
.output()
.unwrap();
Command::new("git")
.args(["config", "user.email", "test@test.com"])
.current_dir(repo)
.output()
.unwrap();
Command::new("git")
.args(["config", "user.name", "Test"])
.current_dir(repo)
.output()
.unwrap();
Command::new("git")
.args(["commit", "--allow-empty", "-m", "init"])
.current_dir(repo)
.output()
.unwrap();
use crate::git_test_support::git_ok;
git_ok(
Command::new("git")
.args(["init"])
.current_dir(repo)
.output(),
"git init",
);
git_ok(
Command::new("git")
.args(["config", "user.email", "test@test.com"])
.current_dir(repo)
.output(),
"git config user.email",
);
git_ok(
Command::new("git")
.args(["config", "user.name", "Test"])
.current_dir(repo)
.output(),
"git config user.name",
);
git_ok(
Command::new("git")
.args(["commit", "--allow-empty", "-m", "init"])
.current_dir(repo)
.output(),
"git commit",
);
}
/// Bug 226: feature_branch_has_unmerged_changes returns true when the
+9 -1
View File
@@ -4,7 +4,7 @@ use serde::{Deserialize, Serialize};
mod squash;
pub(crate) use squash::run_squash_merge;
pub(crate) use squash::{merge_lock_is_free, run_squash_merge};
/// Typed outcome of a completed squash-merge operation.
///
@@ -103,6 +103,9 @@ pub struct MergeJob {
/// than the current server's boot time. This survives `rebuild_and_restart`
/// (which re-execs and keeps the same PID).
pub server_start_time: f64,
/// Unix timestamp (seconds) when this merge job started, used to compute
/// elapsed time for a still-`Running` job.
pub started_at: f64,
}
/// Result of a mergemaster merge operation.
@@ -113,6 +116,11 @@ pub struct MergeReport {
pub result: MergeResult,
pub worktree_cleaned_up: bool,
pub story_archived: bool,
/// Path (relative to the project root) of the full untruncated report
/// written by `service::merge::io::write_merge_report`, if the write
/// succeeded.
#[serde(default)]
pub report_path: Option<String>,
}
#[cfg(test)]
+22 -9
View File
@@ -17,6 +17,27 @@ use crate::config::ProjectConfig;
/// causing `git cherry-pick merge-queue/…` to fail with "bad revision".
static MERGE_LOCK: Mutex<()> = Mutex::new(());
/// Returns `true` when no squash-merge is currently running, i.e. the merge
/// lock is free.
///
/// Used by the build-directory GC pass (story 1199) to decide whether
/// `.huskies/merge_workspace` is safe to reclaim. This is a best-effort,
/// momentary check — the lock is not held across the reclaim itself, so a
/// merge that starts immediately afterward can still race with GC. That's
/// acceptable: the GC pass tolerates races and skips on error rather than
/// failing the whole pass.
pub(crate) fn merge_lock_is_free() -> bool {
MERGE_LOCK.try_lock().is_ok()
}
/// Resolve the base branch for `project_root` from config, or auto-detect it.
fn resolve_base_branch(project_root: &Path) -> String {
let configured = crate::config::ProjectConfig::load(project_root)
.ok()
.and_then(|c| c.base_branch);
crate::worktree::resolve_base_branch(project_root, configured.as_deref())
}
pub(crate) fn run_squash_merge(
project_root: &Path,
branch: &str,
@@ -31,10 +52,7 @@ pub(crate) fn run_squash_merge(
// A zero-commit branch produces an empty squash and a silent "nothing to
// commit" failure. Catch it early with a grep-able error before any merge
// work starts.
let base_branch = crate::config::ProjectConfig::load(project_root)
.ok()
.and_then(|c| c.base_branch.clone())
.unwrap_or_else(|| "master".to_string());
let base_branch = resolve_base_branch(project_root);
let ahead_out = Command::new("git")
.args(["rev-list", "--count", &format!("{base_branch}..{branch}")])
@@ -316,11 +334,6 @@ pub(crate) fn run_squash_merge(
.map(|o| String::from_utf8_lossy(&o.stdout).trim().to_string())
.unwrap_or_default();
let base_branch = crate::config::ProjectConfig::load(project_root)
.ok()
.and_then(|c| c.base_branch.clone())
.unwrap_or_else(|| "master".to_string());
if current_branch != base_branch {
all_output.push_str(&format!(
"=== VERIFICATION FAILED: expected branch '{base_branch}' but HEAD is on \
@@ -3,26 +3,35 @@ use super::*;
use std::process::Command;
fn init_git_repo(repo: &std::path::Path) {
Command::new("git")
.args(["init"])
.current_dir(repo)
.output()
.unwrap();
Command::new("git")
.args(["config", "user.email", "test@test.com"])
.current_dir(repo)
.output()
.unwrap();
Command::new("git")
.args(["config", "user.name", "Test"])
.current_dir(repo)
.output()
.unwrap();
Command::new("git")
.args(["commit", "--allow-empty", "-m", "init"])
.current_dir(repo)
.output()
.unwrap();
use crate::git_test_support::git_ok;
git_ok(
Command::new("git")
.args(["init"])
.current_dir(repo)
.output(),
"git init",
);
git_ok(
Command::new("git")
.args(["config", "user.email", "test@test.com"])
.current_dir(repo)
.output(),
"git config user.email",
);
git_ok(
Command::new("git")
.args(["config", "user.name", "Test"])
.current_dir(repo)
.output(),
"git config user.name",
);
git_ok(
Command::new("git")
.args(["commit", "--allow-empty", "-m", "init"])
.current_dir(repo)
.output(),
"git commit",
);
}
#[tokio::test]
+102 -20
View File
@@ -3,26 +3,35 @@ use super::*;
use std::process::Command;
fn init_git_repo(repo: &std::path::Path) {
Command::new("git")
.args(["init"])
.current_dir(repo)
.output()
.unwrap();
Command::new("git")
.args(["config", "user.email", "test@test.com"])
.current_dir(repo)
.output()
.unwrap();
Command::new("git")
.args(["config", "user.name", "Test"])
.current_dir(repo)
.output()
.unwrap();
Command::new("git")
.args(["commit", "--allow-empty", "-m", "init"])
.current_dir(repo)
.output()
.unwrap();
use crate::git_test_support::git_ok;
git_ok(
Command::new("git")
.args(["init"])
.current_dir(repo)
.output(),
"git init",
);
git_ok(
Command::new("git")
.args(["config", "user.email", "test@test.com"])
.current_dir(repo)
.output(),
"git config user.email",
);
git_ok(
Command::new("git")
.args(["config", "user.name", "Test"])
.current_dir(repo)
.output(),
"git config user.name",
);
git_ok(
Command::new("git")
.args(["commit", "--allow-empty", "-m", "init"])
.current_dir(repo)
.output(),
"git commit",
);
}
#[tokio::test]
@@ -178,6 +187,79 @@ async fn squash_merge_clean_merge_succeeds() {
);
}
#[tokio::test]
async fn squash_merge_succeeds_on_main_based_repo_with_base_branch_unset() {
use std::fs;
use tempfile::tempdir;
let tmp = tempdir().unwrap();
let repo = tmp.path();
// Repo whose default branch is `main` — no `master` branch exists at all,
// and no `.huskies/project.toml` sets `base_branch`. run_squash_merge must
// auto-detect `main` instead of assuming `master` (bug 1176).
Command::new("git")
.args(["init", "-b", "main"])
.current_dir(repo)
.output()
.unwrap();
Command::new("git")
.args(["config", "user.email", "test@test.com"])
.current_dir(repo)
.output()
.unwrap();
Command::new("git")
.args(["config", "user.name", "Test"])
.current_dir(repo)
.output()
.unwrap();
Command::new("git")
.args(["commit", "--allow-empty", "-m", "init"])
.current_dir(repo)
.output()
.unwrap();
Command::new("git")
.args(["checkout", "-b", "feature/story-main_test"])
.current_dir(repo)
.output()
.unwrap();
fs::write(repo.join("new_file.txt"), "new content").unwrap();
Command::new("git")
.args(["add", "."])
.current_dir(repo)
.output()
.unwrap();
Command::new("git")
.args(["commit", "-m", "add new file"])
.current_dir(repo)
.output()
.unwrap();
Command::new("git")
.args(["checkout", "main"])
.current_dir(repo)
.output()
.unwrap();
let result = run_squash_merge(repo, "feature/story-main_test", "main_test").unwrap();
assert!(
matches!(
result,
super::MergeResult::Success {
conflicts_resolved: false,
..
}
),
"clean merge should succeed on a main-based repo; got: {result:?}"
);
assert!(
repo.join("new_file.txt").exists(),
"merged file should exist on main"
);
}
#[tokio::test]
async fn squash_merge_nonexistent_branch_fails() {
use tempfile::tempdir;
@@ -73,7 +73,7 @@ mod tests {
// task eventually fails.
pool.auto_assign_available_work(tmp.path()).await;
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.try_lock().unwrap();
let has_pending = agents.values().any(|a| {
a.agent_name == "coder-1"
&& matches!(a.status, AgentStatus::Pending | AgentStatus::Running)
@@ -115,7 +115,7 @@ mod tests {
pool.auto_assign_available_work(root).await;
// No agent should have been started for the spike.
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.try_lock().unwrap();
assert!(
agents.is_empty(),
"No agents should be assigned to a spike with review_hold"
@@ -155,7 +155,7 @@ mod tests {
pool.auto_assign_available_work(tmp.path()).await;
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.try_lock().unwrap();
// coder-1 must NOT have been assigned to the QA story (wrong stage).
let coder_assigned_to_qa = agents.iter().any(|(key, a)| {
key.contains("9930_story_qa1")
@@ -209,7 +209,7 @@ mod tests {
pool.auto_assign_available_work(tmp.path()).await;
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.try_lock().unwrap();
// coder-1 should have been picked (it matches the stage and is preferred).
let coder1_assigned = agents.values().any(|a| {
a.agent_name == "coder-1"
@@ -262,7 +262,7 @@ mod tests {
// Must not panic.
pool.auto_assign_available_work(tmp.path()).await;
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.try_lock().unwrap();
// No agent should be assigned to the specific QA story (coder-1 may
// be assigned to leaked 2_current items from the global CRDT store).
let assigned_to_qa_story = agents.iter().any(|(key, a)| {
@@ -301,7 +301,7 @@ mod tests {
let pool = AgentPool::new_test(3001);
pool.auto_assign_available_work(root).await;
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.try_lock().unwrap();
// Filter to only agents assigned to our specific story to avoid
// interference from other tests sharing the global CRDT store.
let assigned_to_our_story = agents.iter().any(|(key, a)| {
@@ -347,7 +347,7 @@ mod tests {
let pool = AgentPool::new_test(3001);
pool.auto_assign_available_work(root).await;
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.try_lock().unwrap();
let has_pending = agents.values().any(|a| {
matches!(
a.status,
@@ -553,7 +553,7 @@ mod tests {
let _ = tokio::join!(t1, t2);
// At most one Pending/Running entry should exist for coder-1.
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.try_lock().unwrap();
let active_coder_count = agents
.values()
.filter(|a| {
@@ -602,7 +602,7 @@ mod tests {
pool.auto_assign_available_work(tmp.path()).await;
let count_after_first = {
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.try_lock().unwrap();
agents
.iter()
.filter(|(key, a)| {
@@ -616,7 +616,7 @@ mod tests {
pool.auto_assign_available_work(tmp.path()).await;
let count_after_second = {
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.try_lock().unwrap();
agents
.iter()
.filter(|(key, a)| {
+123 -27
View File
@@ -65,27 +65,43 @@ impl AgentPool {
// AC6: Detect empty-diff stories before starting the merge pipeline.
// If the worktree has no commits on the feature branch, block the
// story immediately via the state machine — no merge job needed.
if let Some(wt_path) = worktree::find_worktree_path(project_root, story_id)
&& !crate::agents::gates::worktree_has_committed_work(&wt_path)
{
let empty_diff_reason = "Feature branch has no code changes — the coder agent \
did not produce any commits.";
slog_warn!(
"[auto-assign] Story '{story_id}' in 4_merge/ has no commits \
on feature branch. Blocking via state machine."
);
if let Err(e) =
crate::agents::lifecycle::transition_to_blocked(story_id, empty_diff_reason)
{
slog_error!("[auto-assign] Failed to transition '{story_id}' to Blocked: {e}");
//
// Bug 1170: worktree_has_committed_work shells out to `git log`
// synchronously. assign_merge_stage runs on the shared tokio
// runtime (it's invoked reactively on every CRDT transition, incl.
// unblock), so calling it inline here blocked a runtime worker
// thread for the duration of the git subprocess — with a story
// whose worktree/agent had crashed, that call could hang
// indefinitely and stall /health and the liveness heartbeat along
// with it. Run it on the blocking-thread pool instead.
if let Some(wt_path) = worktree::find_worktree_path(project_root, story_id) {
let has_commits = tokio::task::spawn_blocking(move || {
crate::agents::gates::worktree_has_committed_work(&wt_path)
})
.await
.unwrap_or(false);
if !has_commits {
let empty_diff_reason = "Feature branch has no code changes — the coder agent \
did not produce any commits.";
slog_warn!(
"[auto-assign] Story '{story_id}' in 4_merge/ has no commits \
on feature branch. Blocking via state machine."
);
if let Err(e) =
crate::agents::lifecycle::transition_to_blocked(story_id, empty_diff_reason)
{
slog_error!(
"[auto-assign] Failed to transition '{story_id}' to Blocked: {e}"
);
}
let _ = self
.watcher_tx
.send(crate::io::watcher::WatcherEvent::StoryBlocked {
story_id: story_id.to_string(),
reason: empty_diff_reason.to_string(),
});
continue;
}
let _ = self
.watcher_tx
.send(crate::io::watcher::WatcherEvent::StoryBlocked {
story_id: story_id.to_string(),
reason: empty_diff_reason.to_string(),
});
continue;
}
// Skip if a merge job is already running for this story (e.g. triggered
@@ -99,13 +115,7 @@ impl AgentPool {
// Skip if an explicit mergemaster LLM agent is already running
// (operator-driven failure recovery path).
let has_mergemaster = {
let agents = match self.agents.lock() {
Ok(a) => a,
Err(e) => {
slog_error!("[auto-assign] Failed to lock agents: {e}");
break;
}
};
let agents = self.agents.lock().await;
is_story_assigned_for_stage(config, &agents, story_id, &PipelineStage::Mergemaster)
};
if has_mergemaster {
@@ -117,3 +127,89 @@ impl AgentPool {
}
}
}
#[cfg(test)]
mod tests {
use super::super::super::AgentPool;
use crate::config::ProjectConfig;
use std::sync::Arc;
use std::sync::atomic::{AtomicU64, Ordering};
/// Bug 1170 regression: `assign_merge_stage` used to call
/// `worktree_has_committed_work` (which shells out to `git`) directly on
/// the async runtime. For a story with a crashed/unassignable agent
/// sitting in `4_merge/`, that synchronous subprocess call had no yield
/// point, so on a runtime with few worker threads it starved every other
/// task — including the liveness heartbeat and `/health` — for the whole
/// scan. After wrapping the call in `spawn_blocking`, the executor stays
/// free to interleave other work while the git subprocess runs
/// off-runtime.
///
/// This reproduces the unblock → merge-auto-assign path: a story sits in
/// `4_merge/` with no active agent entry (the crashed/unassignable case)
/// and a worktree directory that isn't a real git repo, forcing every
/// `git` invocation in the scan to fail — but only after paying the
/// process fork/exec cost, which is what stalls a non-yielding runtime.
#[tokio::test(flavor = "multi_thread", worker_threads = 1)]
async fn assign_merge_stage_does_not_stall_liveness_heartbeat() {
crate::db::ensure_content_store();
let tmp = tempfile::tempdir().unwrap();
let root = tmp.path().to_path_buf();
std::fs::create_dir_all(root.join(".huskies")).unwrap();
std::fs::write(root.join(".huskies/project.toml"), "").unwrap();
let worktrees_dir = root.join(".huskies/worktrees");
std::fs::create_dir_all(&worktrees_dir).unwrap();
// Simulate several stories stuck in 4_merge/ with a crashed/unassignable
// agent: each has a worktree directory (so find_worktree_path succeeds
// and the git-shelling check runs) but is not a real git repo and has
// no active agent entry in the pool.
const STORY_COUNT: usize = 25;
for i in 0..STORY_COUNT {
let story_id = format!("11700_merge_{i:03}");
std::fs::create_dir_all(worktrees_dir.join(&story_id)).unwrap();
crate::db::write_item_with_content(
&story_id,
"4_merge",
"---\nname: Crashed Merge\n---\n",
crate::db::ItemMeta::named("Crashed Merge"),
);
}
let pool = AgentPool::new_test(3200);
let config = ProjectConfig::load(&root).unwrap_or_default();
// Stand in for the liveness heartbeat (tick_loop.rs's
// spawn_liveness_tick) and /health polling: a tight-interval task
// racing the merge scan on the single-worker-thread runtime.
let ticks = Arc::new(AtomicU64::new(0));
let ticks_clone = Arc::clone(&ticks);
let heartbeat = tokio::spawn(async move {
loop {
tokio::time::sleep(std::time::Duration::from_millis(2)).await;
ticks_clone.fetch_add(1, Ordering::SeqCst);
}
});
let start = std::time::Instant::now();
pool.assign_merge_stage(&root, &config).await;
let elapsed = start.elapsed();
heartbeat.abort();
let observed_ticks = ticks.load(Ordering::SeqCst);
// With a 2ms heartbeat cadence, an unstalled runtime should have
// fired roughly elapsed/2ms ticks. Require at least a quarter of that
// as a generous floor — a stalled runtime (pre-fix) produces ~0 ticks
// because the single worker thread never yields during the scan.
let expected_min_ticks = (elapsed.as_millis() / 2 / 4) as u64;
assert!(
observed_ticks >= expected_min_ticks,
"liveness heartbeat stalled during assign_merge_stage: {observed_ticks} tick(s) \
over {elapsed:?} (expected at least ~{expected_min_ticks}); the merge scan likely \
blocked the tokio runtime instead of yielding via spawn_blocking"
);
}
}
@@ -1,18 +1,43 @@
//! TransitionFired subscriber that auto-blocks stories after N consecutive MergeFailure transitions.
//! TransitionFired subscriber that owns the consecutive-MergeFailure budget:
//! auto-blocks stories at the threshold and auto-retries `GatesFailed`
//! failures below it.
//!
//! Listens on the pipeline transition broadcast channel and, for each story,
//! counts how many times it has entered [`Stage::MergeFailure`] consecutively.
//! When the count reaches the configurable threshold (default 3), the story is
//! transitioned to [`Stage::Blocked`] with a reason that names the failure kind.
//! One counter drives two policies sharing the `merge_failure_block_threshold`
//! budget (default 3):
//!
//! The counter for a story resets whenever a non-`MergeFailure` transition fires
//! for that story (e.g. after a successful merge or a `FixupRequested` demotion
//! back to coding).
//! - **Below the threshold**, a `GatesFailed` failure schedules a delayed
//! re-trigger of the deterministic server-side merge (story 1185) — gates
//! failures are dominated by transients (flaky tests, stale base) that a
//! plain re-run fixes. Other kinds still count toward the budget but are
//! not retried: `ConflictDetected` has its own mergemaster recovery path via
//! [`super::merge_failure_subscriber`]; `EmptyDiff`/`NoCommits`/`Other`
//! require human intervention.
//! - **At the threshold**, the story is transitioned to [`Stage::Blocked`]
//! with a reason naming the failure kind.
//!
//! The counter resets when the story leaves `MergeFailure` for a real reason
//! (successful merge, `FixupRequested`, `Block`), but **not** on
//! [`PipelineEvent::MergeRetryStarted`] — that is the `MergeFailure → Merge`
//! bounce a retry itself causes. Treating it as a reset made the budget
//! unreachable and let a deterministic gates failure retry forever (1185
//! review finding 1); counting across the bounce is what makes the budget
//! real.
//!
//! Bug 1025: while a mergemaster is actively running on the story, its
//! iteration loop (squash → fail → fix → retry) generates multiple
//! MergeFailure transitions. Those are NOT consecutive give-ups — they are
//! recovery iterations in progress. We neither count nor schedule retries
//! while a mergemaster is in the pool for the story.
use std::collections::HashMap;
use std::path::{Path, PathBuf};
use std::sync::Arc;
use std::sync::Mutex;
use std::time::Duration;
use crate::io::watcher::WatcherEvent;
use crate::pipeline_state::{MergeFailureKind, PipelineEvent, Stage, Status, StoryId};
use crate::slog;
use crate::slog_warn;
@@ -20,42 +45,102 @@ use crate::slog_warn;
use super::super::super::PipelineStage;
use super::super::AgentPool;
use super::scan::is_story_assigned_for_stage;
use super::story_checks::{
has_review_hold, has_unmet_dependencies, is_story_blocked, is_story_frozen,
};
/// Reconcile: no-op for the merge-failure block subscriber.
/// Delay before an auto-retry re-triggers the server-side merge for a
/// `GatesFailed` failure. Gives transient conditions (a concurrently landing
/// master merge, an exhausted runner) a moment to clear; retrying instantly
/// would just replay the same failure.
const AUTO_RETRY_DELAY: Duration = Duration::from_secs(30);
/// Per-story scheduling generation, shared between the subscriber loop and the
/// delayed retry tasks it spawns.
///
/// The block subscriber maintains an in-memory per-story consecutive-failure counter
/// that cannot be reconstructed from CRDT state alone (only the current stage is
/// stored, not the history of how many times each story failed). Eventual consistency
/// is guaranteed by the live subscriber reacting to each new `MergeFailure` event;
/// the periodic reconciler cannot add value here without risking spurious blocks.
/// Every scheduled retry captures the generation current at schedule time; the
/// timer only acts if that generation is still current when it fires. The
/// subscriber bumps the generation on every (re)schedule and clears the entry
/// on counter reset, so stale timers left over from an earlier failure cycle
/// become no-ops instead of firing unaccounted retries (1185 review finding 6).
type Generations = Arc<Mutex<HashMap<String, u64>>>;
/// What the subscriber decided to do about one transition. Split out from the
/// event loop so the counter/budget policy is synchronous and unit-testable.
#[derive(Debug, PartialEq, Eq)]
enum Decision {
/// Nothing to do (not a MergeFailure, recovery in progress, retry bounce,
/// budget disabled, or a non-retryable kind below the threshold).
Nothing,
/// Schedule a delayed auto-retry: this is consecutive failure `attempt` of
/// a `budget`-sized budget, and the kind is `GatesFailed`.
ScheduleRetry { attempt: u32, budget: u32 },
/// The budget is exhausted: block the story.
Block { count: u32 },
}
/// Reconcile: no-op for the periodic pass.
///
/// The consecutive-failure counter is in-memory and cannot be reconstructed
/// from CRDT state (only the current stage is stored, not the failure
/// history). Restart catch-up for stories already parked in
/// `MergeFailure{GatesFailed}` is handled once, at subscriber startup, by
/// [`reconcile_stranded_gates_failed`] — running it from the periodic
/// reconciler instead would re-schedule retries for budget-exhausted stories
/// on every tick, reintroducing the unbounded-retry bug the startup-only scan
/// avoids.
pub(crate) fn reconcile_merge_failure_block() {}
/// Spawn a background task that blocks stories after N consecutive `MergeFailure` transitions.
///
/// Subscribes to the pipeline transition broadcast channel and tracks a per-story
/// consecutive-failure counter. When a story's count reaches the threshold configured
/// in `project.toml` (`merge_failure_block_threshold`, default 3), the story is
/// transitioned to `Stage::Blocked` with a reason that names the failure kind.
///
/// The counter resets when the story leaves `MergeFailure` (e.g. on `FixupRequested`,
/// `ReQueuedForQa`, or a successful merge via `Unblock → Merge → Done`).
///
/// Bug 1025: while a mergemaster is actively running on the story, its
/// iteration loop (squash → fail → fix → retry) generates multiple
/// MergeFailure transitions. Those are NOT consecutive give-ups — they are
/// recovery iterations in progress. We skip counter increments while a
/// mergemaster is in the pool for the story; the counter only increments on
/// transitions that happen with no recovery agent attached.
/// Spawn the background task that owns the consecutive-MergeFailure budget:
/// auto-retry for `GatesFailed` below the threshold, auto-block at it.
pub(crate) fn spawn_merge_failure_block_subscriber(pool: Arc<AgentPool>, project_root: PathBuf) {
let mut rx = crate::pipeline_state::subscribe_transitions();
tokio::spawn(async move {
let mut counters: HashMap<StoryId, (u32, MergeFailureKind)> = HashMap::new();
let generations: Generations = Generations::default();
// One-shot restart catch-up: stories already sitting in GatesFailed
// when the process starts will never fire another transition on their
// own, so without this they'd silently lose auto-retry coverage
// (1185 review finding 3).
reconcile_stranded_gates_failed(&pool, &project_root, &mut counters, &generations).await;
loop {
match rx.recv().await {
Ok(fired) => {
let recovery_running =
is_mergemaster_running(&pool, &project_root, &fired.story_id.0);
on_transition(&project_root, &fired, &mut counters, recovery_running);
is_mergemaster_running(&pool, &project_root, &fired.story_id.0).await;
let threshold = load_threshold(&project_root).await;
match decide(&fired, &mut counters, recovery_running, threshold) {
Decision::Nothing => {
// A real departure from MergeFailure also
// invalidates any pending retry timer.
if fired.after.status() != Status::MergeFailure
&& !matches!(fired.event, PipelineEvent::MergeRetryStarted)
{
invalidate_generation(&generations, &fired.story_id.0);
}
}
Decision::ScheduleRetry { attempt, budget } => {
schedule_auto_retry(
Arc::clone(&pool),
project_root.clone(),
fired.story_id.0.clone(),
attempt,
budget,
Arc::clone(&generations),
);
}
Decision::Block { count } => {
let kind = counters
.get(&fired.story_id)
.map(|(_, k)| k.clone())
.unwrap_or(MergeFailureKind::Other(String::new()));
apply_block(&fired.story_id, count, &kind);
counters.remove(&fired.story_id);
invalidate_generation(&generations, &fired.story_id.0);
}
}
}
Err(tokio::sync::broadcast::error::RecvError::Lagged(n)) => {
slog_warn!(
@@ -69,94 +154,294 @@ pub(crate) fn spawn_merge_failure_block_subscriber(pool: Arc<AgentPool>, project
});
}
/// Return true if a mergemaster agent is currently in the pool for `story_id`.
/// Used to suppress counter increments while recovery is actively iterating
/// (bug 1025).
fn is_mergemaster_running(pool: &AgentPool, project_root: &Path, story_id: &str) -> bool {
let config = match crate::config::ProjectConfig::load(project_root) {
Ok(c) => c,
Err(_) => return false,
};
let agents = match pool.agents.lock() {
Ok(a) => a,
Err(_) => return false,
};
is_story_assigned_for_stage(&config, &agents, story_id, &PipelineStage::Mergemaster)
}
/// Handle a single transition event: update counters and emit Block if threshold is reached.
/// Pure budget policy: given a fired transition and the per-story counter,
/// decide whether to do nothing, schedule a `GatesFailed` auto-retry, or
/// block the story.
///
/// `recovery_running`: when `true`, a mergemaster is currently in the pool for
/// the story and the failure is part of an in-flight recovery loop. We do NOT
/// increment the consecutive-failure counter in that case (bug 1025).
fn on_transition(
project_root: &Path,
/// `threshold == 0` disables both policies (feature off), matching the
/// pre-1185 block-subscriber behaviour.
fn decide(
fired: &crate::pipeline_state::TransitionFired,
counters: &mut HashMap<StoryId, (u32, MergeFailureKind)>,
recovery_running: bool,
) {
threshold: u32,
) -> Decision {
// Story 1086: gate on the typed `Status` projection — `Status::MergeFailure`
// is precisely the set of stages we count toward the block threshold. We
// still need the variant pattern below to read `kind`.
// is precisely the set of stages we count toward the budget.
if fired.after.status() != Status::MergeFailure {
// The MergeFailure → Merge bounce caused by a retry (auto or manual)
// is not a recovery: the budget must survive it, or a deterministic
// failure retries forever (1185 review finding 1).
if !matches!(fired.event, PipelineEvent::MergeRetryStarted) {
counters.remove(&fired.story_id);
}
return Decision::Nothing;
}
let Stage::MergeFailure { kind, .. } = &fired.after else {
counters.remove(&fired.story_id);
return Decision::Nothing;
};
if recovery_running {
slog!(
"[merge-block-sub] Story '{}' MergeFailure while mergemaster is running; \
not counting toward budget (recovery in progress).",
fired.story_id.0
);
return Decision::Nothing;
}
let entry = counters
.entry(fired.story_id.clone())
.or_insert_with(|| (0, kind.clone()));
entry.0 += 1;
entry.1 = kind.clone();
let count = entry.0;
if threshold == 0 {
return Decision::Nothing;
}
if count >= threshold {
return Decision::Block { count };
}
if matches!(kind, MergeFailureKind::GatesFailed(_)) {
return Decision::ScheduleRetry {
attempt: count,
budget: threshold,
};
}
Decision::Nothing
}
/// Transition `story_id` to `Blocked` with a reason naming the failure kind.
fn apply_block(story_id: &StoryId, count: u32, kind: &MergeFailureKind) {
let kind_str = failure_kind_label(kind);
let reason =
format!("Auto-blocked after {count} consecutive MergeFailure ({kind_str}) transitions.");
let story_id = story_id.0.as_str();
slog!(
"[merge-block-sub] Story '{story_id}' reached {count} consecutive \
MergeFailure ({kind_str}); blocking."
);
if let Err(e) =
crate::pipeline_state::apply_transition(story_id, PipelineEvent::Block { reason }, None)
{
slog_warn!("[merge-block-sub] Failed to block '{story_id}': {e}");
}
}
/// Spawn the delayed auto-retry task for one `GatesFailed` failure.
///
/// Bumps the story's scheduling generation so any earlier still-sleeping timer
/// for the story becomes stale and skips itself.
fn schedule_auto_retry(
pool: Arc<AgentPool>,
project_root: PathBuf,
story_id: String,
attempt: u32,
budget: u32,
generations: Generations,
) {
let generation = next_generation(&generations, &story_id);
slog!(
"[merge-block-sub] Story '{story_id}' GatesFailed (attempt {attempt}/{budget}); \
scheduling auto-retry in {AUTO_RETRY_DELAY:?}."
);
tokio::spawn(async move {
tokio::time::sleep(AUTO_RETRY_DELAY).await;
perform_auto_retry(
&pool,
&project_root,
&story_id,
attempt,
budget,
&generations,
generation,
)
.await;
});
}
/// Fire one auto-retry: re-check that acting is still correct, start the
/// server-side merge, and notify chat only when the merge actually started
/// (1185 review finding 7).
///
/// Guards, in order:
/// - the scheduling generation is still current (no newer schedule/reset);
/// - the story is still parked in `MergeFailure{GatesFailed}`;
/// - the story is not frozen/blocked/on hold/dep-blocked — the same
/// eligibility gates `assign_merge_stage` applies (1185 review finding 4);
/// - no mergemaster is actively recovering the story (1185 review finding 2).
async fn perform_auto_retry(
pool: &Arc<AgentPool>,
project_root: &Path,
story_id: &str,
attempt: u32,
budget: u32,
generations: &Generations,
generation: u64,
) {
if !is_generation_current(generations, story_id, generation) {
slog!(
"[merge-block-sub] Story '{story_id}' auto-retry ({attempt}/{budget}) is stale \
(superseded or reset); skipping."
);
return;
}
match &fired.after {
Stage::MergeFailure { kind, .. } => {
if recovery_running {
slog!(
"[merge-block-sub] Story '{}' MergeFailure while mergemaster is running; \
not counting toward block threshold (recovery in progress).",
fired.story_id.0
);
return;
}
let entry = counters
.entry(fired.story_id.clone())
.or_insert_with(|| (0, kind.clone()));
entry.0 += 1;
entry.1 = kind.clone();
let count = entry.0;
let threshold = load_threshold(project_root);
if threshold == 0 {
return;
let still_gates_failed = matches!(
crate::pipeline_state::read_typed(story_id),
Ok(Some(item)) if matches!(
item.stage,
Stage::MergeFailure {
kind: MergeFailureKind::GatesFailed(_),
..
}
)
);
if !still_gates_failed {
slog!(
"[merge-block-sub] Story '{story_id}' left GatesFailed before auto-retry \
({attempt}/{budget}) fired; skipping."
);
return;
}
if count >= threshold {
let kind_str = failure_kind_label(kind);
let reason = format!(
"Auto-blocked after {count} consecutive MergeFailure ({kind_str}) transitions."
);
let story_id = fired.story_id.0.as_str();
slog!(
"[merge-block-sub] Story '{story_id}' reached {count} consecutive \
MergeFailure ({kind_str}); blocking."
);
if let Err(e) = crate::pipeline_state::apply_transition(
story_id,
PipelineEvent::Block { reason },
None,
) {
slog_warn!("[merge-block-sub] Failed to block '{story_id}': {e}");
} else {
counters.remove(&fired.story_id);
}
}
if has_review_hold(story_id)
|| is_story_frozen(story_id)
|| is_story_blocked(story_id)
|| has_unmet_dependencies(story_id)
{
slog!(
"[merge-block-sub] Story '{story_id}' is held/frozen/blocked/dep-blocked; \
skipping auto-retry ({attempt}/{budget})."
);
return;
}
if is_mergemaster_running(pool, project_root, story_id).await {
slog!(
"[merge-block-sub] Story '{story_id}' has an active mergemaster; \
skipping auto-retry ({attempt}/{budget}) recovery owns the story."
);
return;
}
match pool.start_merge_agent_work(project_root, story_id) {
Ok(()) => {
slog!(
"[merge-block-sub] Auto-retrying merge for '{story_id}' \
(attempt {attempt}/{budget})."
);
let _ = pool.watcher_tx.send(WatcherEvent::MergeAutoRetry {
story_id: story_id.to_string(),
attempt,
budget,
});
}
_ => {
counters.remove(&fired.story_id);
Err(e) => {
slog_warn!(
"[merge-block-sub] Auto-retry for '{story_id}' ({attempt}/{budget}) \
could not start: {e}; not notifying."
);
}
}
}
/// Load the threshold from project config, falling back to the compiled default.
fn load_threshold(project_root: &Path) -> u32 {
crate::config::ProjectConfig::load(project_root)
.map(|c| c.merge_failure_block_threshold)
.unwrap_or(3)
/// One-shot startup scan: schedule a first auto-retry for every story already
/// parked in `MergeFailure{GatesFailed}`.
///
/// The pre-restart attempt count is unrecoverable, so the counter restarts at
/// 1 — worst case a story gets up to `threshold - 1` extra retries across a
/// restart, still bounded per process lifetime.
async fn reconcile_stranded_gates_failed(
pool: &Arc<AgentPool>,
project_root: &Path,
counters: &mut HashMap<StoryId, (u32, MergeFailureKind)>,
generations: &Generations,
) {
let threshold = load_threshold(project_root).await;
if threshold == 0 {
return;
}
for item in crate::pipeline_state::read_all_typed() {
let Stage::MergeFailure { kind, .. } = &item.stage else {
continue;
};
if !matches!(kind, MergeFailureKind::GatesFailed(_)) {
continue;
}
counters.insert(item.story_id.clone(), (1, kind.clone()));
slog!(
"[merge-block-sub] Story '{}' found parked in GatesFailed at startup; \
scheduling catch-up auto-retry (attempt 1/{threshold}).",
item.story_id.0
);
schedule_auto_retry(
Arc::clone(pool),
project_root.to_path_buf(),
item.story_id.0.clone(),
1,
threshold,
Arc::clone(generations),
);
}
}
/// Bump and return the scheduling generation for `story_id`.
fn next_generation(generations: &Generations, story_id: &str) -> u64 {
let mut map = generations.lock().unwrap_or_else(|p| p.into_inner());
let entry = map.entry(story_id.to_string()).or_insert(0);
*entry += 1;
*entry
}
/// Drop the generation entry for `story_id`, making every pending timer stale.
fn invalidate_generation(generations: &Generations, story_id: &str) {
generations
.lock()
.unwrap_or_else(|p| p.into_inner())
.remove(story_id);
}
/// True when `expected` is still the current scheduling generation.
fn is_generation_current(generations: &Generations, story_id: &str, expected: u64) -> bool {
generations
.lock()
.unwrap_or_else(|p| p.into_inner())
.get(story_id)
== Some(&expected)
}
/// Return true if a mergemaster agent is currently in the pool for `story_id`.
/// Used to suppress counting and retries while recovery is actively iterating
/// (bug 1025).
async fn is_mergemaster_running(pool: &AgentPool, project_root: &Path, story_id: &str) -> bool {
let root = project_root.to_path_buf();
let config = match tokio::task::spawn_blocking(move || {
crate::config::ProjectConfig::load(&root)
})
.await
{
Ok(Ok(c)) => c,
_ => return false,
};
let agents = pool.agents.lock().await;
is_story_assigned_for_stage(&config, &agents, story_id, &PipelineStage::Mergemaster)
}
/// Load the budget from project config off the async runtime (the read is
/// synchronous filesystem I/O — bug 1170 class), falling back to the compiled
/// default.
async fn load_threshold(project_root: &Path) -> u32 {
let root = project_root.to_path_buf();
tokio::task::spawn_blocking(move || {
crate::config::ProjectConfig::load(&root)
.map(|c| c.merge_failure_block_threshold)
.unwrap_or(3)
})
.await
.unwrap_or(3)
}
/// Short human-readable label for a [`MergeFailureKind`] variant.
@@ -178,11 +463,7 @@ mod tests {
use crate::pipeline_state::{BranchName, PipelineEvent, Stage, StoryId, TransitionFired};
use std::num::NonZeroU32;
fn setup_project(tmp: &tempfile::TempDir) {
let sk = tmp.path().join(".huskies");
std::fs::create_dir_all(&sk).unwrap();
std::fs::write(sk.join("project.toml"), "[[agent]]\nname = \"coder\"\n").unwrap();
}
const THRESHOLD: u32 = 3;
fn seed_at_merge(story_id: &str) {
crate::crdt_state::init_for_test();
@@ -233,218 +514,232 @@ mod tests {
}
}
/// AC3 (threshold-not-reached): 2 consecutive failures below threshold of 3 must NOT block.
#[test]
fn below_threshold_does_not_block() {
let tmp = tempfile::tempdir().unwrap();
setup_project(&tmp);
let story_id = "1018_below";
seed_at_merge(story_id);
// Transition to MergeFailure once to establish the stage.
crate::agents::lifecycle::transition_to_merge_failure(
story_id,
MergeFailureKind::GatesFailed("error".to_string()),
)
.expect("initial MergeFailure transition");
let mut counters: HashMap<StoryId, (u32, MergeFailureKind)> = HashMap::new();
let kind = MergeFailureKind::GatesFailed("error".to_string());
// Fire 2 MergeFailure events (default threshold is 3).
for _ in 0..2 {
let fired = make_merge_failure_fired(story_id, kind.clone());
on_transition(tmp.path(), &fired, &mut counters, false);
/// The MergeFailure → Merge bounce a retry causes.
fn make_retry_started_fired(story_id: &str) -> TransitionFired {
TransitionFired {
story_id: StoryId(story_id.to_string()),
before: Stage::MergeFailure {
kind: MergeFailureKind::GatesFailed("error".to_string()),
feature_branch: BranchName("feature/test".to_string()),
commits_ahead: NonZeroU32::new(1).unwrap(),
},
after: Stage::Merge {
feature_branch: BranchName("feature/test".to_string()),
commits_ahead: NonZeroU32::new(1).unwrap(),
claim: None,
retries: 1,
server_start_time: None,
},
event: PipelineEvent::MergeRetryStarted,
at: chrono::Utc::now(),
}
}
// Story must still be in MergeFailure (not Blocked).
let item = crate::pipeline_state::read_typed(story_id)
.expect("read")
.expect("item");
assert!(
matches!(item.stage, Stage::MergeFailure { .. }),
"story must still be in MergeFailure after 2 failures (threshold 3): {:?}",
item.stage
fn gates_failed() -> MergeFailureKind {
MergeFailureKind::GatesFailed("error".to_string())
}
/// Below the threshold, GatesFailed schedules a retry with the right
/// attempt numbering.
#[test]
fn gates_failed_below_threshold_schedules_retry() {
let mut counters = HashMap::new();
let fired = make_merge_failure_fired("t_sched", gates_failed());
assert_eq!(
decide(&fired, &mut counters, false, THRESHOLD),
Decision::ScheduleRetry {
attempt: 1,
budget: THRESHOLD
}
);
assert_eq!(
decide(&fired, &mut counters, false, THRESHOLD),
Decision::ScheduleRetry {
attempt: 2,
budget: THRESHOLD
}
);
}
/// AC3 (threshold-reached): 3 consecutive failures at threshold of 3 must block.
/// 1185 review finding 1 (regression): the retry's own MergeFailure→Merge
/// bounce must NOT reset the counter — the third consecutive failure
/// blocks even though retries happened in between.
#[test]
fn at_threshold_blocks_with_failure_kind_in_reason() {
let tmp = tempfile::tempdir().unwrap();
setup_project(&tmp);
fn merge_retry_started_does_not_reset_counter() {
let mut counters = HashMap::new();
let story = "t_no_reset";
let fail = make_merge_failure_fired(story, gates_failed());
let bounce = make_retry_started_fired(story);
assert!(matches!(
decide(&fail, &mut counters, false, THRESHOLD),
Decision::ScheduleRetry { attempt: 1, .. }
));
assert_eq!(
decide(&bounce, &mut counters, false, THRESHOLD),
Decision::Nothing
);
assert!(matches!(
decide(&fail, &mut counters, false, THRESHOLD),
Decision::ScheduleRetry { attempt: 2, .. }
));
assert_eq!(
decide(&bounce, &mut counters, false, THRESHOLD),
Decision::Nothing
);
// Third consecutive failure: budget exhausted despite the bounces.
assert_eq!(
decide(&fail, &mut counters, false, THRESHOLD),
Decision::Block { count: 3 }
);
}
/// A real departure (FixupRequested → Coding) still resets the counter.
#[test]
fn real_departure_resets_counter() {
let mut counters = HashMap::new();
let story = "t_reset";
let fail = make_merge_failure_fired(story, gates_failed());
decide(&fail, &mut counters, false, THRESHOLD);
decide(&fail, &mut counters, false, THRESHOLD);
assert_eq!(
counters.get(&StoryId(story.to_string())).map(|e| e.0),
Some(2)
);
decide(&make_coding_fired(story), &mut counters, false, THRESHOLD);
assert!(!counters.contains_key(&StoryId(story.to_string())));
// Fresh failures start a fresh budget.
assert!(matches!(
decide(&fail, &mut counters, false, THRESHOLD),
Decision::ScheduleRetry { attempt: 1, .. }
));
}
/// Non-GatesFailed kinds count toward the block budget but never schedule
/// a retry (ConflictDetected has its own mergemaster path; the rest need
/// humans).
#[test]
fn non_gates_failed_counts_but_does_not_retry() {
let mut counters = HashMap::new();
let story = "t_conflict";
let conflict = make_merge_failure_fired(story, MergeFailureKind::ConflictDetected(None));
assert_eq!(
decide(&conflict, &mut counters, false, THRESHOLD),
Decision::Nothing
);
assert_eq!(
counters.get(&StoryId(story.to_string())).map(|e| e.0),
Some(1)
);
assert_eq!(
decide(&conflict, &mut counters, false, THRESHOLD),
Decision::Nothing
);
assert_eq!(
decide(&conflict, &mut counters, false, THRESHOLD),
Decision::Block { count: 3 }
);
}
/// Mixed kinds share one budget: GatesFailed and ConflictDetected
/// interleavings block at the same total count (1185 review finding 5).
#[test]
fn mixed_kinds_share_one_budget() {
let mut counters = HashMap::new();
let story = "t_mixed";
let fail = make_merge_failure_fired(story, gates_failed());
let conflict = make_merge_failure_fired(story, MergeFailureKind::ConflictDetected(None));
assert!(matches!(
decide(&fail, &mut counters, false, THRESHOLD),
Decision::ScheduleRetry { attempt: 1, .. }
));
assert_eq!(
decide(&conflict, &mut counters, false, THRESHOLD),
Decision::Nothing
);
assert_eq!(
decide(&fail, &mut counters, false, THRESHOLD),
Decision::Block { count: 3 }
);
}
/// Bug 1025: recovery in progress neither counts nor schedules.
#[test]
fn mergemaster_running_suppresses_counting_and_retry() {
let mut counters = HashMap::new();
let story = "t_recovery";
let fail = make_merge_failure_fired(story, gates_failed());
for _ in 0..3 {
assert_eq!(
decide(&fail, &mut counters, true, THRESHOLD),
Decision::Nothing
);
}
assert!(!counters.contains_key(&StoryId(story.to_string())));
}
/// threshold == 0 disables both policies.
#[test]
fn threshold_zero_disables_block_and_retry() {
let mut counters = HashMap::new();
let fail = make_merge_failure_fired("t_disabled", gates_failed());
for _ in 0..5 {
assert_eq!(decide(&fail, &mut counters, false, 0), Decision::Nothing);
}
}
/// Applying a Block decision transitions the story and names the kind.
#[test]
fn apply_block_blocks_with_failure_kind_in_reason() {
let story_id = "1018_at_threshold";
seed_at_merge(story_id);
crate::agents::lifecycle::transition_to_merge_failure(
story_id,
MergeFailureKind::GatesFailed("fmt error".to_string()),
)
.expect("initial MergeFailure transition");
let mut counters: HashMap<StoryId, (u32, MergeFailureKind)> = HashMap::new();
let kind = MergeFailureKind::GatesFailed("fmt error".to_string());
// Fire 3 MergeFailure events — the 3rd must trigger the block.
for _ in 0..3 {
let fired = make_merge_failure_fired(story_id, kind.clone());
on_transition(tmp.path(), &fired, &mut counters, false);
}
apply_block(
&StoryId(story_id.to_string()),
3,
&MergeFailureKind::GatesFailed("fmt error".to_string()),
);
let item = crate::pipeline_state::read_typed(story_id)
.expect("read")
.expect("item");
assert!(
matches!(item.stage, Stage::Blocked { .. }),
"story must be Blocked after 3 consecutive MergeFailures: {:?}",
item.stage
);
// The block reason must name the failure kind.
if let Stage::Blocked { reason } = &item.stage {
assert!(
reason.contains("GatesFailed"),
"block reason must name the failure kind: {reason}"
);
match &item.stage {
Stage::Blocked { reason } => {
assert!(
reason.contains("GatesFailed"),
"block reason must name the failure kind: {reason}"
);
}
other => panic!("story must be Blocked: {other:?}"),
}
}
/// AC3 (reset): counter clears after a non-MergeFailure transition.
///
/// 2 failures → FixupRequested reset → 2 more failures: still below threshold, no block.
/// 1185 review finding 6 (regression): a newer schedule or a reset makes
/// earlier timers stale.
#[test]
fn counter_resets_on_non_merge_failure_transition() {
let tmp = tempfile::tempdir().unwrap();
setup_project(&tmp);
let story_id = "1018_reset";
seed_at_merge(story_id);
fn stale_generations_are_not_current() {
let generations: Generations = Generations::default();
let g1 = next_generation(&generations, "s");
assert!(is_generation_current(&generations, "s", g1));
crate::agents::lifecycle::transition_to_merge_failure(
story_id,
MergeFailureKind::ConflictDetected(None),
)
.expect("initial MergeFailure transition");
let g2 = next_generation(&generations, "s");
assert!(!is_generation_current(&generations, "s", g1));
assert!(is_generation_current(&generations, "s", g2));
let mut counters: HashMap<StoryId, (u32, MergeFailureKind)> = HashMap::new();
let kind = MergeFailureKind::ConflictDetected(None);
// Fire 2 MergeFailure events.
for _ in 0..2 {
let fired = make_merge_failure_fired(story_id, kind.clone());
on_transition(tmp.path(), &fired, &mut counters, false);
}
assert_eq!(
counters.get(&StoryId(story_id.to_string())).map(|e| e.0),
Some(2),
"counter must be 2 after 2 failures"
);
// Simulate FixupRequested (non-MergeFailure transition).
let reset_fired = make_coding_fired(story_id);
on_transition(tmp.path(), &reset_fired, &mut counters, false);
assert!(
!counters.contains_key(&StoryId(story_id.to_string())),
"counter must be cleared after non-MergeFailure transition"
);
// Re-seed to MergeFailure so we can apply the block transition.
crate::agents::lifecycle::transition_to_merge_failure(
story_id,
MergeFailureKind::ConflictDetected(None),
)
.expect("re-enter MergeFailure after reset");
// Fire 2 more MergeFailure events — still below threshold.
for _ in 0..2 {
let fired = make_merge_failure_fired(story_id, kind.clone());
on_transition(tmp.path(), &fired, &mut counters, false);
}
let item = crate::pipeline_state::read_typed(story_id)
.expect("read")
.expect("item");
assert!(
matches!(item.stage, Stage::MergeFailure { .. }),
"story must still be in MergeFailure after reset + 2 new failures: {:?}",
item.stage
);
}
/// Bug 1025: while a mergemaster is running, MergeFailure transitions are
/// recovery iterations, not consecutive give-ups. 3 failures with
/// `recovery_running=true` must NOT block.
#[test]
fn mergemaster_running_suppresses_block() {
let tmp = tempfile::tempdir().unwrap();
setup_project(&tmp);
let story_id = "1025_recovery_running";
seed_at_merge(story_id);
crate::agents::lifecycle::transition_to_merge_failure(
story_id,
MergeFailureKind::ConflictDetected(None),
)
.expect("initial MergeFailure transition");
let mut counters: HashMap<StoryId, (u32, MergeFailureKind)> = HashMap::new();
let kind = MergeFailureKind::ConflictDetected(None);
// Fire 3 MergeFailure events WHILE a mergemaster is running (gated).
for _ in 0..3 {
let fired = make_merge_failure_fired(story_id, kind.clone());
on_transition(tmp.path(), &fired, &mut counters, true);
}
// Counter must NOT have incremented at all — recovery in progress.
assert!(
!counters.contains_key(&StoryId(story_id.to_string())),
"counter must not increment while mergemaster is running"
);
// And the story must still be in MergeFailure (not Blocked).
let item = crate::pipeline_state::read_typed(story_id)
.expect("read")
.expect("item");
assert!(
matches!(item.stage, Stage::MergeFailure { .. }),
"story must NOT be blocked while mergemaster is running (recovery in progress): {:?}",
item.stage
);
}
/// Bug 1025 regression guard: the genuinely-stuck case (no mergemaster
/// running) still blocks at the threshold, so the original 1018 behaviour
/// is preserved.
#[test]
fn no_mergemaster_still_blocks_at_threshold() {
let tmp = tempfile::tempdir().unwrap();
setup_project(&tmp);
let story_id = "1025_genuine_stuck";
seed_at_merge(story_id);
crate::agents::lifecycle::transition_to_merge_failure(
story_id,
MergeFailureKind::ConflictDetected(None),
)
.expect("initial MergeFailure transition");
let mut counters: HashMap<StoryId, (u32, MergeFailureKind)> = HashMap::new();
let kind = MergeFailureKind::ConflictDetected(None);
// Fire 3 MergeFailure events with NO mergemaster (recovery_running=false).
for _ in 0..3 {
let fired = make_merge_failure_fired(story_id, kind.clone());
on_transition(tmp.path(), &fired, &mut counters, false);
}
// Story must be Blocked (genuine-stuck case unchanged).
let item = crate::pipeline_state::read_typed(story_id)
.expect("read")
.expect("item");
assert!(
matches!(item.stage, Stage::Blocked { .. }),
"story must still block when no mergemaster is running: {:?}",
item.stage
);
invalidate_generation(&generations, "s");
assert!(!is_generation_current(&generations, "s", g2));
}
}
@@ -100,15 +100,7 @@ async fn on_merge_failure_transition(
};
let agent_name = {
let agents = match pool.agents.lock() {
Ok(a) => a,
Err(e) => {
slog_warn!(
"[merge-failure-sub] Failed to lock agent pool for '{story_id}': {e}"
);
return;
}
};
let agents = pool.agents.lock().await;
if is_story_assigned_for_stage(
&config,
&agents,
@@ -228,7 +220,7 @@ mod tests {
);
on_merge_failure_transition(&pool, tmp.path(), &fired).await;
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.lock().await;
assert!(
agents.iter().any(|(key, a)| {
key.contains(story_id)
@@ -259,7 +251,7 @@ mod tests {
// Give the subscriber time to run (it should do nothing).
tokio::time::sleep(std::time::Duration::from_millis(100)).await;
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.lock().await;
let spawned = agents.iter().any(|(key, a)| {
key.contains(story_id)
&& a.agent_name == "mergemaster"
@@ -287,7 +279,7 @@ mod tests {
tokio::time::sleep(std::time::Duration::from_millis(100)).await;
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.lock().await;
let spawned = agents.iter().any(|(key, a)| {
key.contains(story_id)
&& a.agent_name == "mergemaster"
@@ -315,7 +307,7 @@ mod tests {
tokio::time::sleep(std::time::Duration::from_millis(100)).await;
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.lock().await;
let spawned = agents.iter().any(|(key, a)| {
key.contains(story_id)
&& a.agent_name == "mergemaster"
@@ -343,7 +335,7 @@ mod tests {
tokio::time::sleep(std::time::Duration::from_millis(100)).await;
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.lock().await;
let spawned = agents.iter().any(|(key, a)| {
key.contains(story_id)
&& a.agent_name == "mergemaster"
@@ -374,7 +366,7 @@ mod tests {
// First call — spawns mergemaster (agent enters Pending).
on_merge_failure_transition(&pool, tmp.path(), &fired).await;
{
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.lock().await;
assert!(
agents.iter().any(|(key, a)| {
key.contains(story_id)
@@ -388,7 +380,7 @@ mod tests {
// Second call (self-loop) — agent is still Pending; guard must prevent double-spawn.
on_merge_failure_transition(&pool, tmp.path(), &fired).await;
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.lock().await;
let active_count = agents
.iter()
.filter(|(key, a)| {
+2 -1
View File
@@ -4,7 +4,8 @@
mod auto_assign;
mod backlog;
mod merge;
/// TransitionFired subscriber that auto-blocks stories after N consecutive MergeFailure transitions.
/// TransitionFired subscriber owning the consecutive-MergeFailure budget:
/// auto-retries GatesFailed below the threshold, auto-blocks at it.
pub(crate) mod merge_failure_block_subscriber;
/// TransitionFired subscriber that auto-spawns mergemaster on ConflictDetected merge failures.
pub(crate) mod merge_failure_subscriber;
+2 -15
View File
@@ -5,7 +5,6 @@ use std::path::Path;
use crate::config::ProjectConfig;
use crate::pipeline_state::Stage;
use crate::slog;
use crate::slog_error;
use super::super::super::PipelineStage;
use super::super::AgentPool;
@@ -80,13 +79,7 @@ impl AgentPool {
if *stage == PipelineStage::Coder
&& let Some(max) = config.max_coders
{
let agents_lock = match self.agents.lock() {
Ok(a) => a,
Err(e) => {
slog_error!("[auto-assign] Failed to lock agents: {e}");
break;
}
};
let agents_lock = self.agents.lock().await;
let active = count_active_agents_for_stage(config, &agents_lock, stage);
if active >= max {
slog!(
@@ -102,13 +95,7 @@ impl AgentPool {
// stage_mismatch=true means the preferred agent's stage doesn't match the
// pipeline stage, so we fell back to a generic stage agent.
let (already_assigned, free_agent, preferred_busy, stage_mismatch) = {
let agents = match self.agents.lock() {
Ok(a) => a,
Err(e) => {
slog_error!("[auto-assign] Failed to lock agents: {e}");
break;
}
};
let agents = self.agents.lock().await;
let assigned = is_story_assigned_for_stage(config, &agents, story_id, stage);
if assigned {
(true, None, false, false)
+6 -6
View File
@@ -256,7 +256,7 @@ mod tests {
let pool = AgentPool::new_test(3001);
pool.inject_test_agent("42_story_foo", "coder-1", AgentStatus::Running);
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.try_lock().unwrap();
assert!(is_story_assigned_for_stage(
&config,
&agents,
@@ -285,7 +285,7 @@ mod tests {
let pool = AgentPool::new_test(3001);
pool.inject_test_agent("42_story_foo", "coder-1", AgentStatus::Completed);
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.try_lock().unwrap();
// Completed agents don't count as assigned
assert!(!is_story_assigned_for_stage(
&config,
@@ -309,7 +309,7 @@ stage = "qa"
let pool = AgentPool::new_test(3001);
pool.inject_test_agent("42_story_foo", "qa-2", AgentStatus::Running);
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.try_lock().unwrap();
// qa-2 with stage=qa should be recognised as a QA agent
assert!(
is_story_assigned_for_stage(&config, &agents, "42_story_foo", &PipelineStage::Qa),
@@ -338,7 +338,7 @@ name = "coder-2"
pool.inject_test_agent("s1", "coder-1", AgentStatus::Running);
pool.inject_test_agent("s2", "coder-2", AgentStatus::Running);
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.try_lock().unwrap();
let free = find_free_agent_for_stage(&config, &agents, &PipelineStage::Coder);
assert!(free.is_none(), "no free coders should be available");
}
@@ -361,7 +361,7 @@ name = "coder-3"
// coder-1 is busy, coder-2 is free
pool.inject_test_agent("s1", "coder-1", AgentStatus::Running);
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.try_lock().unwrap();
let free = find_free_agent_for_stage(&config, &agents, &PipelineStage::Coder);
assert_eq!(
free,
@@ -384,7 +384,7 @@ name = "coder-1"
// coder-1 completed its previous story — it's free for a new one
pool.inject_test_agent("s1", "coder-1", AgentStatus::Completed);
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.try_lock().unwrap();
let free = find_free_agent_for_stage(&config, &agents, &PipelineStage::Coder);
assert_eq!(free, Some("coder-1"), "completed coder-1 should be free");
}
@@ -2,7 +2,7 @@
use std::collections::HashMap;
use std::path::Path;
use std::sync::Mutex;
use tokio::sync::Mutex;
use tokio::sync::broadcast;
use crate::agents::pool::StoryAgent;
@@ -83,7 +83,7 @@ pub(crate) fn count_turns_in_log(path: &Path) -> u64 {
/// Turns and budget are counted from the **current session's** log file
/// only — prior sessions are excluded so that restart counts from earlier
/// runs do not accumulate against the limits.
pub(super) fn check_agent_limits(
pub(super) async fn check_agent_limits(
agents: &Mutex<HashMap<String, StoryAgent>>,
project_root: &Path,
) -> Vec<(String, TerminationReason)> {
@@ -94,10 +94,7 @@ pub(super) fn check_agent_limits(
// Snapshot running agents: (key, story_id, agent_name, tx, log_session_id).
let running: Vec<RunningAgentSnapshot> = {
let lock = match agents.lock() {
Ok(l) => l,
Err(_) => return Vec::new(),
};
let lock = agents.lock().await;
lock.iter()
.filter(|(_, agent)| agent.status == AgentStatus::Running)
.map(|(key, agent)| {
@@ -4,9 +4,11 @@
mod budget;
mod limits;
mod orphan;
mod reap;
#[cfg(test)]
mod tests;
use std::collections::HashSet;
use std::path::Path;
use crate::agents::AgentStatus;
@@ -18,6 +20,7 @@ use crate::slog_warn;
use super::super::AgentPool;
use limits::check_agent_limits;
use orphan::check_orphaned_agents;
use reap::reap_failed_agents;
pub(crate) use budget::{compute_budget_from_logs, compute_budget_from_single_log};
pub(crate) use limits::{count_turns_in_log, resolve_session_log};
@@ -25,8 +28,8 @@ pub(crate) use limits::{count_turns_in_log, resolve_session_log};
impl AgentPool {
/// Run a single watchdog pass synchronously (test helper).
#[cfg(test)]
pub fn run_watchdog_once(&self) {
check_orphaned_agents(&self.agents);
pub async fn run_watchdog_once(&self) {
check_orphaned_agents(&self.agents).await;
}
/// Run one watchdog pass: detect orphans, enforce limits, kill offenders.
@@ -39,21 +42,24 @@ impl AgentPool {
/// `retry_count` is incremented and the story stays in `2_current/` for
/// re-attempt. This prevents the original kill-respawn loop (bug 646)
/// while restoring the `max_retries` semantic for turn/budget overruns.
pub fn run_watchdog_pass(&self, project_root: Option<&Path>) -> usize {
let orphaned = check_orphaned_agents(&self.agents);
pub async fn run_watchdog_pass(&self, project_root: Option<&Path>) -> usize {
let orphaned = check_orphaned_agents(&self.agents).await;
if let Some(root) = project_root {
let terminated = check_agent_limits(&self.agents, root);
let terminated = check_agent_limits(&self.agents, root).await;
let config = ProjectConfig::load(root).unwrap_or_default();
let mut just_terminated: HashSet<String> = HashSet::new();
for (key, reason) in &terminated {
just_terminated.insert(key.clone());
// Step 1: snapshot the agent's worktree path so we can find every
// process running in it (claude + any subprocesses). This must
// happen BEFORE we mutate the agent record so we can read the
// worktree info safely.
let worktree_path = self.agents.lock().ok().and_then(|lock| {
let worktree_path = {
let lock = self.agents.lock().await;
lock.get(key)
.and_then(|a| a.worktree_info.as_ref().map(|wt| wt.path.clone()))
});
};
// Step 2: SIGKILL every process running in the worktree and
// BLOCK until verified gone. The previous mechanism — portable_pty's
@@ -85,23 +91,24 @@ impl AgentPool {
"[watchdog] No worktree path recorded for '{key}'; cannot tree-kill, \
falling back to portable_pty SIGHUP (likely no-op for claude-code)."
);
self.kill_child_for_key(key);
self.kill_child_for_key(key).await;
}
// Step 3: NOW update the agent record. The process is verified
// gone (or we logged that SIGKILL didn't take effect, which is
// exceptional), so flipping status away from Running can no
// longer open a window for a concurrent spawn.
if let Ok(mut lock) = self.agents.lock()
&& let Some(agent) = lock.get_mut(key)
{
agent.status = AgentStatus::Failed;
agent.termination_reason = Some(reason.clone());
if let Some(handle) = agent.task_handle.take() {
// Best-effort abort of the outer tokio task. The PTY
// blocking thread already returned (claude is dead),
// so this is bookkeeping rather than load-bearing.
handle.abort();
let mut lock = self.agents.lock().await;
if let Some(agent) = lock.get_mut(key) {
agent.status = AgentStatus::Failed;
agent.termination_reason = Some(reason.clone());
if let Some(handle) = agent.task_handle.take() {
// Best-effort abort of the outer tokio task. The PTY
// blocking thread already returned (claude is dead),
// so this is bookkeeping rather than load-bearing.
handle.abort();
}
}
}
@@ -130,7 +137,16 @@ impl AgentPool {
if !terminated.is_empty() {
Self::notify_agent_state_changed(&self.watcher_tx);
}
return orphaned + terminated.len();
// Bug 1198: reap any other Failed pool entry with no live process
// — orphan-detected above, or left behind by a spawn error
// (inactivity-watchdog kill, worktree timeout, runtime error)
// that never routed through the retry/respawn path. Entries the
// limits loop above just processed are excluded so their retry
// count isn't bumped twice.
let reaped = reap_failed_agents(self, root, &config, &just_terminated).await;
return orphaned + terminated.len() + reaped;
}
orphaned
@@ -1,7 +1,7 @@
//! Orphan detection: marks running agents whose backing task has exited.
use std::collections::HashMap;
use std::sync::Mutex;
use tokio::sync::Mutex;
use tokio::sync::broadcast;
use crate::agents::pool::StoryAgent;
@@ -15,11 +15,8 @@ use crate::slog;
/// without updating the agent status — for example when the process is killed
/// externally and the PTY master fd returns EOF before our inactivity timeout
/// fires, but some other edge case prevents the normal cleanup path from running.
pub(super) fn check_orphaned_agents(agents: &Mutex<HashMap<String, StoryAgent>>) -> usize {
let mut lock = match agents.lock() {
Ok(l) => l,
Err(_) => return 0,
};
pub(super) async fn check_orphaned_agents(agents: &Mutex<HashMap<String, StoryAgent>>) -> usize {
let mut lock = agents.lock().await;
// Collect orphaned entries: Running or Pending agents whose task handle is finished.
// Pending agents can be orphaned if worktree creation panics before setting status.
@@ -0,0 +1,99 @@
//! Reap: removes stale `Failed` pool entries left behind by orphan detection,
//! watchdog kills, or internal spawn failures, and respawns the story's agent
//! (or blocks the story once its retry budget is exhausted).
//!
//! `check_orphaned_agents` only scans `Running`/`Pending` entries, so once an
//! entry is marked `Failed` it becomes invisible to every later watchdog
//! pass. Without this step a `Failed` entry with no live process sits in the
//! pool forever: `list_agents` keeps showing it, and the story never gets a
//! new agent unless some unrelated CRDT transition happens to trigger a
//! system-wide auto-assign scan.
use std::collections::HashSet;
use std::path::Path;
use crate::agents::AgentStatus;
use crate::agents::pool::AgentPool;
use crate::agents::pool::pipeline::should_block_story;
use crate::config::ProjectConfig;
use crate::io::watcher::WatcherEvent;
use crate::{slog, slog_warn};
/// Reap every `Failed` pool entry with no live process, except keys in
/// `exclude_keys` (already handled by the caller's own retry/block logic in
/// this same pass — e.g. limit-exceeded kills).
///
/// For each reaped entry: removes it from the pool (so `list_agents` stops
/// showing it), increments the story's retry count via [`should_block_story`]
/// and, unless that blocks the story, respawns the agent by name.
/// `start_agent`'s own session-store lookup resumes the prior session
/// automatically whenever one was recorded — no explicit session plumbing
/// needed here.
pub(super) async fn reap_failed_agents(
pool: &AgentPool,
project_root: &Path,
config: &ProjectConfig,
exclude_keys: &HashSet<String>,
) -> usize {
let dead: Vec<(String, String, String)> = {
let mut agents = pool.agents.lock().await;
let keys: Vec<String> = agents
.iter()
.filter(|(key, agent)| {
agent.status == AgentStatus::Failed
&& !exclude_keys.contains(*key)
&& agent
.task_handle
.as_ref()
.map(|h| h.is_finished())
.unwrap_or(true)
})
.map(|(key, _)| key.clone())
.collect();
keys.into_iter()
.filter_map(|key| {
agents.remove(&key).map(|agent| {
let story_id = key
.rsplit_once(':')
.map(|(s, _)| s.to_string())
.unwrap_or_else(|| key.clone());
(key, story_id, agent.agent_name)
})
})
.collect()
};
let count = dead.len();
for (key, story_id, agent_name) in dead {
if let Some(block_reason) = should_block_story(&story_id, config.max_retries, "watchdog") {
let _ = pool.watcher_tx.send(WatcherEvent::StoryBlocked {
story_id: story_id.clone(),
reason: block_reason,
});
slog!(
"[watchdog] Story '{story_id}' blocked after exceeding retry limit \
(reaped dead pool entry '{key}')."
);
continue;
}
slog!(
"[watchdog] Reaping dead pool entry '{key}'; respawning '{agent_name}' \
for '{story_id}'."
);
if let Err(e) = pool
.start_agent(project_root, &story_id, Some(&agent_name), None, None)
.await
{
slog_warn!(
"[watchdog] Failed to respawn '{agent_name}' for '{story_id}' after reap: {e}"
);
}
}
if count > 0 {
AgentPool::notify_agent_state_changed(&pool.watcher_tx);
}
count
}
@@ -10,8 +10,8 @@ use crate::agents::{AgentEvent, AgentStatus, TerminationReason};
// ── Limit enforcement integration tests (bug 624) ────────────────────────
#[test]
fn watchdog_terminates_agent_exceeding_turn_limit() {
#[tokio::test]
async fn watchdog_terminates_agent_exceeding_turn_limit() {
let tmp = tempfile::tempdir().unwrap();
let root = tmp.path();
@@ -37,12 +37,12 @@ max_turns = 10
);
let mut rx = tx.subscribe();
let found = pool.run_watchdog_pass(Some(root));
let found = pool.run_watchdog_pass(Some(root)).await;
assert!(found >= 1, "watchdog should detect the over-limit agent");
// Agent should now be Failed with TurnLimit reason.
{
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.try_lock().unwrap();
let key = composite_key("story_a", "coder-1");
let agent = agents.get(&key).unwrap();
assert_eq!(agent.status, AgentStatus::Failed);
@@ -60,8 +60,8 @@ max_turns = 10
);
}
#[test]
fn watchdog_terminates_agent_exceeding_budget_limit() {
#[tokio::test]
async fn watchdog_terminates_agent_exceeding_budget_limit() {
let tmp = tempfile::tempdir().unwrap();
let root = tmp.path();
@@ -87,11 +87,11 @@ max_budget_usd = 5.00
);
let mut rx = tx.subscribe();
let found = pool.run_watchdog_pass(Some(root));
let found = pool.run_watchdog_pass(Some(root)).await;
assert!(found >= 1, "watchdog should detect the over-budget agent");
{
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.try_lock().unwrap();
let key = composite_key("story_b", "coder-1");
let agent = agents.get(&key).unwrap();
assert_eq!(agent.status, AgentStatus::Failed);
@@ -106,8 +106,8 @@ max_budget_usd = 5.00
assert!(matches!(event, AgentEvent::Error { .. }));
}
#[test]
fn watchdog_does_not_terminate_agent_under_limits() {
#[tokio::test]
async fn watchdog_does_not_terminate_agent_under_limits() {
let tmp = tempfile::tempdir().unwrap();
let root = tmp.path();
@@ -133,11 +133,11 @@ max_budget_usd = 10.00
// has 25 turns < 50 so no violation).
pool.inject_test_agent_with_session("story_c", "coder-1", AgentStatus::Running, "sess-ok");
let found = pool.run_watchdog_pass(Some(root));
let found = pool.run_watchdog_pass(Some(root)).await;
assert_eq!(found, 0, "agent under limits should not be terminated");
{
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.try_lock().unwrap();
let key = composite_key("story_c", "coder-1");
let agent = agents.get(&key).unwrap();
assert_eq!(
@@ -153,8 +153,8 @@ max_budget_usd = 10.00
/// coder-1 with max_turns=50, max_budget_usd=5.00 ran 5.6× over the turn
/// limit (280 turns). The watchdog must terminate at the turn limit (turns
/// hit first in the observed trace), with reason TurnLimit.
#[test]
fn regression_bug624_coder1_story623_trajectory() {
#[tokio::test]
async fn regression_bug624_coder1_story623_trajectory() {
let tmp = tempfile::tempdir().unwrap();
let root = tmp.path();
@@ -183,11 +183,11 @@ max_budget_usd = 5.00
);
let mut rx = tx.subscribe();
let found = pool.run_watchdog_pass(Some(root));
let found = pool.run_watchdog_pass(Some(root)).await;
assert!(found >= 1, "watchdog must catch the turn-limit violation");
{
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.try_lock().unwrap();
let key = composite_key("story_623", "coder-1");
let agent = agents.get(&key).unwrap();
assert_eq!(agent.status, AgentStatus::Failed);
@@ -218,8 +218,8 @@ max_budget_usd = 5.00
///
/// This test seeds a single session that legitimately exceeds the limit
/// and uses `max_retries = 1` so that the first violation blocks.
#[test]
fn watchdog_marks_story_blocked_after_limit_termination() {
#[tokio::test]
async fn watchdog_marks_story_blocked_after_limit_termination() {
crate::db::ensure_content_store();
crate::crdt_state::init_for_test();
@@ -263,7 +263,7 @@ max_turns = 10
"sess-runaway",
);
let found = pool.run_watchdog_pass(Some(root));
let found = pool.run_watchdog_pass(Some(root)).await;
assert!(found >= 1, "watchdog should detect the over-limit agent");
// With max_retries=1, the first violation blocks immediately via the state machine.
@@ -278,7 +278,7 @@ max_turns = 10
// Sanity: the agent itself is also Failed with the right reason.
{
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.try_lock().unwrap();
let key = composite_key(story_id, "coder-1");
let agent = agents.get(&key).unwrap();
assert_eq!(agent.status, AgentStatus::Failed);
@@ -297,8 +297,8 @@ max_turns = 10
/// fresh session_id whose log has fewer events than `max_turns`.
/// Assert the agent is NOT terminated (per-session count is under the
/// limit) AND the story is NOT marked blocked.
#[test]
fn per_session_counting_does_not_terminate_under_limit() {
#[tokio::test]
async fn per_session_counting_does_not_terminate_under_limit() {
let tmp = tempfile::tempdir().unwrap();
let root = tmp.path();
@@ -323,14 +323,14 @@ max_turns = 10
let pool = AgentPool::new_test(3001);
pool.inject_test_agent_with_session("story_d", "coder-1", AgentStatus::Running, "new-sess");
let found = pool.run_watchdog_pass(Some(root));
let found = pool.run_watchdog_pass(Some(root)).await;
assert_eq!(
found, 0,
"agent under per-session limit should NOT be terminated"
);
{
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.try_lock().unwrap();
let key = composite_key("story_d", "coder-1");
let agent = agents.get(&key).unwrap();
assert_eq!(
@@ -345,8 +345,8 @@ max_turns = 10
/// Same setup as per_session_counting_does_not_terminate_under_limit, but
/// the new agent's own session log exceeds `max_turns`. Assert the agent
/// IS terminated AND (with max_retries=1) the story IS marked blocked.
#[test]
fn per_session_counting_terminates_over_limit() {
#[tokio::test]
async fn per_session_counting_terminates_over_limit() {
crate::db::ensure_content_store();
crate::crdt_state::init_for_test();
@@ -390,14 +390,14 @@ max_turns = 10
pool.inject_test_agent_with_session(story_id, "coder-1", AgentStatus::Running, "new-sess");
let mut rx = tx.subscribe();
let found = pool.run_watchdog_pass(Some(root));
let found = pool.run_watchdog_pass(Some(root)).await;
assert!(
found >= 1,
"agent over per-session limit must be terminated"
);
{
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.try_lock().unwrap();
let key = composite_key(story_id, "coder-1");
let agent = agents.get(&key).unwrap();
assert_eq!(agent.status, AgentStatus::Failed);
@@ -423,8 +423,8 @@ max_turns = 10
/// `max_turns`. After session 1: retry_count=1, NOT blocked. After
/// session 2: retry_count=2, NOT blocked. After session 3:
/// retry_count=3 >= max_retries, story IS blocked.
#[test]
fn watchdog_retry_semantic_blocks_after_max_retries() {
#[tokio::test]
async fn watchdog_retry_semantic_blocks_after_max_retries() {
crate::db::ensure_content_store();
let tmp = tempfile::tempdir().unwrap();
@@ -453,7 +453,7 @@ max_turns = 10
write_fake_session_log(root, story_id, "coder-1", "session-1", 12);
let pool = AgentPool::new_test(3001);
pool.inject_test_agent_with_session(story_id, "coder-1", AgentStatus::Running, "session-1");
pool.run_watchdog_pass(Some(root));
pool.run_watchdog_pass(Some(root)).await;
let item = crate::crdt_state::read_item(story_id).expect("story must be in CRDT");
assert_eq!(
@@ -473,7 +473,7 @@ max_turns = 10
write_fake_session_log(root, story_id, "coder-1", "session-2", 12);
let pool = AgentPool::new_test(3001);
pool.inject_test_agent_with_session(story_id, "coder-1", AgentStatus::Running, "session-2");
pool.run_watchdog_pass(Some(root));
pool.run_watchdog_pass(Some(root)).await;
let item = crate::crdt_state::read_item(story_id).expect("story must be in CRDT");
assert_eq!(
@@ -493,7 +493,7 @@ max_turns = 10
write_fake_session_log(root, story_id, "coder-1", "session-3", 12);
let pool = AgentPool::new_test(3001);
pool.inject_test_agent_with_session(story_id, "coder-1", AgentStatus::Running, "session-3");
pool.run_watchdog_pass(Some(root));
pool.run_watchdog_pass(Some(root)).await;
let item = crate::crdt_state::read_item(story_id).expect("story must be in CRDT");
assert_eq!(
@@ -518,8 +518,8 @@ max_turns = 10
/// must not count against the watchdog's turn budget. A session log with
/// 5 tool turns and 30 narration turns reports turns_used == 5, so an
/// agent with max_tool_turns = 10 stays Running.
#[test]
fn watchdog_does_not_count_narration_only_turns() {
#[tokio::test]
async fn watchdog_does_not_count_narration_only_turns() {
let tmp = tempfile::tempdir().unwrap();
let root = tmp.path();
@@ -542,13 +542,13 @@ max_turns = 200
let pool = AgentPool::new_test(3001);
pool.inject_test_agent_with_session("story_923", "coder-1", AgentStatus::Running, "sess-narr");
let found = pool.run_watchdog_pass(Some(root));
let found = pool.run_watchdog_pass(Some(root)).await;
assert_eq!(
found, 0,
"agent must not be terminated: only 5 tool turns of a 10-turn budget"
);
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.try_lock().unwrap();
let key = composite_key("story_923", "coder-1");
let agent = agents.get(&key).unwrap();
assert_eq!(agent.status, AgentStatus::Running);
@@ -558,8 +558,8 @@ max_turns = 200
/// Story 923: max_tool_turns takes precedence over max_turns when both are
/// set. With max_tool_turns = 3 and max_turns = 200, an agent that has 4
/// tool turns is killed even though total turns (4) is far below max_turns.
#[test]
fn watchdog_max_tool_turns_overrides_max_turns() {
#[tokio::test]
async fn watchdog_max_tool_turns_overrides_max_turns() {
let tmp = tempfile::tempdir().unwrap();
let root = tmp.path();
@@ -585,13 +585,13 @@ max_turns = 200
);
let mut rx = tx.subscribe();
let found = pool.run_watchdog_pass(Some(root));
let found = pool.run_watchdog_pass(Some(root)).await;
assert!(
found >= 1,
"watchdog must terminate when tool turns exceed max_tool_turns"
);
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.try_lock().unwrap();
let key = composite_key("story_923b", "coder-1");
let agent = agents.get(&key).unwrap();
assert_eq!(agent.status, AgentStatus::Failed);
@@ -4,6 +4,7 @@ use std::path::Path;
mod limits_tests;
mod orphan_tests;
mod reap_tests;
/// Write a fake session log file with `n` tool-using assistant turn entries.
///
@@ -20,18 +20,18 @@ async fn check_orphaned_agents_returns_count_of_orphaned_agents() {
pool.inject_test_agent_with_handle("story_a", "coder", AgentStatus::Running, h1);
pool.inject_test_agent_with_handle("story_b", "coder", AgentStatus::Running, h2);
let found = check_orphaned_agents(&pool.agents);
let found = check_orphaned_agents(&pool.agents).await;
assert_eq!(found, 2, "should detect both orphaned agents");
}
#[test]
fn check_orphaned_agents_returns_zero_when_no_orphans() {
#[tokio::test]
async fn check_orphaned_agents_returns_zero_when_no_orphans() {
let pool = AgentPool::new_test(3001);
// Inject agents in terminal states — not orphaned.
pool.inject_test_agent("story_a", "coder", AgentStatus::Completed);
pool.inject_test_agent("story_b", "qa", AgentStatus::Failed);
let found = check_orphaned_agents(&pool.agents);
let found = check_orphaned_agents(&pool.agents).await;
assert_eq!(
found, 0,
"no orphans should be detected for terminal agents"
@@ -53,10 +53,10 @@ async fn watchdog_detects_orphaned_running_agent() {
pool.inject_test_agent_with_handle("orphan_story", "coder", AgentStatus::Running, handle);
let mut rx = tx.subscribe();
pool.run_watchdog_once();
pool.run_watchdog_once().await;
{
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.try_lock().unwrap();
let key = composite_key("orphan_story", "coder");
let agent = agents.get(&key).unwrap();
assert_eq!(
@@ -87,13 +87,13 @@ async fn watchdog_orphan_detection_returns_nonzero_enabling_auto_assign() {
// Before watchdog: agent is Running.
{
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.try_lock().unwrap();
let key = composite_key("orphan_story", "coder");
assert_eq!(agents.get(&key).unwrap().status, AgentStatus::Running);
}
// Run watchdog pass — should return 1 (orphan found).
let found = check_orphaned_agents(&pool.agents);
let found = check_orphaned_agents(&pool.agents).await;
assert_eq!(
found, 1,
"watchdog must return 1 for a single orphaned agent"
@@ -101,7 +101,7 @@ async fn watchdog_orphan_detection_returns_nonzero_enabling_auto_assign() {
// After watchdog: agent is Failed.
{
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.try_lock().unwrap();
let key = composite_key("orphan_story", "coder");
assert_eq!(
agents.get(&key).unwrap().status,
@@ -0,0 +1,238 @@
//! Regression tests for the reap pass (bug 1198): a Failed pool entry with
//! no live process must be respawned by the next watchdog pass, with its
//! story's retry count bumped, and must stop appearing in list_agents.
use super::super::super::super::{AgentPool, composite_key};
use super::{write_fake_session_log, write_project_config};
use crate::agents::AgentStatus;
/// AC1 + AC4: a Failed entry with no live process (simulating e.g. the
/// inactivity watchdog kill landing in spawn.rs's generic Err arm) is
/// respawned by the very next `run_watchdog_pass` — no manual
/// stop_agent/start_agent needed.
#[tokio::test]
async fn reap_respawns_stale_failed_agent() {
crate::db::ensure_content_store();
crate::crdt_state::init_for_test();
let tmp = tempfile::tempdir().unwrap();
let root = tmp.path();
write_project_config(
root,
r#"
[[agent]]
name = "coder-1"
runtime = "claude-code"
"#,
);
let story_id = "1198_story_stuck";
crate::db::write_content(
crate::db::ContentKey::Story(story_id),
"---\nname: Stuck Story\n---\n",
);
crate::crdt_state::write_item_str(story_id, "2_current", Some("Stuck Story"), None, None, None);
// Inject a Failed entry with no task_handle — exactly what spawn.rs's
// generic Err arm leaves behind after e.g. an inactivity-watchdog kill.
let pool = AgentPool::new_test(3001);
pool.inject_test_agent(story_id, "coder-1", AgentStatus::Failed);
let found = pool.run_watchdog_pass(Some(root)).await;
assert!(found >= 1, "reap should count the stale Failed entry");
// A fresh entry must exist for the same story — the agent respawned.
let agents = pool.agents.try_lock().unwrap();
let key = composite_key(story_id, "coder-1");
let agent = agents.get(&key).expect("agent must have been respawned");
assert_ne!(
agent.status,
AgentStatus::Failed,
"respawned entry must not still be Failed"
);
}
/// AC2: reaping a stale Failed entry bumps the story's retry count via the
/// existing should_block_story path.
#[tokio::test]
async fn reap_increments_retry_count() {
crate::db::ensure_content_store();
crate::crdt_state::init_for_test();
let tmp = tempfile::tempdir().unwrap();
let root = tmp.path();
write_project_config(
root,
r#"
max_retries = 5
[[agent]]
name = "coder-1"
runtime = "claude-code"
"#,
);
let story_id = "1198_story_retry";
crate::db::write_content(
crate::db::ContentKey::Story(story_id),
"---\nname: Retry Story\n---\n",
);
crate::crdt_state::write_item_str(story_id, "2_current", Some("Retry Story"), None, None, None);
let pool = AgentPool::new_test(3001);
pool.inject_test_agent(story_id, "coder-1", AgentStatus::Failed);
pool.run_watchdog_pass(Some(root)).await;
let item = crate::crdt_state::read_item(story_id).expect("story must be in CRDT");
assert_eq!(
item.retry_count(),
1,
"reaping a stale Failed entry must bump retry_count exactly once"
);
}
/// AC2: exhausting max_retries via a reap blocks the story via the existing
/// should_block_story path (same mechanism as the limits watchdog).
#[tokio::test]
async fn reap_blocks_story_after_max_retries() {
crate::db::ensure_content_store();
crate::crdt_state::init_for_test();
let tmp = tempfile::tempdir().unwrap();
let root = tmp.path();
write_project_config(
root,
r#"
max_retries = 1
[[agent]]
name = "coder-1"
runtime = "claude-code"
"#,
);
let story_id = "1198_story_block";
crate::db::write_content(
crate::db::ContentKey::Story(story_id),
"---\nname: Block Story\n---\n",
);
crate::crdt_state::write_item_str(story_id, "2_current", Some("Block Story"), None, None, None);
let pool = AgentPool::new_test(3001);
pool.inject_test_agent(story_id, "coder-1", AgentStatus::Failed);
pool.run_watchdog_pass(Some(root)).await;
let item = crate::crdt_state::read_item(story_id).expect("story must be in CRDT");
assert_eq!(
item.stage().dir_name(),
"blocked",
"story must be blocked after exhausting max_retries=1 via reap"
);
}
/// AC3: list_agents never shows a Failed entry after the next watchdog pass
/// reaps it — regardless of whether the story blocks or respawns.
#[tokio::test]
async fn reap_removes_failed_entry_from_list_agents() {
crate::db::ensure_content_store();
crate::crdt_state::init_for_test();
let tmp = tempfile::tempdir().unwrap();
let root = tmp.path();
write_project_config(
root,
r#"
max_retries = 1
[[agent]]
name = "coder-1"
runtime = "claude-code"
"#,
);
let story_id = "1198_story_listing";
crate::db::write_content(
crate::db::ContentKey::Story(story_id),
"---\nname: Listing Story\n---\n",
);
crate::crdt_state::write_item_str(
story_id,
"2_current",
Some("Listing Story"),
None,
None,
None,
);
let pool = AgentPool::new_test(3001);
pool.inject_test_agent(story_id, "coder-1", AgentStatus::Failed);
pool.run_watchdog_pass(Some(root)).await;
let listed = pool.list_agents().await.unwrap();
assert!(
!listed
.iter()
.any(|a| a.story_id == story_id && a.status == AgentStatus::Failed),
"list_agents must not show a Failed entry for '{story_id}' after the next watchdog pass"
);
}
/// The limits-termination path (turn/budget overrun) must not have its
/// retry_count double-bumped by the reap pass running in the same
/// `run_watchdog_pass` call.
#[tokio::test]
async fn reap_does_not_double_bump_limits_terminated_agent() {
crate::db::ensure_content_store();
crate::crdt_state::init_for_test();
let tmp = tempfile::tempdir().unwrap();
let root = tmp.path();
write_project_config(
root,
r#"
max_retries = 5
[[agent]]
name = "coder-1"
runtime = "claude-code"
max_turns = 10
"#,
);
let story_id = "1198_story_no_double_bump";
crate::db::write_content(
crate::db::ContentKey::Story(story_id),
"---\nname: No Double Bump\n---\n",
);
crate::crdt_state::write_item_str(
story_id,
"2_current",
Some("No Double Bump"),
None,
None,
None,
);
write_fake_session_log(root, story_id, "coder-1", "sess-overrun", 12);
let pool = AgentPool::new_test(3001);
pool.inject_test_agent_with_session(story_id, "coder-1", AgentStatus::Running, "sess-overrun");
pool.run_watchdog_pass(Some(root)).await;
let item = crate::crdt_state::read_item(story_id).expect("story must be in CRDT");
assert_eq!(
item.retry_count(),
1,
"a single limit-termination pass must bump retry_count by exactly 1, \
not twice (once from the limits branch, once from reap)"
);
}
+7 -5
View File
@@ -19,8 +19,8 @@ mod test_helpers;
use crate::io::watcher::WatcherEvent;
use crate::service::status::StatusBroadcaster;
use std::collections::HashMap;
use std::sync::{Arc, Mutex};
use tokio::sync::broadcast;
use std::sync::Arc;
use tokio::sync::{Mutex, broadcast};
// Bring pool-internal types into pool's namespace so that sub-modules
// (auto_assign, pipeline, etc.) can access them via `use super::...`.
@@ -87,10 +87,12 @@ impl AgentPool {
_ => continue,
};
let key = composite_key(&story_id, &agent_name);
if let Ok(mut agents) = agents_clone.lock()
&& let Some(agent) = agents.get_mut(&key)
{
agent.throttled = Some(crate::agents::AgentExecution::Throttled { until });
let mut agents = agents_clone.lock().await;
if let Some(agent) = agents.get_mut(&key) {
agent.throttled =
Some(crate::agents::AgentExecution::Throttled { until });
}
}
let _ = watcher_tx_clone.send(WatcherEvent::AgentStateChanged);
}
@@ -3,7 +3,7 @@
use std::collections::HashMap;
use std::path::PathBuf;
use std::sync::{Arc, Mutex};
use std::sync::Arc;
use tokio::sync::broadcast;
@@ -16,7 +16,7 @@ use std::path::Path;
/// type cycle between `start_agent` and `run_server_owned_completion`.
#[allow(clippy::too_many_arguments)]
pub(crate) fn spawn_pipeline_advance(
agents: Arc<Mutex<HashMap<String, StoryAgent>>>,
agents: Arc<tokio::sync::Mutex<HashMap<String, StoryAgent>>>,
port: u16,
story_id: &str,
agent_name: &str,
@@ -694,7 +694,7 @@ impl AgentPool {
if let Err(e) = crate::agents::lifecycle::move_story_to_done(story_id) {
slog_error!("[pipeline] Failed to move '{story_id}' to done: {e}");
}
self.remove_agents_for_story(story_id);
self.remove_agents_for_story(story_id).await;
crate::crdt_state::delete_merge_job(story_id);
// TODO: Re-enable worktree cleanup once we have persistent agent logs.
// Removing worktrees destroys evidence needed to debug empty-commit agents.
@@ -104,7 +104,7 @@ async fn mergemaster_blocks_and_sends_story_blocked_when_no_commits_ahead() {
);
// No mergemaster agent should have been started.
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.try_lock().unwrap();
let mergemaster_started = agents
.values()
.any(|a| a.agent_name.contains("mergemaster"));
@@ -162,7 +162,7 @@ stage = "qa"
// Verify that 293 cannot get a QA agent right now (QA is busy).
{
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.try_lock().unwrap();
assert!(
!is_agent_free(&agents, "qa"),
"qa should be busy on story 292"
@@ -172,7 +172,7 @@ stage = "qa"
// Simulate QA completing on story 292: remove the agent from the pool
// (as run_server_owned_completion does) then run pipeline advance.
{
let mut agents = pool.agents.lock().unwrap();
let mut agents = pool.agents.try_lock().unwrap();
agents.remove(&composite_key("292_story_first", "qa"));
}
@@ -193,7 +193,7 @@ stage = "qa"
.await;
// After pipeline advance, auto_assign should have started QA on story 293.
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.try_lock().unwrap();
let qa_on_293 = agents.values().any(|a| {
a.agent_name == "qa" && matches!(a.status, AgentStatus::Pending | AgentStatus::Running)
});
@@ -278,7 +278,7 @@ async fn stale_mergemaster_advance_for_done_story_is_noop() {
.await;
// No agents should have been started.
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.try_lock().unwrap();
assert!(
agents.is_empty(),
"No agents should be started for a stale advance on a done story. \
@@ -871,7 +871,7 @@ stage = "coder"
.await;
// The coder must be re-spawned — Pending or Running.
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.try_lock().unwrap();
let coder_restarted = agents.values().any(|a| {
a.agent_name == "coder-1" && matches!(a.status, AgentStatus::Pending | AgentStatus::Running)
});
@@ -957,7 +957,7 @@ stage = "coder"
.await;
// The recovery respawn must have been issued — coder-1 should be Pending/Running.
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.try_lock().unwrap();
let coder_restarted = agents.values().any(|a| {
a.agent_name == "coder-1" && matches!(a.status, AgentStatus::Pending | AgentStatus::Running)
});
@@ -1328,7 +1328,7 @@ async fn coder_completion_with_test_evidence_and_zero_commits_does_not_advance()
);
// No QA or merge agent should have been started.
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.try_lock().unwrap();
let qa_or_merge_started = agents
.values()
.any(|a| a.agent_name.contains("qa") || a.agent_name.contains("merge"));
@@ -28,7 +28,7 @@ impl AgentPool {
// Verify agent exists, is Running, and grab its worktree path.
let worktree_path = {
let agents = self.agents.lock().map_err(|e| e.to_string())?;
let agents = self.agents.lock().await;
let agent = agents
.get(&key)
.ok_or_else(|| format!("No agent '{agent_name}' for story '{story_id}'"))?;
@@ -82,7 +82,7 @@ impl AgentPool {
merge_failure_reported_for_advance,
session_id_for_advance,
) = {
let mut agents = self.agents.lock().map_err(|e| e.to_string())?;
let mut agents = self.agents.lock().await;
let agent = agents.get_mut(&key).ok_or_else(|| {
format!("Agent '{agent_name}' for story '{story_id}' disappeared during gate check")
})?;
@@ -2,7 +2,8 @@
use crate::io::watcher::WatcherEvent;
use crate::slog;
use std::collections::HashMap;
use std::sync::{Arc, Mutex};
use std::sync::Arc;
use tokio::sync::Mutex;
use tokio::sync::broadcast;
use super::super::super::super::{AgentEvent, CompletionReport, PipelineStage, pipeline_stage};
@@ -45,10 +46,7 @@ pub(in crate::agents::pool) async fn run_server_owned_completion(
// Guard: skip if completion was already recorded (legacy path).
{
let lock = match agents.lock() {
Ok(a) => a,
Err(_) => return,
};
let lock = agents.lock().await;
match lock.get(&key) {
Some(agent) if agent.completion.is_some() => {
slog!(
@@ -64,10 +62,7 @@ pub(in crate::agents::pool) async fn run_server_owned_completion(
// Get worktree path for running gates.
let worktree_path = {
let lock = match agents.lock() {
Ok(a) => a,
Err(_) => return,
};
let lock = agents.lock().await;
lock.get(&key)
.and_then(|a| a.worktree_info.as_ref().map(|wt| wt.path.clone()))
};
@@ -175,6 +170,14 @@ pub(in crate::agents::pool) async fn run_server_owned_completion(
"[agents] Server-owned completion for '{story_id}:{agent_name}': gates_passed={gates_passed}"
);
crate::history::record_agent_run(
story_id,
agent_name,
session_id.as_deref(),
gates_passed,
&gate_output,
);
// Notify chat transports of the agent completion result.
let _ = watcher_tx.send(WatcherEvent::AgentCompleted {
story_id: story_id.to_string(),
@@ -192,10 +195,7 @@ pub(in crate::agents::pool) async fn run_server_owned_completion(
// Store completion report, extract data for pipeline advance, then
// remove the entry so completed agents never appear in list_agents.
let (tx, project_root_for_advance, wt_path_for_advance, merge_failure_reported_for_advance) = {
let mut lock = match agents.lock() {
Ok(a) => a,
Err(_) => return,
};
let mut lock = agents.lock().await;
let agent = match lock.get_mut(&key) {
Some(a) => a,
None => return,
@@ -6,26 +6,35 @@ use std::path::PathBuf;
use std::process::Command;
fn init_git_repo(repo: &std::path::Path) {
Command::new("git")
.args(["init"])
.current_dir(repo)
.output()
.unwrap();
Command::new("git")
.args(["config", "user.email", "test@test.com"])
.current_dir(repo)
.output()
.unwrap();
Command::new("git")
.args(["config", "user.name", "Test"])
.current_dir(repo)
.output()
.unwrap();
Command::new("git")
.args(["commit", "--allow-empty", "-m", "init"])
.current_dir(repo)
.output()
.unwrap();
use crate::git_test_support::git_ok;
git_ok(
Command::new("git")
.args(["init"])
.current_dir(repo)
.output(),
"git init",
);
git_ok(
Command::new("git")
.args(["config", "user.email", "test@test.com"])
.current_dir(repo)
.output(),
"git config user.email",
);
git_ok(
Command::new("git")
.args(["config", "user.name", "Test"])
.current_dir(repo)
.output(),
"git config user.name",
);
git_ok(
Command::new("git")
.args(["commit", "--allow-empty", "-m", "init"])
.current_dir(repo)
.output(),
"git commit",
);
}
// ── report_completion tests ────────────────────────────────────
@@ -108,7 +117,7 @@ async fn server_owned_completion_skips_when_already_completed() {
);
// Subscribe before calling so we can check if Done event was emitted.
let mut rx = pool.subscribe("s10", "coder-1").unwrap();
let mut rx = pool.subscribe("s10", "coder-1").await.unwrap();
run_server_owned_completion(
&pool.agents,
@@ -121,7 +130,7 @@ async fn server_owned_completion_skips_when_already_completed() {
.await;
// Status should remain Completed (unchanged) — no gate re-run.
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.try_lock().unwrap();
let key = super::super::super::composite_key("s10", "coder-1");
let agent = agents.get(&key).unwrap();
assert_eq!(agent.status, AgentStatus::Completed);
@@ -147,7 +156,7 @@ async fn server_owned_completion_runs_gates_on_clean_worktree() {
let pool = AgentPool::new_test(3001);
pool.inject_test_agent_with_path("s11", "coder-1", AgentStatus::Running, repo.to_path_buf());
let mut rx = pool.subscribe("s11", "coder-1").unwrap();
let mut rx = pool.subscribe("s11", "coder-1").await.unwrap();
run_server_owned_completion(
&pool.agents,
@@ -160,7 +169,7 @@ async fn server_owned_completion_runs_gates_on_clean_worktree() {
.await;
// Agent entry should be removed from the map after completion.
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.try_lock().unwrap();
let key = super::super::super::composite_key("s11", "coder-1");
assert!(
agents.get(&key).is_none(),
@@ -192,7 +201,7 @@ async fn server_owned_completion_fails_on_dirty_worktree() {
let pool = AgentPool::new_test(3001);
pool.inject_test_agent_with_path("s12", "coder-1", AgentStatus::Running, repo.to_path_buf());
let mut rx = pool.subscribe("s12", "coder-1").unwrap();
let mut rx = pool.subscribe("s12", "coder-1").await.unwrap();
run_server_owned_completion(
&pool.agents,
@@ -205,7 +214,7 @@ async fn server_owned_completion_fails_on_dirty_worktree() {
.await;
// Agent entry should be removed from the map after completion (even on failure).
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.try_lock().unwrap();
let key = super::super::super::composite_key("s12", "coder-1");
assert!(
agents.get(&key).is_none(),
@@ -307,7 +316,7 @@ async fn server_owned_completion_is_noop_for_mergemaster() {
// The agent entry should remain in the pool (lifecycle cleanup is the
// caller's responsibility, not run_server_owned_completion's).
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.try_lock().unwrap();
let key = super::super::super::composite_key("99_story_merge445", "mergemaster");
assert!(
agents.get(&key).is_some(),
@@ -361,7 +370,7 @@ async fn server_owned_completion_preserves_dirty_worktree_with_committed_work()
let pool = AgentPool::new_test(3001);
pool.inject_test_agent_with_path("645_test", "coder-1", AgentStatus::Running, wt_path.clone());
let mut rx = pool.subscribe("645_test", "coder-1").unwrap();
let mut rx = pool.subscribe("645_test", "coder-1").await.unwrap();
run_server_owned_completion(
&pool.agents,
@@ -34,35 +34,29 @@ impl AgentPool {
/// If the agent was already removed from the pool (race: `remove_agents_for_story`
/// ran first) this is a no-op; the `ContentKey::MergeSuccess` DB key written
/// by the caller acts as the authoritative fallback in that case.
pub fn set_merge_success_reported(&self, story_id: &str) {
match self.agents.lock() {
Ok(mut lock) => {
let found = lock.iter_mut().find(|(key, agent)| {
let key_story_id = key
.rsplit_once(':')
.map(|(sid, _)| sid)
.unwrap_or(key.as_str());
key_story_id == story_id
&& pipeline_stage(&agent.agent_name) == PipelineStage::Mergemaster
});
match found {
Some((_, agent)) => {
agent.merge_success_reported = true;
slog!(
"[pipeline] Merge success flag set for '{story_id}:{}'",
agent.agent_name
);
}
None => {
slog!(
"[pipeline] set_merge_success_reported: no running mergemaster \
for '{story_id}' DB key is the authoritative fallback"
);
}
}
pub async fn set_merge_success_reported(&self, story_id: &str) {
let mut lock = self.agents.lock().await;
let found = lock.iter_mut().find(|(key, agent)| {
let key_story_id = key
.rsplit_once(':')
.map(|(sid, _)| sid)
.unwrap_or(key.as_str());
key_story_id == story_id
&& pipeline_stage(&agent.agent_name) == PipelineStage::Mergemaster
});
match found {
Some((_, agent)) => {
agent.merge_success_reported = true;
slog!(
"[pipeline] Merge success flag set for '{story_id}:{}'",
agent.agent_name
);
}
Err(e) => {
slog_error!("[pipeline] set_merge_success_reported: could not lock agents: {e}");
None => {
slog!(
"[pipeline] set_merge_success_reported: no running mergemaster \
for '{story_id}' DB key is the authoritative fallback"
);
}
}
}
@@ -74,35 +68,29 @@ impl AgentPool {
/// that `run_pipeline_advance` can block advancement to `5_done/` even when
/// the server-owned gate check returns `gates_passed=true` (those gates run
/// in the feature-branch worktree, not on master).
pub fn set_merge_failure_reported(&self, story_id: &str) {
match self.agents.lock() {
Ok(mut lock) => {
let found = lock.iter_mut().find(|(key, agent)| {
let key_story_id = key
.rsplit_once(':')
.map(|(sid, _)| sid)
.unwrap_or(key.as_str());
key_story_id == story_id
&& pipeline_stage(&agent.agent_name) == PipelineStage::Mergemaster
});
match found {
Some((_, agent)) => {
agent.merge_failure_reported = true;
slog!(
"[pipeline] Merge failure flag set for '{story_id}:{}'",
agent.agent_name
);
}
None => {
slog_warn!(
"[pipeline] set_merge_failure_reported: no running mergemaster found \
for story '{story_id}' flag not set"
);
}
}
pub async fn set_merge_failure_reported(&self, story_id: &str) {
let mut lock = self.agents.lock().await;
let found = lock.iter_mut().find(|(key, agent)| {
let key_story_id = key
.rsplit_once(':')
.map(|(sid, _)| sid)
.unwrap_or(key.as_str());
key_story_id == story_id
&& pipeline_stage(&agent.agent_name) == PipelineStage::Mergemaster
});
match found {
Some((_, agent)) => {
agent.merge_failure_reported = true;
slog!(
"[pipeline] Merge failure flag set for '{story_id}:{}'",
agent.agent_name
);
}
Err(e) => {
slog_error!("[pipeline] set_merge_failure_reported: could not lock agents: {e}");
None => {
slog_warn!(
"[pipeline] set_merge_failure_reported: no running mergemaster found \
for story '{story_id}' flag not set"
);
}
}
}
+59 -60
View File
@@ -32,7 +32,19 @@ impl AgentPool {
/// Called at the top of [`start_merge_agent_work`] to unblock retries,
/// and also by the periodic background reaper in the tick loop so stale
/// entries are cleaned up even when no new merge is triggered.
///
/// A job's `server_start` round-trips through JSON text (see
/// [`encode_server_start_time`]/[`decode_server_start_time`]), and
/// `serde_json`'s float parser is not guaranteed bit-exact for
/// high-precision Unix timestamps — it can decode a value a couple of
/// ULPs below the original. Comparing with a bare `<` against a
/// freshly-read `current_boot` would then occasionally treat a job
/// written by *this very server instance* as belonging to a previous
/// one. Real server restarts are always seconds apart at minimum, so a
/// generous tolerance absorbs that noise without weakening genuine
/// stale-boot detection.
pub(crate) fn reap_stale_merge_jobs(&self) {
const STALE_TOLERANCE_SECS: f64 = 1.0;
if let Some(jobs) = crate::crdt_state::read_all_merge_jobs() {
let current_boot = server_start_time();
for job in jobs {
@@ -40,7 +52,7 @@ impl AgentPool {
continue;
}
let stale = match decode_server_start_time(job.error.as_deref()) {
Some(t) => t < current_boot,
Some(t) => t < current_boot - STALE_TOLERANCE_SECS,
None => true, // Legacy (pid-encoded) or malformed: stale
};
if stale {
@@ -186,50 +198,6 @@ impl AgentPool {
.map(|k| k.is_self_evident_fix())
.unwrap_or(false);
// Bug 1101 diagnostic: log the classified failure_kind and the
// matched classifier-trigger substring with surrounding context,
// so we can confirm whether classify() is incorrectly matching
// a passing-step stdout substring (e.g. "Diff in " inside a
// failing test's panic message) and bouncing the story to a
// fixup coder. Remove once the fix lands.
if let Ok(r) = report.as_ref()
&& let crate::agents::merge::MergeResult::GateFailure {
output: gate_output,
failure_kind: Some(k),
} = &r.result
{
const TRIGGERS: &[&str] = &[
"CONFLICT (content):",
"Merge conflict:",
"Diff in ",
"would reformat",
"missing-docs direction",
"error[clippy::",
"warning[clippy::",
"missing_doc_comments",
"error[E",
];
let matched = TRIGGERS
.iter()
.find_map(|t| gate_output.find(t).map(|i| (*t, i)));
let (trigger, context) = match matched {
Some((t, i)) => {
let start = i.saturating_sub(30);
let end = (i + t.len() + 60).min(gate_output.len());
let ctx = gate_output
.get(start..end)
.unwrap_or("<context unavailable>")
.replace('\n', " ");
(Some(t), ctx)
}
None => (None, String::from("<no trigger matched>")),
};
slog!(
"[merge] classify diagnostic for '{sid}': failure_kind={k:?} \
is_fixup={is_fixup} trigger={trigger:?} context='{context}'"
);
}
if is_no_commits {
let reason = kind.display_reason();
if let Err(e) = crate::agents::lifecycle::transition_to_blocked(&sid, &reason) {
@@ -249,7 +217,7 @@ impl AgentPool {
// retry_count=1 so maybe_inject_gate_failure injects gate output
// into --append-system-prompt on the fixup spawn.
// transition_to_merge_failure also writes ContentKey::GateOutput.
let display = kind.display_reason();
let display = crate::service::merge::summarize_merge_failure_kind(&kind);
let _ =
crate::agents::lifecycle::transition_to_merge_failure(sid.as_str(), kind);
match crate::agents::lifecycle::move_story_to_stage(&sid, "current") {
@@ -290,7 +258,7 @@ impl AgentPool {
// Transition through the state machine (Merge → MergeFailure).
// Only send the notification when the stage actually changed; if the
// story was already in MergeFailure (self-loop), suppress the duplicate.
let display = kind.display_reason();
let display = crate::service::merge::summarize_merge_failure_kind(&kind);
let should_notify = match crate::agents::lifecycle::transition_to_merge_failure(
sid.as_str(),
kind,
@@ -327,14 +295,21 @@ impl AgentPool {
&& let Ok(ref r) = report
&& r.story_archived
{
pool.set_merge_success_reported(&sid);
pool.set_merge_success_reported(&sid).await;
crate::db::write_content(crate::db::ContentKey::MergeSuccess(&sid), "1");
}
// Update CRDT with terminal status.
// Update CRDT with terminal status. The full untruncated output is
// already on disk (write_merge_report, called from
// run_merge_pipeline for the Ok(r) case above, or below for the
// Err(e) case) — only a bounded summary plus that pointer goes
// into the replicated `merge_jobs.error` field, so a large gate
// failure doesn't bloat every node's CRDT state.
match &report {
Ok(r) => {
let report_json = serde_json::to_string(r).unwrap_or_else(|_| String::new());
let bounded = crate::service::merge::bound_report_for_storage(r);
let report_json =
serde_json::to_string(&bounded).unwrap_or_else(|_| String::new());
crate::crdt_state::write_merge_job(
&sid,
"completed",
@@ -344,12 +319,15 @@ impl AgentPool {
);
}
Err(e) => {
let report_path = crate::service::merge::io::write_merge_report(&root, &sid, e);
let bounded =
crate::service::merge::bound_plain_error(e, report_path.as_deref());
crate::crdt_state::write_merge_job(
&sid,
"failed",
started_at,
Some(finished_at),
Some(e),
Some(&bounded),
);
}
}
@@ -384,24 +362,44 @@ impl AgentPool {
merge_result,
crate::agents::merge::MergeResult::Success { .. }
) {
let report_path = crate::service::merge::io::write_merge_report(
project_root,
story_id,
merge_result.output(),
);
return Ok(crate::agents::merge::MergeReport {
story_id: story_id.to_string(),
result: merge_result,
worktree_cleaned_up: false,
story_archived: false,
report_path,
});
}
let story_archived = crate::agents::lifecycle::move_story_to_done(story_id).is_ok();
if story_archived {
self.remove_agents_for_story(story_id);
}
let report_path = crate::service::merge::io::write_merge_report(
project_root,
story_id,
merge_result.output(),
);
let worktree_cleaned_up = if wt_path.exists() {
let config = crate::config::ProjectConfig::load(project_root).unwrap_or_default();
worktree::remove_worktree_by_story_id(project_root, story_id, &config)
.await
.is_ok()
let story_archived = crate::agents::lifecycle::move_story_to_done(story_id).is_ok();
// Story 1178: only delete the feature branch once the state transition
// to Done is confirmed. Deleting it unconditionally here meant a
// successful squash merge whose CRDT transition failed (e.g. the story
// was in a stage `move_story_to_done` didn't yet handle) would still
// lose its feature branch, making the failure unrecoverable — the
// story couldn't be retried because the branch it needed was gone.
let worktree_cleaned_up = if story_archived {
self.remove_agents_for_story(story_id).await;
if wt_path.exists() {
let config = crate::config::ProjectConfig::load(project_root).unwrap_or_default();
worktree::remove_worktree_by_story_id(project_root, story_id, &config)
.await
.is_ok()
} else {
false
}
} else {
false
};
@@ -413,6 +411,7 @@ impl AgentPool {
result: merge_result,
worktree_cleaned_up,
story_archived,
report_path,
})
}
}
@@ -29,6 +29,7 @@ impl AgentPool {
},
worktree_cleaned_up: false,
story_archived: false,
report_path: None,
});
(crate::agents::merge::MergeJobStatus::Completed(report), 0.0)
}
@@ -41,6 +42,7 @@ impl AgentPool {
story_id: story_id.to_string(),
status,
server_start_time,
started_at: view.started_at,
})
}
}
+138 -22
View File
@@ -34,26 +34,35 @@ fn serial_test_lock() -> std::sync::MutexGuard<'static, ()> {
}
fn init_git_repo(repo: &std::path::Path) {
Command::new("git")
.args(["init"])
.current_dir(repo)
.output()
.unwrap();
Command::new("git")
.args(["config", "user.email", "test@test.com"])
.current_dir(repo)
.output()
.unwrap();
Command::new("git")
.args(["config", "user.name", "Test"])
.current_dir(repo)
.output()
.unwrap();
Command::new("git")
.args(["commit", "--allow-empty", "-m", "init"])
.current_dir(repo)
.output()
.unwrap();
use crate::git_test_support::git_ok;
git_ok(
Command::new("git")
.args(["init"])
.current_dir(repo)
.output(),
"git init",
);
git_ok(
Command::new("git")
.args(["config", "user.email", "test@test.com"])
.current_dir(repo)
.output(),
"git config user.email",
);
git_ok(
Command::new("git")
.args(["config", "user.name", "Test"])
.current_dir(repo)
.output(),
"git config user.name",
);
git_ok(
Command::new("git")
.args(["commit", "--allow-empty", "-m", "init"])
.current_dir(repo)
.output(),
"git commit",
);
}
// ── bug 498: stale Running job blocks retry ───────────────────────────────
@@ -113,6 +122,22 @@ async fn stale_running_merge_job_is_cleared_and_retry_succeeds() {
result.is_ok(),
"start_merge_agent_work must succeed after stale Running job is cleared; got: {result:?}"
);
// start_merge_agent_work spawns the actual pipeline as a background
// tokio task and returns immediately. Wait for it to reach a terminal
// state before the test ends: otherwise the task keeps running after
// `_serial` is released and can still be touching CRDT state
// (write_merge_job / delete_merge_job) while the next merge-pipeline
// test has already called init_for_test(), corrupting that test's
// thread-local state.
loop {
tokio::time::sleep(std::time::Duration::from_millis(50)).await;
if let Some(job) = pool.get_merge_status("77_story_stale")
&& !matches!(job.status, MergeJobStatus::Running)
{
break;
}
}
}
// ── story 852: periodic background reaper ────────────────────────────────
@@ -149,7 +174,7 @@ async fn reap_stale_merge_jobs_removes_old_running_entry_without_merge() {
);
// No agents must have been spawned (no merge was triggered).
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.try_lock().unwrap();
assert!(
agents.is_empty(),
"reap must not spawn any agents; got {} agent(s)",
@@ -393,6 +418,97 @@ async fn merge_agent_work_succeeds_on_clean_branch() {
}
}
/// Regression test (story 1178, AC3): when the squash merge itself succeeds
/// but the CRDT state transition to Done fails (here: no CRDT entry exists
/// for the story, so `move_story_to_done` errors and `story_archived` is
/// false), the feature branch must NOT be deleted — otherwise a retry has
/// nothing to merge from.
#[tokio::test]
async fn merge_success_without_story_archived_keeps_feature_branch() {
let _serial = serial_test_lock();
use std::fs;
use tempfile::tempdir;
crate::crdt_state::init_for_test();
let tmp = tempdir().unwrap();
let repo = tmp.path();
init_git_repo(repo);
let branch = "feature/story-1178_branch_kept";
Command::new("git")
.args(["checkout", "-b", branch])
.current_dir(repo)
.output()
.unwrap();
fs::write(repo.join("feature.txt"), "feature content").unwrap();
Command::new("git")
.args(["add", "."])
.current_dir(repo)
.output()
.unwrap();
Command::new("git")
.args(["commit", "-m", "add feature"])
.current_dir(repo)
.output()
.unwrap();
Command::new("git")
.args(["checkout", "master"])
.current_dir(repo)
.output()
.unwrap();
let merge_dir = repo.join(".huskies/work/4_merge");
fs::create_dir_all(&merge_dir).unwrap();
fs::write(
merge_dir.join("1178_branch_kept.md"),
"---\nname: Branch Kept Test\n---\n",
)
.unwrap();
Command::new("git")
.args(["add", "."])
.current_dir(repo)
.output()
.unwrap();
Command::new("git")
.args(["commit", "-m", "add story in merge"])
.current_dir(repo)
.output()
.unwrap();
let pool = Arc::new(AgentPool::new_test(3001));
// Note: no CRDT entry is written for this story, so `move_story_to_done`
// will fail with NotFound — `story_archived` will be false regardless of
// git merge outcome. That is exactly the scenario this test protects.
let job = run_merge_to_completion(&pool, repo, "1178_branch_kept").await;
let MergeJobStatus::Completed(report) = &job.status else {
panic!("expected a completed job, got: {:?}", job.status);
};
if matches!(
report.result,
crate::agents::merge::MergeResult::Success { .. }
) {
assert!(
!report.story_archived,
"story_archived should be false: no CRDT entry exists for this story"
);
assert!(
!report.worktree_cleaned_up,
"worktree/branch must not be cleaned up when story_archived is false"
);
let branch_check = Command::new("git")
.args(["rev-parse", "--verify", branch])
.current_dir(repo)
.output()
.unwrap();
assert!(
branch_check.status.success(),
"feature branch '{branch}' must still exist after a merge whose state \
transition to Done failed, so a retry stays possible"
);
}
}
// ── quality gate ordering test ────────────────────────────────
/// Regression test for bug 142: quality gates must run BEFORE the fast-forward
@@ -811,7 +927,7 @@ async fn server_side_merge_happy_path_advances_to_done() {
}
// Verify no LLM agent was spawned.
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.try_lock().unwrap();
assert!(
agents.is_empty(),
"no LLM agents should be spawned for deterministic merge; pool has {} agents",
@@ -2,10 +2,10 @@
/// Wall-clock time captured the first time this server process touches the
/// merge subsystem. Used to detect merge_jobs left over from a previous
/// server instance: a re-exec on `rebuild_and_restart` keeps the same PID,
/// so PID alone cannot distinguish "current" vs "previous" server. This
/// timestamp is fresh per-process (the static is reset by execve) and is
/// the source of truth for stale-merge detection.
/// server instance: PIDs can collide across restarts (PID 1 in a container
/// is always the server), so PID alone cannot distinguish "current" vs
/// "previous" server. This timestamp is fresh per-process and is the source
/// of truth for stale-merge detection.
static SERVER_START_TIME: std::sync::OnceLock<f64> = std::sync::OnceLock::new();
/// Return this server process's start time (lazily captured on first call).
+13 -17
View File
@@ -25,11 +25,9 @@ impl AgentPool {
/// continuing to run after the server exits. Collects each agent's worktree
/// path, then SIGKILLs every process running inside that path and verifies
/// termination before returning.
pub fn kill_all_children(&self) {
pub async fn kill_all_children(&self) {
let worktree_paths: Vec<(String, std::path::PathBuf)> = {
let Ok(agents) = self.agents.lock() else {
return;
};
let agents = self.agents.lock().await;
agents
.iter()
.filter_map(|(key, agent)| {
@@ -69,11 +67,9 @@ impl AgentPool {
/// Fallback used by `stop_agent` when no worktree path is recorded for the
/// agent. Also the primary kill path for any caller that has only a composite
/// key and not a worktree path directly.
pub(super) fn kill_child_for_key(&self, key: &str) {
pub(super) async fn kill_child_for_key(&self, key: &str) {
let worktree_path = {
let Ok(agents) = self.agents.lock() else {
return;
};
let agents = self.agents.lock().await;
agents
.get(key)
.and_then(|a| a.worktree_info.as_ref().map(|wt| wt.path.clone()))
@@ -124,18 +120,18 @@ mod tests {
.unwrap_or(false)
}
#[test]
fn kill_all_children_is_safe_on_empty_pool() {
#[tokio::test]
async fn kill_all_children_is_safe_on_empty_pool() {
let pool = AgentPool::new_test(3001);
pool.kill_all_children(); // must not panic
pool.kill_all_children().await; // must not panic
}
/// AC 4 — `kill_child_for_key` SIGKILLs the single agent's process and
/// verifies it is gone within 2 s. The sleeper has the worktree path in
/// its argv[0] so `pgrep -f` can locate it, mirroring how claude-code is
/// launched with `--directory <worktree>` in production.
#[test]
fn kill_child_for_key_kills_real_process() {
#[tokio::test]
async fn kill_child_for_key_kills_real_process() {
use std::os::unix::process::CommandExt;
let pool = AgentPool::new_test(3002);
@@ -165,7 +161,7 @@ mod tests {
"sleeper pid {pid} should be running before kill_child_for_key"
);
pool.kill_child_for_key("story-1090-kill:coder");
pool.kill_child_for_key("story-1090-kill:coder").await;
let _ = child.wait(); // reap zombie so ps -p returns false
assert!(
@@ -176,8 +172,8 @@ mod tests {
/// AC 5 — `kill_all_children` SIGKILLs all agents' processes. Two agents
/// with distinct worktree paths are injected; both must be gone after the call.
#[test]
fn kill_all_children_kills_multiple_real_processes() {
#[tokio::test]
async fn kill_all_children_kills_multiple_real_processes() {
use std::os::unix::process::CommandExt;
let pool = AgentPool::new_test(3003);
@@ -213,7 +209,7 @@ mod tests {
);
}
pool.kill_all_children();
pool.kill_all_children().await;
for (pid, child, _tmp) in &mut sleepers {
let _ = child.wait(); // reap zombie
+45 -16
View File
@@ -10,12 +10,12 @@ use super::types::{agent_info_from_entry, composite_key};
impl AgentPool {
/// Return the names of configured agents for `stage` that are not currently
/// running or pending.
pub fn available_agents_for_stage(
pub async fn available_agents_for_stage(
&self,
config: &ProjectConfig,
stage: &PipelineStage,
) -> Result<Vec<String>, String> {
let agents = self.agents.lock().map_err(|e| e.to_string())?;
let agents = self.agents.lock().await;
Ok(config
.agent
.iter()
@@ -44,8 +44,8 @@ impl AgentPool {
}
/// List all agents with their status.
pub fn list_agents(&self) -> Result<Vec<AgentInfo>, String> {
let agents = self.agents.lock().map_err(|e| e.to_string())?;
pub async fn list_agents(&self) -> Result<Vec<AgentInfo>, String> {
let agents = self.agents.lock().await;
Ok(agents
.iter()
.map(|(key, agent)| {
@@ -59,14 +59,35 @@ impl AgentPool {
.collect())
}
/// Best-effort agent list for sync callers (chat commands, rendering).
///
/// Uses `try_lock()` so it never blocks the thread. Returns an empty list
/// if the lock is currently held — callers that refresh periodically (htop,
/// pipeline board) tolerate this gracefully.
pub fn list_agents_nonblocking(&self) -> Vec<AgentInfo> {
let Ok(agents) = self.agents.try_lock() else {
return Vec::new();
};
agents
.iter()
.map(|(key, agent)| {
let story_id = key
.rsplit_once(':')
.map(|(sid, _)| sid.to_string())
.unwrap_or_else(|| key.clone());
agent_info_from_entry(&story_id, agent)
})
.collect()
}
/// Subscribe to events for a story agent.
pub fn subscribe(
pub async fn subscribe(
&self,
story_id: &str,
agent_name: &str,
) -> Result<broadcast::Receiver<AgentEvent>, String> {
let key = composite_key(story_id, agent_name);
let agents = self.agents.lock().map_err(|e| e.to_string())?;
let agents = self.agents.lock().await;
let agent = agents
.get(&key)
.ok_or_else(|| format!("No agent '{agent_name}' for story '{story_id}'"))?;
@@ -74,13 +95,13 @@ impl AgentPool {
}
/// Drain accumulated events for polling. Returns all events since the last drain.
pub fn drain_events(
pub async fn drain_events(
&self,
story_id: &str,
agent_name: &str,
) -> Result<Vec<AgentEvent>, String> {
let key = composite_key(story_id, agent_name);
let agents = self.agents.lock().map_err(|e| e.to_string())?;
let agents = self.agents.lock().await;
let agent = agents
.get(&key)
.ok_or_else(|| format!("No agent '{agent_name}' for story '{story_id}'"))?;
@@ -91,9 +112,13 @@ impl AgentPool {
/// Get the log session ID and project root for an agent, if available.
///
/// Used by MCP tools to find the persistent log file for a completed agent.
pub fn get_log_info(&self, story_id: &str, agent_name: &str) -> Option<(String, PathBuf)> {
pub async fn get_log_info(
&self,
story_id: &str,
agent_name: &str,
) -> Option<(String, PathBuf)> {
let key = composite_key(story_id, agent_name);
let agents = self.agents.lock().ok()?;
let agents = self.agents.lock().await;
let agent = agents.get(&key)?;
let session_id = agent.log_session_id.clone()?;
let project_root = agent.project_root.clone()?;
@@ -111,8 +136,8 @@ mod tests {
ProjectConfig::parse(toml_str).unwrap()
}
#[test]
fn available_agents_for_stage_returns_idle_agents() {
#[tokio::test]
async fn available_agents_for_stage_returns_idle_agents() {
let config = make_config(
r#"
[[agent]]
@@ -133,17 +158,19 @@ stage = "qa"
let available = pool
.available_agents_for_stage(&config, &PipelineStage::Coder)
.await
.unwrap();
assert_eq!(available, vec!["coder-2"]);
let available_qa = pool
.available_agents_for_stage(&config, &PipelineStage::Qa)
.await
.unwrap();
assert_eq!(available_qa, vec!["qa"]);
}
#[test]
fn available_agents_for_stage_returns_empty_when_all_busy() {
#[tokio::test]
async fn available_agents_for_stage_returns_empty_when_all_busy() {
let config = make_config(
r#"
[[agent]]
@@ -156,12 +183,13 @@ stage = "coder"
let available = pool
.available_agents_for_stage(&config, &PipelineStage::Coder)
.await
.unwrap();
assert!(available.is_empty());
}
#[test]
fn available_agents_for_stage_ignores_completed_agents() {
#[tokio::test]
async fn available_agents_for_stage_ignores_completed_agents() {
let config = make_config(
r#"
[[agent]]
@@ -174,6 +202,7 @@ stage = "coder"
let available = pool
.available_agents_for_stage(&config, &PipelineStage::Coder)
.await
.unwrap();
assert_eq!(available, vec!["coder-1"]);
}
+27 -24
View File
@@ -3,8 +3,10 @@
use crate::agent_log::AgentLogWriter;
use crate::config::ProjectConfig;
use crate::slog_error;
use std::future::Future;
use std::path::Path;
use std::sync::{Arc, Mutex};
use std::pin::Pin;
use std::sync::Arc;
use tokio::sync::broadcast;
use super::super::runtime::{
@@ -38,48 +40,48 @@ impl AgentPool {
/// `resume_context` (if any) is sent as the new message. This lets
/// the agent re-enter the previous conversation without re-reading
/// CLAUDE.md and README, satisfying story 543.
pub async fn start_agent(
&self,
project_root: &Path,
story_id: &str,
agent_name: Option<&str>,
resume_context: Option<&str>,
pub fn start_agent<'a>(
&'a self,
project_root: &'a Path,
story_id: &'a str,
agent_name: Option<&'a str>,
resume_context: Option<&'a str>,
session_id_to_resume: Option<String>,
) -> Result<AgentInfo, String> {
self.start_agent_inner(
) -> Pin<Box<dyn Future<Output = Result<AgentInfo, String>> + Send + 'a>> {
Box::pin(self.start_agent_inner(
project_root,
story_id,
agent_name,
resume_context,
session_id_to_resume,
None,
)
))
}
/// Start an agent with an `AppContext` for direct MCP tool dispatch.
///
/// API-based runtimes (Gemini, OpenAI) need the `AppContext` to invoke MCP
/// tools without an HTTP round-trip. CLI-based runtimes (Claude Code) do not.
pub fn start_agent_with_ctx(
&self,
project_root: &Path,
story_id: &str,
agent_name: Option<&str>,
resume_context: Option<&str>,
pub fn start_agent_with_ctx<'a>(
&'a self,
project_root: &'a Path,
story_id: &'a str,
agent_name: Option<&'a str>,
resume_context: Option<&'a str>,
session_id_to_resume: Option<String>,
app_ctx: Arc<crate::http::context::AppContext>,
) -> Result<AgentInfo, String> {
self.start_agent_inner(
) -> Pin<Box<dyn Future<Output = Result<AgentInfo, String>> + Send + 'a>> {
Box::pin(self.start_agent_inner(
project_root,
story_id,
agent_name,
resume_context,
session_id_to_resume,
Some(app_ctx),
)
))
}
fn start_agent_inner(
async fn start_agent_inner(
&self,
project_root: &Path,
story_id: &str,
@@ -100,7 +102,8 @@ impl AgentPool {
// Create name-independent shared resources before the lock so they are
// ready for the atomic check-and-insert (story 132).
let (tx, _) = broadcast::channel::<AgentEvent>(1024);
let event_log: Arc<Mutex<Vec<AgentEvent>>> = Arc::new(Mutex::new(Vec::new()));
let event_log: Arc<std::sync::Mutex<Vec<AgentEvent>>> =
Arc::new(std::sync::Mutex::new(Vec::new()));
let log_session_id = uuid::Uuid::new_v4().to_string();
// Create the per-session status buffer subscribed to this project's
@@ -149,7 +152,7 @@ impl AgentPool {
// agent turn (story 736).
let prior_events: Option<String>;
{
let mut agents = self.agents.lock().map_err(|e| e.to_string())?;
let mut agents = self.agents.lock().await;
resolved_name = match agent_name {
Some(name) => name.to_string(),
@@ -371,7 +374,7 @@ impl AgentPool {
// the atomic resolution above).
let log_writer =
match AgentLogWriter::new(project_root, story_id, &resolved_name, &log_session_id) {
Ok(w) => Some(Arc::new(Mutex::new(w))),
Ok(w) => Some(Arc::new(std::sync::Mutex::new(w))),
Err(e) => {
eprintln!(
"[agents] Failed to create log writer for {story_id}:{resolved_name}: {e}"
@@ -436,7 +439,7 @@ impl AgentPool {
// Store the task handle while the agent is still Pending.
{
let mut agents = self.agents.lock().map_err(|e| e.to_string())?;
let mut agents = self.agents.lock().await;
if let Some(agent) = agents.get_mut(&key) {
agent.task_handle = Some(handle);
}
+134 -72
View File
@@ -116,6 +116,23 @@ pub(super) fn maybe_inject_gate_failure(args: &mut Vec<String>, story_id: &str)
}
}
/// Append `Edit,Write,Bash` to the `--disallowedTools` flag so worktree agents
/// cannot write to the master tree via Claude's built-in tools. If
/// `--disallowedTools` is already present (from agent config), the three names
/// are appended to the existing value rather than replacing it.
pub(super) fn inject_worktree_disallowed_tools(args: &mut Vec<String>) {
const BLOCKED: &str = "Edit,Write,Bash";
if let Some(pos) = args.iter().position(|a| a == "--disallowedTools") {
if let Some(val) = args.get_mut(pos + 1) {
val.push(',');
val.push_str(BLOCKED);
}
} else {
args.push("--disallowedTools".to_string());
args.push(BLOCKED.to_string());
}
}
/// Run the background worktree-creation + agent-launch flow.
///
/// Caller (`AgentPool::start_agent`) wraps this in `tokio::spawn` and stores
@@ -129,7 +146,7 @@ pub(super) async fn run_agent_spawn(
story_id: String,
agent_name: String,
tx: broadcast::Sender<AgentEvent>,
agents: Arc<Mutex<HashMap<String, StoryAgent>>>,
agents: Arc<tokio::sync::Mutex<HashMap<String, StoryAgent>>>,
key: String,
event_log: Arc<Mutex<Vec<AgentEvent>>>,
port: u16,
@@ -172,10 +189,10 @@ pub(super) async fn run_agent_spawn(
let wt_info = {
let wt_path = crate::worktree::worktree_path(&project_root_clone, &sid);
let branch = format!("feature/story-{sid}");
let base_branch = config_clone
.base_branch
.clone()
.unwrap_or_else(|| crate::worktree::detect_base_branch(&project_root_clone));
let base_branch = crate::worktree::resolve_base_branch(
&project_root_clone,
config_clone.base_branch.as_deref(),
);
let deadline =
tokio::time::Instant::now() + std::time::Duration::from_secs(worktree_wait_secs);
loop {
@@ -201,10 +218,11 @@ pub(super) async fn run_agent_spawn(
log.push(event.clone());
}
let _ = tx_clone.send(event);
if let Ok(mut agents) = agents_ref.lock()
&& let Some(agent) = agents.get_mut(&key_clone)
{
agent.status = AgentStatus::Failed;
let mut agents = agents_ref.lock().await;
if let Some(agent) = agents.get_mut(&key_clone) {
agent.status = AgentStatus::Failed;
}
}
AgentPool::notify_agent_state_changed(&watcher_tx_clone);
return;
@@ -216,21 +234,33 @@ pub(super) async fn run_agent_spawn(
// Step 1.1: Install the pre-commit quality-gate hook in the worktree.
// Non-fatal — if installation fails the agent can still run; the failure
// is logged so the operator can investigate.
if let Err(e) = crate::worktree::install_pre_commit_hook(&wt_info.path) {
slog_error!("[agents] pre-commit hook install failed for {sid}: {e}");
// Runs in spawn_blocking because install_pre_commit_hook executes
// synchronous git-config subprocesses that would otherwise pin a
// tokio worker thread and contribute to runtime starvation under
// concurrent agent spawns.
{
let hook_path = wt_info.path.clone();
let hook_result = tokio::task::spawn_blocking(move || {
crate::worktree::install_pre_commit_hook(&hook_path)
})
.await
.unwrap_or_else(|e| Err(format!("spawn_blocking panicked: {e}")));
if let Err(e) = hook_result {
slog_error!("[agents] pre-commit hook install failed for {sid}: {e}");
}
}
// Step 2: store worktree info and render agent command/args/prompt.
let wt_path_str = wt_info.path.to_string_lossy().to_string();
{
if let Ok(mut agents) = agents_ref.lock()
&& let Some(agent) = agents.get_mut(&key_clone)
{
let mut agents = agents_ref.lock().await;
if let Some(agent) = agents.get_mut(&key_clone) {
agent.worktree_info = Some(wt_info.clone());
}
}
let (command, mut args, mut prompt) = match config_clone.render_agent_args(
&project_root_clone,
&wt_path_str,
&sid,
Some(&aname),
@@ -249,10 +279,11 @@ pub(super) async fn run_agent_spawn(
log.push(event.clone());
}
let _ = tx_clone.send(event);
if let Ok(mut agents) = agents_ref.lock()
&& let Some(agent) = agents.get_mut(&key_clone)
{
agent.status = AgentStatus::Failed;
let mut agents = agents_ref.lock().await;
if let Some(agent) = agents.get_mut(&key_clone) {
agent.status = AgentStatus::Failed;
}
}
AgentPool::notify_agent_state_changed(&watcher_tx_clone);
return;
@@ -264,6 +295,10 @@ pub(super) async fn run_agent_spawn(
maybe_inject_gate_failure(&mut args, &sid);
// Cap turns and budget for merge-gate fixup sessions (story 981).
maybe_cap_for_merge_fixup(&mut args, &sid);
// Every agent that runs inside a worktree must use the validated MCP
// edit/write tools instead of Claude's built-in Edit/Write/Bash. This
// prevents accidental writes to the master worktree (stories 1127, 1136).
inject_worktree_disallowed_tools(&mut args);
// Append project-local prompt content (.huskies/AGENT.md) to the
// baked-in prompt so every agent role sees project-specific guidance
@@ -337,9 +372,8 @@ pub(super) async fn run_agent_spawn(
// Step 3: transition to Running now that the worktree is ready.
{
if let Ok(mut agents) = agents_ref.lock()
&& let Some(agent) = agents.get_mut(&key_clone)
{
let mut agents = agents_ref.lock().await;
if let Some(agent) = agents.get_mut(&key_clone) {
agent.status = AgentStatus::Running;
}
}
@@ -436,25 +470,26 @@ pub(super) async fn run_agent_spawn(
match run_result {
Ok(result) => {
// Persist token usage if the agent reported it.
if let Some(ref usage) = result.token_usage
&& let Ok(agents) = agents_ref.lock()
&& let Some(agent) = agents.get(&key_clone)
&& let Some(ref pr) = agent.project_root
{
let model_for_record = config_clone
.find_agent(&aname)
.and_then(|a| a.model.clone());
let record = crate::agents::token_usage::build_record(
&sid,
&aname,
model_for_record,
usage.clone(),
);
if let Err(e) = crate::agents::token_usage::append_record(pr, &record) {
slog_error!(
"[agents] Failed to persist token usage for \
{sid}:{aname}: {e}"
if let Some(ref usage) = result.token_usage {
let agents = agents_ref.lock().await;
if let Some(agent) = agents.get(&key_clone)
&& let Some(ref pr) = agent.project_root
{
let model_for_record = config_clone
.find_agent(&aname)
.and_then(|a| a.model.clone());
let record = crate::agents::token_usage::build_record(
&sid,
&aname,
model_for_record,
usage.clone(),
);
if let Err(e) = crate::agents::token_usage::append_record(pr, &record) {
slog_error!(
"[agents] Failed to persist token usage for \
{sid}:{aname}: {e}"
);
}
}
}
@@ -505,10 +540,7 @@ pub(super) async fn run_agent_spawn(
// Remove the agent entry from the pool and emit Done so that
// any caller blocked on wait_for_agent is unblocked.
let tx_done = {
let mut lock = match agents_ref.lock() {
Ok(a) => a,
Err(_) => return,
};
let mut lock = agents_ref.lock().await;
if let Some(agent) = lock.remove(&key_clone) {
agent.tx
} else {
@@ -587,10 +619,7 @@ pub(super) async fn run_agent_spawn(
if stage == PipelineStage::Mergemaster {
let (tx_done, done_session_id, merge_failure_reported, merge_success_reported) = {
let mut lock = match agents_ref.lock() {
Ok(a) => a,
Err(_) => return,
};
let mut lock = agents_ref.lock().await;
if let Some(agent) = lock.remove(&key_clone) {
(
agent.tx,
@@ -627,15 +656,14 @@ pub(super) async fn run_agent_spawn(
// Do NOT send WorkItem/reassign — story is already Done.
// Drain one queued ConflictDetected story now that this
// mergemaster slot is free (story 1044).
if let Some((candidate_id, candidate_agent)) =
crate::config::ProjectConfig::load(&project_root_clone)
.ok()
.and_then(|cfg| {
agents_ref.lock().ok().as_ref().and_then(|agts| {
pick_queued_conflict_detected(&cfg, agts, &sid)
})
})
{
let candidate =
if let Ok(cfg) = crate::config::ProjectConfig::load(&project_root_clone) {
let agts = agents_ref.lock().await;
pick_queued_conflict_detected(&cfg, &agts, &sid)
} else {
None
};
if let Some((candidate_id, candidate_agent)) = candidate {
slog!(
"[agents] Mergemaster exit for '{sid}' (success): \
queued ConflictDetected story '{candidate_id}' found; \
@@ -745,17 +773,14 @@ pub(super) async fn run_agent_spawn(
});
// Drain one queued ConflictDetected story now that this
// mergemaster slot is free (story 1044).
if let Some((candidate_id, candidate_agent)) =
crate::config::ProjectConfig::load(&project_root_clone)
.ok()
.and_then(|cfg| {
agents_ref
.lock()
.ok()
.as_ref()
.and_then(|agts| pick_queued_conflict_detected(&cfg, agts, &sid))
})
{
let candidate =
if let Ok(cfg) = crate::config::ProjectConfig::load(&project_root_clone) {
let agts = agents_ref.lock().await;
pick_queued_conflict_detected(&cfg, &agts, &sid)
} else {
None
};
if let Some((candidate_id, candidate_agent)) = candidate {
slog!(
"[agents] Mergemaster exit for '{sid}': queued ConflictDetected \
story '{candidate_id}' found; spawning '{candidate_agent}'."
@@ -812,10 +837,7 @@ pub(super) async fn run_agent_spawn(
// Remove agent from the pool and unblock any wait_for_agent callers.
let tx_done = {
let mut lock = match agents_ref.lock() {
Ok(a) => a,
Err(_) => return,
};
let mut lock = agents_ref.lock().await;
if let Some(agent) = lock.remove(&key_clone) {
agent.tx
} else {
@@ -910,10 +932,11 @@ pub(super) async fn run_agent_spawn(
log.push(event.clone());
}
let _ = tx_clone.send(event);
if let Ok(mut agents) = agents_ref.lock()
&& let Some(agent) = agents.get_mut(&key_clone)
{
agent.status = AgentStatus::Failed;
let mut agents = agents_ref.lock().await;
if let Some(agent) = agents.get_mut(&key_clone) {
agent.status = AgentStatus::Failed;
}
}
AgentPool::notify_agent_state_changed(&watcher_tx_clone);
}
@@ -1297,4 +1320,43 @@ mod tests {
item.stage().dir_name()
);
}
// ── inject_worktree_disallowed_tools (AC1, story 1142) ───────────
/// AC3(c) proxy: worktree agents get `--disallowedTools Edit,Write,Bash`.
#[test]
fn worktree_disallowed_tools_added_when_absent() {
let mut args: Vec<String> = vec!["--verbose".to_string()];
inject_worktree_disallowed_tools(&mut args);
let pos = args
.iter()
.position(|a| a == "--disallowedTools")
.expect("--disallowedTools must be present");
let val = &args[pos + 1];
assert!(val.contains("Edit"), "must include Edit");
assert!(val.contains("Write"), "must include Write");
assert!(val.contains("Bash"), "must include Bash");
}
/// Existing `--disallowedTools` value is extended, not replaced.
#[test]
fn worktree_disallowed_tools_appended_to_existing() {
let mut args = vec!["--disallowedTools".to_string(), "SomeOtherTool".to_string()];
inject_worktree_disallowed_tools(&mut args);
// Only one --disallowedTools flag.
let count = args
.iter()
.filter(|a| a.as_str() == "--disallowedTools")
.count();
assert_eq!(count, 1, "must not duplicate --disallowedTools");
let pos = args.iter().position(|a| a == "--disallowedTools").unwrap();
let val = &args[pos + 1];
assert!(
val.contains("SomeOtherTool"),
"original tool must be preserved"
);
assert!(val.contains("Edit"), "Edit must be added");
assert!(val.contains("Write"), "Write must be added");
assert!(val.contains("Bash"), "Bash must be added");
}
}
@@ -109,7 +109,7 @@ async fn start_agent_cleans_up_pending_entry_on_failure() {
"agent must transition to Failed after worktree creation error"
);
let agents = pool.agents.lock().unwrap();
let agents = pool.agents.lock().await;
let failed_entry = agents
.values()
.find(|a| a.agent_name == "coder-1" && a.status == AgentStatus::Failed);
@@ -121,6 +121,7 @@ async fn start_agent_cleans_up_pending_entry_on_failure() {
let events = pool
.drain_events("50_story_test", "coder-1")
.await
.expect("drain_events should succeed");
let has_error_event = events.iter().any(|e| matches!(e, AgentEvent::Error { .. }));
assert!(
@@ -736,7 +737,7 @@ async fn reconcile_canonical_agents_stops_stale_coder_in_qa_stage() {
let pool = AgentPool::new_test(3099);
pool.inject_test_agent("777_story_reconcile", "coder-1", AgentStatus::Running);
let before = pool.list_agents().unwrap();
let before = pool.list_agents().await.unwrap();
assert!(
before.iter().any(|a| a.agent_name == "coder-1"
&& matches!(a.status, AgentStatus::Running | AgentStatus::Pending)),
@@ -745,7 +746,7 @@ async fn reconcile_canonical_agents_stops_stale_coder_in_qa_stage() {
pool.reconcile_canonical_agents(root).await;
let after = pool.list_agents().unwrap();
let after = pool.list_agents().await.unwrap();
let still_active = after.iter().any(|a| {
a.story_id == "777_story_reconcile"
&& a.agent_name == "coder-1"
@@ -786,7 +787,7 @@ async fn reconcile_canonical_agents_leaves_correct_stage_agent_alone() {
pool.reconcile_canonical_agents(root).await;
let after = pool.list_agents().unwrap();
let after = pool.list_agents().await.unwrap();
let still_active = after.iter().any(|a| {
a.story_id == "555_story_correct"
&& a.agent_name == "coder-1"
@@ -851,7 +852,7 @@ async fn regression_1100_stale_coder_blocks_mergemaster_then_reconciler_clears()
pool.reconcile_canonical_agents(root).await;
// coder-1 must be gone from the active pool.
let remaining = pool.list_agents().unwrap();
let remaining = pool.list_agents().await.unwrap();
assert!(
!remaining.iter().any(|a| {
a.story_id == "1100_reg"
+18 -27
View File
@@ -1,7 +1,6 @@
//! Agent stop — terminates a running agent while preserving its worktree.
use crate::process_kill::{pids_matching, sigkill_pids_and_verify};
use crate::slog;
use crate::slog_error;
use crate::slog_warn;
use std::path::Path;
@@ -40,7 +39,7 @@ impl AgentPool {
// Step 1: snapshot the worktree path (no status mutation yet).
let worktree_info = {
let agents = self.agents.lock().map_err(|e| e.to_string())?;
let agents = self.agents.lock().await;
let agent = agents
.get(&key)
.ok_or_else(|| format!("No agent '{agent_name}' for story '{story_id}'"))?;
@@ -71,12 +70,12 @@ impl AgentPool {
"[stop_agent] No worktree path recorded for '{key}'; cannot tree-kill, \
falling back to portable_pty SIGHUP (likely no-op for claude-code)."
);
self.kill_child_for_key(&key);
self.kill_child_for_key(&key).await;
}
// Step 3: now safe to mutate. Status flip and handle abort.
let (task_handle, tx) = {
let mut agents = self.agents.lock().map_err(|e| e.to_string())?;
let mut agents = self.agents.lock().await;
let agent = agents
.get_mut(&key)
.ok_or_else(|| format!("No agent '{agent_name}' for story '{story_id}'"))?;
@@ -107,7 +106,7 @@ impl AgentPool {
// Remove from map.
{
let mut agents = self.agents.lock().map_err(|e| e.to_string())?;
let mut agents = self.agents.lock().await;
agents.remove(&key);
}
@@ -138,9 +137,7 @@ impl AgentPool {
// Snapshot active LLM agents without holding the lock during async stops.
let snapshot: Vec<(String, String, PipelineStage)> = {
let Ok(agents) = self.agents.lock() else {
return;
};
let agents = self.agents.lock().await;
agents
.iter()
.filter_map(|(key, a)| {
@@ -197,14 +194,8 @@ impl AgentPool {
///
/// Called when a story is archived so that stale entries don't accumulate.
/// Returns the number of entries removed.
pub fn remove_agents_for_story(&self, story_id: &str) -> usize {
let mut agents = match self.agents.lock() {
Ok(a) => a,
Err(e) => {
slog_error!("[agents] Failed to lock pool for cleanup of '{story_id}': {e}");
return 0;
}
};
pub async fn remove_agents_for_story(&self, story_id: &str) -> usize {
let mut agents = self.agents.lock().await;
let prefix = format!("{story_id}:");
let keys_to_remove: Vec<String> = agents
.keys()
@@ -229,30 +220,30 @@ mod tests {
// ── remove_agents_for_story tests ────────────────────────────────────────
#[test]
fn remove_agents_for_story_removes_all_entries() {
#[tokio::test]
async fn remove_agents_for_story_removes_all_entries() {
let pool = AgentPool::new_test(3001);
pool.inject_test_agent("story_a", "coder-1", AgentStatus::Completed);
pool.inject_test_agent("story_a", "qa", AgentStatus::Failed);
pool.inject_test_agent("story_b", "coder-1", AgentStatus::Running);
let removed = pool.remove_agents_for_story("story_a");
let removed = pool.remove_agents_for_story("story_a").await;
assert_eq!(removed, 2, "should remove both agents for story_a");
let agents = pool.list_agents().unwrap();
let agents = pool.list_agents().await.unwrap();
assert_eq!(agents.len(), 1, "only story_b agent should remain");
assert_eq!(agents[0].story_id, "story_b");
}
#[test]
fn remove_agents_for_story_returns_zero_when_no_match() {
#[tokio::test]
async fn remove_agents_for_story_returns_zero_when_no_match() {
let pool = AgentPool::new_test(3001);
pool.inject_test_agent("story_a", "coder-1", AgentStatus::Running);
let removed = pool.remove_agents_for_story("nonexistent");
let removed = pool.remove_agents_for_story("nonexistent").await;
assert_eq!(removed, 0);
let agents = pool.list_agents().unwrap();
let agents = pool.list_agents().await.unwrap();
assert_eq!(agents.len(), 1, "existing agents should not be affected");
}
@@ -283,12 +274,12 @@ mod tests {
pool.inject_test_agent("60_story_cleanup", "qa", AgentStatus::Completed);
pool.inject_test_agent("61_story_other", "coder-1", AgentStatus::Running);
assert_eq!(pool.list_agents().unwrap().len(), 3);
assert_eq!(pool.list_agents().await.unwrap().len(), 3);
move_story_to_done("60_story_cleanup").unwrap();
pool.remove_agents_for_story("60_story_cleanup");
pool.remove_agents_for_story("60_story_cleanup").await;
let remaining = pool.list_agents().unwrap();
let remaining = pool.list_agents().await.unwrap();
assert_eq!(
remaining.len(),
1,
+8 -8
View File
@@ -20,7 +20,7 @@ impl AgentPool {
) -> broadcast::Sender<AgentEvent> {
let (tx, _) = broadcast::channel::<AgentEvent>(64);
let key = composite_key(story_id, agent_name);
let mut agents = self.agents.lock().unwrap();
let mut agents = self.agents.try_lock().unwrap();
agents.insert(
key,
StoryAgent {
@@ -55,7 +55,7 @@ impl AgentPool {
) -> broadcast::Sender<AgentEvent> {
let (tx, _) = broadcast::channel::<AgentEvent>(64);
let key = composite_key(story_id, agent_name);
let mut agents = self.agents.lock().unwrap();
let mut agents = self.agents.try_lock().unwrap();
agents.insert(
key,
StoryAgent {
@@ -95,7 +95,7 @@ impl AgentPool {
) -> broadcast::Sender<AgentEvent> {
let (tx, _) = broadcast::channel::<AgentEvent>(64);
let key = composite_key(story_id, agent_name);
let mut agents = self.agents.lock().unwrap();
let mut agents = self.agents.try_lock().unwrap();
agents.insert(
key,
StoryAgent {
@@ -130,7 +130,7 @@ impl AgentPool {
) -> broadcast::Sender<AgentEvent> {
let (tx, _) = broadcast::channel::<AgentEvent>(64);
let key = composite_key(story_id, agent_name);
let mut agents = self.agents.lock().unwrap();
let mut agents = self.agents.try_lock().unwrap();
agents.insert(
key,
StoryAgent {
@@ -165,7 +165,7 @@ impl AgentPool {
) -> broadcast::Sender<AgentEvent> {
let (tx, _) = broadcast::channel::<AgentEvent>(64);
let key = composite_key(story_id, agent_name);
let mut agents = self.agents.lock().unwrap();
let mut agents = self.agents.try_lock().unwrap();
agents.insert(
key,
StoryAgent {
@@ -197,7 +197,7 @@ impl AgentPool {
story_id: &str,
agent_name: &str,
) -> Option<Vec<BufferedItem>> {
let agents = self.agents.lock().unwrap();
let agents = self.agents.try_lock().unwrap();
let key = composite_key(story_id, agent_name);
agents
.get(&key)
@@ -219,7 +219,7 @@ impl AgentPool {
) -> broadcast::Sender<AgentEvent> {
let (tx, _) = broadcast::channel::<AgentEvent>(64);
let key = composite_key(story_id, agent_name);
let mut agents = self.agents.lock().unwrap();
let mut agents = self.agents.try_lock().unwrap();
agents.insert(
key,
StoryAgent {
@@ -258,7 +258,7 @@ impl AgentPool {
) -> broadcast::Sender<AgentEvent> {
let (tx, _) = broadcast::channel::<AgentEvent>(64);
let key = composite_key(story_id, agent_name);
let mut agents = self.agents.lock().unwrap();
let mut agents = self.agents.try_lock().unwrap();
agents.insert(
key,
StoryAgent {
+7 -5
View File
@@ -3,8 +3,8 @@ use crate::slog;
use crate::worktree::WorktreeInfo;
use std::collections::HashMap;
use std::path::PathBuf;
use std::sync::{Arc, Mutex};
use tokio::sync::broadcast;
use std::sync::Arc;
use tokio::sync::{Mutex, broadcast};
use super::super::{AgentEvent, AgentInfo, AgentStatus, CompletionReport};
@@ -45,8 +45,10 @@ impl PendingGuard {
impl Drop for PendingGuard {
fn drop(&mut self) {
if self.armed
&& let Ok(mut agents) = self.agents.lock()
if !self.armed {
return;
}
if let Ok(mut agents) = self.agents.try_lock()
&& agents
.get(&self.key)
.is_some_and(|a| a.status == AgentStatus::Pending)
@@ -68,7 +70,7 @@ pub(super) struct StoryAgent {
pub(super) tx: broadcast::Sender<AgentEvent>,
pub(super) task_handle: Option<tokio::task::JoinHandle<()>>,
/// Accumulated events for polling via get_agent_output.
pub(super) event_log: Arc<Mutex<Vec<AgentEvent>>>,
pub(super) event_log: Arc<std::sync::Mutex<Vec<AgentEvent>>>,
/// Set when the agent calls report_completion.
pub(super) completion: Option<CompletionReport>,
/// Project root, stored for pipeline advancement after completion.
+5 -5
View File
@@ -17,11 +17,11 @@ impl AgentPool {
) -> Result<AgentInfo, String> {
// Subscribe before checking status so we don't miss the terminal event
// if the agent completes in the window between the two operations.
let mut rx = self.subscribe(story_id, agent_name)?;
let mut rx = self.subscribe(story_id, agent_name).await?;
// Return immediately if already in a terminal state.
{
let agents = self.agents.lock().map_err(|e| e.to_string())?;
let agents = self.agents.lock().await;
let key = composite_key(story_id, agent_name);
if let Some(agent) = agents.get(&key)
&& matches!(agent.status, AgentStatus::Completed | AgentStatus::Failed)
@@ -48,7 +48,7 @@ impl AgentPool {
_ => false,
};
if is_terminal {
let agents = self.agents.lock().map_err(|e| e.to_string())?;
let agents = self.agents.lock().await;
let key = composite_key(story_id, agent_name);
return Ok(if let Some(agent) = agents.get(&key) {
agent_info_from_entry(story_id, agent)
@@ -78,7 +78,7 @@ impl AgentPool {
}
Ok(Err(broadcast::error::RecvError::Lagged(_))) => {
// Missed some buffered events — check current status before resuming.
let agents = self.agents.lock().map_err(|e| e.to_string())?;
let agents = self.agents.lock().await;
let key = composite_key(story_id, agent_name);
if let Some(agent) = agents.get(&key)
&& matches!(agent.status, AgentStatus::Completed | AgentStatus::Failed)
@@ -89,7 +89,7 @@ impl AgentPool {
}
Ok(Err(broadcast::error::RecvError::Closed)) => {
// Channel closed: no more events will arrive. Return current state.
let agents = self.agents.lock().map_err(|e| e.to_string())?;
let agents = self.agents.lock().await;
let key = composite_key(story_id, agent_name);
if let Some(agent) = agents.get(&key) {
return Ok(agent_info_from_entry(story_id, agent));
+47 -14
View File
@@ -129,7 +129,22 @@ pub(crate) async fn on_coding_transition(project_root: &Path, port: u16, story_i
"[worktree-create-sub] Worktree ready for '{story_id}' at {}",
info.path.display()
);
if let Err(e) = crate::worktree::install_pre_commit_hook(&info.path) {
let hook_path = info.path.clone();
slog!("[worktree] starting pre-commit hook install for '{story_id}'");
let hook_result = tokio::task::spawn_blocking(move || {
crate::worktree::install_pre_commit_hook(&hook_path)
})
.await
.unwrap_or_else(|e| Err(format!("spawn_blocking panicked: {e}")));
match &hook_result {
Ok(()) => {
slog!("[worktree] finished pre-commit hook install for '{story_id}' result=ok")
}
Err(e) => slog!(
"[worktree] finished pre-commit hook install for '{story_id}' result=err: {e}"
),
}
if let Err(e) = hook_result {
slog_warn!(
"[worktree-create-sub] Pre-commit hook install failed for '{story_id}': {e}"
);
@@ -142,6 +157,12 @@ pub(crate) async fn on_coding_transition(project_root: &Path, port: u16, story_i
}
/// Remove the worktree and feature branch for `story_id` after it reaches a terminal stage.
///
/// Story 1199: bytes-reclaimed accounting for the worktree's `target/` dir
/// happens one layer down, in `remove_worktree` — the single choke point
/// shared by this subscriber, the `remove_worktree` MCP tool, and
/// `worktree::cleanup`/`sweep`. Logging it there (once) avoids a duplicate
/// log line here.
pub(crate) async fn on_terminal_transition(project_root: &Path, story_id: &str) {
let config = match crate::config::ProjectConfig::load(project_root) {
Ok(c) => c,
@@ -163,23 +184,11 @@ pub(crate) async fn on_terminal_transition(project_root: &Path, story_id: &str)
#[cfg(test)]
mod tests {
use super::*;
use crate::worktree::test_support::init_git_repo;
use std::fs;
use std::process::Command;
use tempfile::TempDir;
fn init_git_repo(dir: &Path) {
Command::new("git")
.args(["init"])
.current_dir(dir)
.output()
.expect("git init");
Command::new("git")
.args(["commit", "--allow-empty", "-m", "init"])
.current_dir(dir)
.output()
.expect("git commit");
}
fn setup_project(tmp: &TempDir) -> PathBuf {
let root = tmp.path().join("project");
fs::create_dir_all(root.join(".huskies")).unwrap();
@@ -255,6 +264,30 @@ mod tests {
);
}
/// Story 1199 AC4: on_terminal_transition reclaims the worktree's target/
/// dir (and everything else in the worktree) when removing it.
#[tokio::test]
async fn terminal_transition_reclaims_target_dir() {
let tmp = TempDir::new().unwrap();
let root = setup_project(&tmp);
let story_id = "1006_test_reclaim_target";
on_coding_transition(&root, 3001, story_id).await;
let wt_path = crate::worktree::worktree_path(&root, story_id);
let target_dir = wt_path.join("target");
fs::create_dir_all(&target_dir).unwrap();
fs::write(target_dir.join("build_artifact.bin"), vec![0u8; 512]).unwrap();
assert!(target_dir.exists(), "target/ must exist before cleanup");
on_terminal_transition(&root, story_id).await;
assert!(
!target_dir.exists(),
"target/ dir must be reclaimed when the worktree is removed"
);
assert!(!wt_path.exists());
}
/// AC2: on_terminal_transition is a no-op (non-fatal) when no worktree exists.
#[tokio::test]
async fn terminal_transition_noop_when_no_worktree() {
+220
View File
@@ -247,6 +247,104 @@ mod tests {
);
}
// ── story 1196: in-flight MCP tool calls suspend the inactivity deadline ──
/// A `tool_use` block in an `assistant` message marks a call as in
/// flight; the inactivity deadline must be suspended while it is
/// outstanding (so a slow MCP tool like `run_tests` doesn't get the
/// agent killed), and must resume as soon as the matching `tool_result`
/// arrives in a `user` message.
///
/// Script: emits a tool_use, sleeps 2s (would fail a 1s timeout without
/// suspension), emits the matching tool_result, then sleeps 2s again
/// (now with no call in flight, the 1s timeout must fire).
#[tokio::test]
async fn tool_call_in_flight_suspends_then_resumes_inactivity_deadline() {
use std::os::unix::fs::PermissionsExt;
let tmp = tempfile::tempdir().unwrap();
let script = tmp.path().join("tool_call_then_silence.sh");
let body = "#!/bin/sh\n\
printf '%s\\n' '{\"type\":\"assistant\",\"message\":{\"content\":[{\"type\":\"tool_use\",\"id\":\"tool1\",\"name\":\"run_tests\",\"input\":{}}]}}'\n\
sleep 2\n\
printf '%s\\n' '{\"type\":\"user\",\"message\":{\"content\":[{\"type\":\"tool_result\",\"tool_use_id\":\"tool1\",\"content\":\"ok\"}]}}'\n\
sleep 2\n";
std::fs::write(&script, body).unwrap();
std::fs::set_permissions(&script, std::fs::Permissions::from_mode(0o755)).unwrap();
let (tx, _rx) = broadcast::channel::<AgentEvent>(64);
let (watcher_tx, _watcher_rx) = broadcast::channel::<WatcherEvent>(16);
let event_log = Arc::new(Mutex::new(Vec::new()));
let result = run_agent_pty_streaming(
"1196_story_tool_call_in_flight",
"coder-1",
"sh",
&[script.to_string_lossy().to_string()],
"--",
"/tmp",
&tx,
&event_log,
None,
1, // inactivity_timeout_secs = 1s
watcher_tx,
None,
None,
)
.await;
match result {
Err(err) => assert!(
err.contains("inactivity timeout"),
"expected an inactivity timeout error after the tool call resolved, got: {err}"
),
Ok(_) => panic!(
"agent must still be killed once the tool call resolves and \
the process falls genuinely silent again"
),
}
}
/// A genuinely hung agent (no output at all, no tool call in flight)
/// must still be killed by the inactivity watchdog after the configured
/// timeout — the suspension added for in-flight tool calls must not
/// mask a real hang.
#[tokio::test]
async fn genuinely_silent_agent_with_no_in_flight_call_is_killed() {
use std::os::unix::fs::PermissionsExt;
let tmp = tempfile::tempdir().unwrap();
let script = tmp.path().join("silent.sh");
std::fs::write(&script, "#!/bin/sh\nsleep 2\n").unwrap();
std::fs::set_permissions(&script, std::fs::Permissions::from_mode(0o755)).unwrap();
let (tx, _rx) = broadcast::channel::<AgentEvent>(64);
let (watcher_tx, _watcher_rx) = broadcast::channel::<WatcherEvent>(16);
let event_log = Arc::new(Mutex::new(Vec::new()));
let result = run_agent_pty_streaming(
"1196_story_genuine_hang",
"coder-1",
"sh",
&[script.to_string_lossy().to_string()],
"--",
"/tmp",
&tx,
&event_log,
None,
1, // inactivity_timeout_secs = 1s
watcher_tx,
None,
None,
)
.await;
match result {
Err(err) => assert!(err.contains("inactivity timeout")),
Ok(_) => panic!("a genuinely silent agent with no in-flight tool call must be killed"),
}
}
#[test]
fn test_emit_event_writes_to_log_writer() {
let tmp = tempfile::tempdir().unwrap();
@@ -503,4 +601,126 @@ mod tests {
"Expected RateLimitWarning for status=allowed, got: {evt:?}"
);
}
// ── story 1211: deterministic crash notification on mid-turn death ──────
/// AC1/AC3/AC4/AC5: a child killed mid-turn (inactivity watchdog, never
/// emits a `"result"` event) produces exactly one `AgentCrashed` watcher
/// event, carrying the exit code and the last line the child printed
/// before it died. The watchdog-kill path used to `return Err(...)`
/// directly from inside the receive loop, bypassing the bottom
/// cleanup/notification code entirely — this test guards against that
/// regression.
#[tokio::test]
async fn killed_mid_turn_sends_exactly_one_crash_notification() {
use std::os::unix::fs::PermissionsExt;
let tmp = tempfile::tempdir().unwrap();
let script = tmp.path().join("crash_then_hang.sh");
std::fs::write(
&script,
"#!/bin/sh\nprintf '%s\\n' 'assertion failed: output.write(&bytes).is_ok()'\nsleep 5\n",
)
.unwrap();
std::fs::set_permissions(&script, std::fs::Permissions::from_mode(0o755)).unwrap();
let (tx, _rx) = broadcast::channel::<AgentEvent>(64);
let (watcher_tx, mut watcher_rx) = broadcast::channel::<WatcherEvent>(16);
let event_log = Arc::new(Mutex::new(Vec::new()));
let result = run_agent_pty_streaming(
"1211_story_crash",
"coder-1",
"sh",
&[script.to_string_lossy().to_string()],
"--",
"/tmp",
&tx,
&event_log,
None,
1, // inactivity_timeout_secs = 1s
watcher_tx,
None,
None,
)
.await;
assert!(
result.is_err(),
"a watchdog-killed agent must still return the inactivity timeout error"
);
let evt = watcher_rx
.try_recv()
.expect("Expected exactly one AgentCrashed watcher event");
match evt {
WatcherEvent::AgentCrashed {
story_id,
agent_name,
exit_code,
last_error_line,
} => {
assert_eq!(story_id, "1211_story_crash");
assert_eq!(agent_name, "coder-1");
assert!(exit_code.is_some(), "exit code should be captured");
assert_eq!(
last_error_line.as_deref(),
Some("assertion failed: output.write(&bytes).is_ok()"),
"last line before death should be captured"
);
}
other => panic!("Expected AgentCrashed, got: {other:?}"),
}
// AC5: exactly one crash notification — no second event queued.
assert!(
watcher_rx.try_recv().is_err(),
"must not emit a duplicate AgentCrashed for the same death event"
);
}
/// AC2: a turn that ends cleanly (a `"result"` event observed) followed
/// by a normal child exit must produce zero `AgentCrashed` notifications.
#[tokio::test]
async fn clean_turn_end_sends_zero_crash_notifications() {
use std::os::unix::fs::PermissionsExt;
let tmp = tempfile::tempdir().unwrap();
let script = tmp.path().join("clean_result.sh");
std::fs::write(
&script,
"#!/bin/sh\nprintf '%s\\n' '{\"type\":\"result\",\"subtype\":\"success\"}'\n",
)
.unwrap();
std::fs::set_permissions(&script, std::fs::Permissions::from_mode(0o755)).unwrap();
let (tx, _rx) = broadcast::channel::<AgentEvent>(64);
let (watcher_tx, mut watcher_rx) = broadcast::channel::<WatcherEvent>(16);
let event_log = Arc::new(Mutex::new(Vec::new()));
let result = run_agent_pty_streaming(
"1211_story_clean",
"coder-1",
"sh",
&[script.to_string_lossy().to_string()],
"--",
"/tmp",
&tx,
&event_log,
None,
0, // no inactivity timeout
watcher_tx,
None,
None,
)
.await;
assert!(result.is_ok(), "PTY run should succeed: {:?}", result.err());
assert!(result.unwrap().exit_ok, "clean exit should report exit_ok");
assert!(
watcher_rx.try_recv().is_err(),
"a clean turn end followed by normal exit must not send AgentCrashed"
);
}
}
+92 -12
View File
@@ -246,14 +246,41 @@ fn run_agent_pty_blocking(
// can distinguish a rate-limit exit from a genuine no-progress exit (bug 1053).
let mut rate_limit_hard_block_seen = false;
let mut rate_limit_reset_at_captured: Option<chrono::DateTime<chrono::Utc>> = None;
// Tool-use ids from `assistant` messages that haven't yet seen a matching
// `tool_result` in a `user` message. While an MCP tool call (e.g.
// run_tests) is in flight, the server can legitimately run for many
// minutes with no PTY output at all — the CLI is blocked waiting on the
// MCP response, not hung. The inactivity deadline is suspended entirely
// while this set is non-empty, and resumes as soon as the matching
// tool_result clears the last in-flight id (story 1196).
let mut tool_calls_in_flight: std::collections::HashSet<String> =
std::collections::HashSet::new();
// Tracks whether a `"result"` event was observed — the CLI's signal that
// the current turn completed normally (see
// llm/providers/claude_code/events/mod.rs). If the child dies before
// this is set, the death is a mid-turn crash (AC1/AC2 of story 1211)
// rather than a clean completion.
let mut saw_result_event = false;
// Most recent non-empty line emitted by the child before it died — used
// as the "last error line" in the crash notification (AC4) since panics
// and assertion failures print to the PTY's combined stdout/stderr.
let mut last_line: Option<String> = None;
// Set when the inactivity watchdog kills the child (AC3 of story 1211):
// tracked instead of returning early so the death still flows through
// the single crash-detection funnel below before the function returns.
let mut timed_out = false;
loop {
let effective_timeout = base_timeout.map(|base| {
let extra = block_until
.and_then(|t| (t - chrono::Utc::now()).to_std().ok())
.unwrap_or(std::time::Duration::ZERO);
base + extra
});
let effective_timeout = if !tool_calls_in_flight.is_empty() {
None
} else {
base_timeout.map(|base| {
let extra = block_until
.and_then(|t| (t - chrono::Utc::now()).to_std().ok())
.unwrap_or(std::time::Duration::ZERO);
base + extra
})
};
let recv_result = match effective_timeout {
Some(dur) => line_rx.recv_timeout(dur),
@@ -278,10 +305,8 @@ fn run_agent_pty_blocking(
{inactivity_timeout_secs}s with no output. Killing process."
);
let _ = child.kill();
let _ = child.wait();
return Err(format!(
"Agent inactivity timeout: no output received for {inactivity_timeout_secs}s"
));
timed_out = true;
break;
}
};
@@ -289,6 +314,7 @@ fn run_agent_pty_blocking(
if trimmed.is_empty() {
continue;
}
last_line = Some(trimmed.to_string());
// Try to parse as JSON
let json: serde_json::Value = match serde_json::from_str(trimmed) {
@@ -339,8 +365,31 @@ fn run_agent_pty_blocking(
}
// Complete assistant events are skipped for content extraction
// because thinking and text already arrived via stream_event.
// The raw JSON is still forwarded as AgentJson below.
"assistant" | "user" => {}
// The raw JSON is still forwarded as AgentJson below. A tool_use
// block marks an MCP call as in flight so the inactivity deadline
// is suspended until its tool_result arrives (story 1196).
"assistant" => {
if let Some(blocks) = json.pointer("/message/content").and_then(|c| c.as_array()) {
for block in blocks {
if block.get("type").and_then(|t| t.as_str()) == Some("tool_use")
&& let Some(id) = block.get("id").and_then(|i| i.as_str())
{
tool_calls_in_flight.insert(id.to_string());
}
}
}
}
"user" => {
if let Some(blocks) = json.pointer("/message/content").and_then(|c| c.as_array()) {
for block in blocks {
if block.get("type").and_then(|t| t.as_str()) == Some("tool_result")
&& let Some(id) = block.get("tool_use_id").and_then(|i| i.as_str())
{
tool_calls_in_flight.remove(id);
}
}
}
}
"rate_limit_event" => {
let rate_limit_info = json.get("rate_limit_info");
let status = rate_limit_info
@@ -404,6 +453,10 @@ fn run_agent_pty_blocking(
}
}
"result" => {
// A "result" event signals the CLI turn completed — mark the
// turn as clean so the post-loop crash check (AC1/AC2 of
// story 1211) does not treat this exit as a mid-turn death.
saw_result_event = true;
// Extract token usage from the result event.
if let Some(usage) = TokenUsage::from_result_event(&json) {
slog!(
@@ -445,6 +498,7 @@ fn run_agent_pty_blocking(
false
}
};
let exit_code = wait_result.as_ref().ok().map(|status| status.exit_code());
// Wait for the reader thread to finish so it releases the cloned PTY
// master fd before we return. Without this, the next PTY spawn for the
@@ -453,6 +507,32 @@ fn run_agent_pty_blocking(
slog!("[agent:{story_id}:{agent_name}] Reader thread panicked: {e:?}");
}
// AC1-AC3 (story 1211): single crash-detection funnel. Every exit path
// above (EOF, reader disconnect, IO error, or watchdog kill on timeout)
// breaks out of the loop into this one spot instead of returning early,
// so a mid-turn death is detected and notified exactly once regardless
// of which path triggered it (AC5). A turn is "clean" once a `"result"`
// event has been observed; anything else was killed or crashed before
// finishing its turn.
if !saw_result_event {
slog_warn!(
"[agent:{story_id}:{agent_name}] Agent died mid-turn (no result event observed); \
exit_code={exit_code:?}, last_line={last_line:?}"
);
let _ = watcher_tx.send(WatcherEvent::AgentCrashed {
story_id: story_id.to_string(),
agent_name: agent_name.to_string(),
exit_code,
last_error_line: last_line.clone(),
});
}
if timed_out {
return Err(format!(
"Agent inactivity timeout: no output received for {inactivity_timeout_secs}s"
));
}
// Log whether session was created — Session: None indicates CLI died
// before emitting any events (possible causes: rate limit, budget
// exhaustion, PTY write failure, CLI crash).
+404
View File
@@ -0,0 +1,404 @@
//! Shared helpers for the API-based agent runtimes (Gemini, OpenAI), which
//! talk directly to a provider's REST API rather than spawning a CLI over a
//! PTY. Both runtimes drive an almost-identical turn loop against different
//! wire formats; this module holds the logic that doesn't vary between them.
use std::sync::atomic::{AtomicBool, Ordering};
use std::sync::{Arc, Mutex};
use serde_json::Value;
use tokio::sync::broadcast;
use crate::agent_log::AgentLogWriter;
use super::super::{AgentEvent, TokenUsage};
use super::{RuntimeContext, RuntimeResult, RuntimeStatus};
/// Cancellation flag shared by the API-based runtimes' `stop()`/`get_status()`.
pub(super) struct CancellationFlag {
cancelled: Arc<AtomicBool>,
}
impl CancellationFlag {
/// Create a fresh, un-cancelled flag.
pub(super) fn new() -> Self {
Self {
cancelled: Arc::new(AtomicBool::new(false)),
}
}
/// Clone of the underlying flag, for the conversation loop to poll.
pub(super) fn handle(&self) -> Arc<AtomicBool> {
Arc::clone(&self.cancelled)
}
/// Request a stop.
pub(super) fn stop(&self) {
self.cancelled.store(true, Ordering::Relaxed);
}
/// Report `Failed` once stopped, `Idle` otherwise.
pub(super) fn status(&self) -> RuntimeStatus {
if self.cancelled.load(Ordering::Relaxed) {
RuntimeStatus::Failed
} else {
RuntimeStatus::Idle
}
}
}
/// Build the event-emitting closure shared by API-based runtimes: forwards
/// events to the broadcast channel, the in-memory event log, and (optionally)
/// the on-disk log writer.
pub(super) fn make_emit(
tx: broadcast::Sender<AgentEvent>,
event_log: Arc<Mutex<Vec<AgentEvent>>>,
log_writer: Option<Arc<Mutex<AgentLogWriter>>>,
) -> impl Fn(AgentEvent) {
move |event: AgentEvent| {
super::super::pty::emit_event(
event,
&tx,
&event_log,
log_writer.as_ref().map(|w| w.as_ref()),
);
}
}
/// Zeroed token-usage accumulator, seeded before an API runtime's
/// conversation loop starts accumulating per-turn usage.
pub(super) fn zero_usage() -> TokenUsage {
TokenUsage {
input_tokens: 0,
output_tokens: 0,
cache_creation_input_tokens: 0,
cache_read_input_tokens: 0,
total_cost_usd: 0.0,
}
}
/// Emit the initial "running" status event.
pub(super) fn emit_running(ctx: &RuntimeContext, emit: &impl Fn(AgentEvent)) {
emit(AgentEvent::Status {
story_id: ctx.story_id.clone(),
agent_name: ctx.agent_name.clone(),
status: "running".to_string(),
});
}
/// Set up an API-based runtime's conversation loop: builds the event
/// emitter, emits the initial "running" status, and returns it alongside a
/// zeroed usage accumulator and the turn counter (starting at 0).
pub(super) fn start_conversation_loop(
ctx: &RuntimeContext,
tx: broadcast::Sender<AgentEvent>,
event_log: Arc<Mutex<Vec<AgentEvent>>>,
log_writer: Option<Arc<Mutex<AgentLogWriter>>>,
) -> (impl Fn(AgentEvent), TokenUsage, u32) {
let emit = make_emit(tx, event_log, log_writer);
emit_running(ctx, &emit);
(emit, zero_usage(), 0u32)
}
/// Build a successful `RuntimeResult` carrying the given token usage. All
/// API-based runtimes report `exit_ok: true` and leave the CLI-only fields
/// (`aborted_signal`, `rate_limit_exit`, `rate_limit_reset_at`) at their
/// defaults, since those concepts don't apply outside the PTY runtime.
pub(super) fn api_runtime_result(total_usage: TokenUsage) -> RuntimeResult {
RuntimeResult {
session_id: None,
token_usage: Some(total_usage),
exit_ok: true,
aborted_signal: false,
rate_limit_exit: false,
rate_limit_reset_at: None,
}
}
/// Emit the `Done` event and build the final successful result once the
/// model produces a response with no further tool/function calls.
pub(super) fn done_result(
ctx: &RuntimeContext,
emit: &impl Fn(AgentEvent),
total_usage: TokenUsage,
) -> RuntimeResult {
emit(AgentEvent::Done {
story_id: ctx.story_id.clone(),
agent_name: ctx.agent_name.clone(),
session_id: None,
});
api_runtime_result(total_usage)
}
/// Safety limit on conversation turns for API-based runtimes, shared so
/// both the guard check and its error message stay in sync.
const MAX_TURNS: u32 = 200;
/// Check the per-turn cancellation/max-turns guard at the top of an API
/// runtime's conversation loop. Returns `Some(result)` when the loop should
/// stop immediately (either the user requested a stop, or the safety turn
/// limit was exceeded); otherwise increments `*turn` and returns `None`.
pub(super) fn check_loop_guard(
ctx: &RuntimeContext,
cancelled: &AtomicBool,
turn: &mut u32,
total_usage: &TokenUsage,
emit: &impl Fn(AgentEvent),
) -> Option<RuntimeResult> {
if cancelled.load(Ordering::Relaxed) {
emit(AgentEvent::Error {
story_id: ctx.story_id.clone(),
agent_name: ctx.agent_name.clone(),
message: "Agent was stopped by user".to_string(),
});
return Some(api_runtime_result(total_usage.clone()));
}
*turn += 1;
if *turn > MAX_TURNS {
emit(AgentEvent::Error {
story_id: ctx.story_id.clone(),
agent_name: ctx.agent_name.clone(),
message: format!("Exceeded maximum turns ({MAX_TURNS})"),
});
return Some(api_runtime_result(total_usage.clone()));
}
None
}
/// Extract the model name for an API-based runtime: the agent pool stashes
/// the model directly in `ctx.command` for non-CLI runtimes, detected here
/// via `is_command_a_model`; otherwise fall back to a `--model` arg, and
/// finally `default_model`.
pub(super) fn extract_model(
ctx: &RuntimeContext,
is_command_a_model: impl Fn(&str) -> bool,
default_model: &str,
) -> String {
if is_command_a_model(&ctx.command) {
ctx.command.clone()
} else {
ctx.args
.iter()
.position(|a| a == "--model")
.and_then(|i| ctx.args.get(i + 1))
.cloned()
.unwrap_or_else(|| default_model.to_string())
}
}
/// Build the default system-prompt text shared by both API-based runtimes:
/// prefers an explicit `--append-system-prompt` arg (set by the agent pool),
/// else falls back to a generic tool-calling preamble.
pub(super) fn build_system_text(ctx: &RuntimeContext) -> String {
ctx.args
.iter()
.position(|a| a == "--append-system-prompt")
.and_then(|i| ctx.args.get(i + 1))
.cloned()
.unwrap_or_else(|| {
format!(
"You are an AI coding agent working on story {}. \
You have access to tools via function calling. \
Use them to complete the task. \
Work in the directory: {}",
ctx.story_id, ctx.cwd
)
})
}
/// Recursively clean an MCP JSON-Schema `properties` object into a provider
/// function-calling schema: strips `$schema` (always) and
/// `additionalProperties` (when the provider doesn't support it, e.g.
/// Gemini) from the top level and from nested `properties`/`items`.
pub(super) fn clean_schema_properties(
properties: &Value,
strip_additional_properties: bool,
) -> Value {
let Some(obj) = properties.as_object() else {
return properties.clone();
};
let mut cleaned = serde_json::Map::new();
for (key, value) in obj {
let mut prop = value.clone();
if let Some(p) = prop.as_object_mut() {
p.remove("$schema");
if strip_additional_properties {
p.remove("additionalProperties");
}
if let Some(nested_props) = p.get("properties").cloned() {
p.insert(
"properties".to_string(),
clean_schema_properties(&nested_props, strip_additional_properties),
);
}
if let Some(items) = p.get("items").cloned()
&& let Some(items_obj) = items.as_object()
{
let mut cleaned_items = items_obj.clone();
cleaned_items.remove("$schema");
if strip_additional_properties {
cleaned_items.remove("additionalProperties");
}
p.insert("items".to_string(), Value::Object(cleaned_items));
}
}
cleaned.insert(key.clone(), prop);
}
Value::Object(cleaned)
}
// ── Test helpers ─────────────────────────────────────────────────────
/// Build a throwaway `AppContext` backed by a temp directory, for tests
/// that need a `RuntimeContext.app_ctx` but don't exercise it.
#[cfg(test)]
pub(super) fn test_app_ctx() -> Arc<crate::http::context::AppContext> {
let tmp = tempfile::tempdir().unwrap();
Arc::new(crate::http::context::AppContext::new_test(
tmp.path().to_path_buf(),
))
}
/// Build a `RuntimeContext` with sensible test defaults, overriding only
/// `command` and `args` (the fields the API-runtime tests vary).
#[cfg(test)]
pub(super) fn test_runtime_context(command: &str, args: Vec<String>) -> RuntimeContext {
RuntimeContext {
story_id: "42_story_test".to_string(),
agent_name: "coder-1".to_string(),
command: command.to_string(),
args,
prompt: "Do the thing".to_string(),
cwd: "/tmp/wt".to_string(),
inactivity_timeout_secs: 300,
app_ctx: Some(test_app_ctx()),
session_id_to_resume: None,
fresh_prompt: None,
project_root: std::path::PathBuf::from("/tmp/project"),
model: None,
}
}
// ── Tests ────────────────────────────────────────────────────────────
#[cfg(test)]
mod tests {
use super::*;
use serde_json::json;
#[test]
fn clean_schema_strips_dollar_schema_always() {
let schema = json!({
"name": { "type": "string", "$schema": "http://json-schema.org/draft-07/schema#" }
});
let result = clean_schema_properties(&schema, false);
assert!(result["name"].get("$schema").is_none());
}
#[test]
fn clean_schema_strips_additional_properties_when_requested() {
let schema = json!({
"name": { "type": "string", "additionalProperties": false }
});
let result = clean_schema_properties(&schema, true);
assert!(result["name"].get("additionalProperties").is_none());
}
#[test]
fn clean_schema_keeps_additional_properties_when_not_requested() {
let schema = json!({
"name": { "type": "object", "additionalProperties": false }
});
let result = clean_schema_properties(&schema, false);
assert!(result["name"].get("additionalProperties").is_some());
}
#[test]
fn clean_schema_recurses_into_nested_object_properties() {
let schema = json!({
"config": {
"type": "object",
"properties": {
"key": { "type": "string", "$schema": "x" }
}
}
});
let result = clean_schema_properties(&schema, false);
assert!(result["config"]["properties"]["key"].is_object());
assert!(
result["config"]["properties"]["key"]
.get("$schema")
.is_none()
);
}
#[test]
fn clean_schema_recurses_into_array_items() {
let schema = json!({
"items": {
"type": "array",
"items": {
"type": "object",
"properties": { "name": { "type": "string" } },
"additionalProperties": false,
"$schema": "x"
}
}
});
let result = clean_schema_properties(&schema, true);
let items_schema = &result["items"]["items"];
assert!(items_schema.get("additionalProperties").is_none());
assert!(items_schema.get("$schema").is_none());
}
#[test]
fn extract_model_uses_command_when_it_matches() {
let ctx = test_runtime_context("gpt-4o", vec![]);
assert_eq!(
extract_model(&ctx, |c| c.starts_with("gpt"), "fallback"),
"gpt-4o"
);
}
#[test]
fn extract_model_falls_back_to_args() {
let ctx = test_runtime_context("claude", vec!["--model".to_string(), "custom".to_string()]);
assert_eq!(
extract_model(&ctx, |c| c.starts_with("gpt"), "fallback"),
"custom"
);
}
#[test]
fn extract_model_falls_back_to_default() {
let ctx = test_runtime_context("claude", vec![]);
assert_eq!(
extract_model(&ctx, |c| c.starts_with("gpt"), "fallback"),
"fallback"
);
}
#[test]
fn build_system_text_uses_args() {
let ctx = test_runtime_context(
"gpt-4o",
vec![
"--append-system-prompt".to_string(),
"Custom system prompt".to_string(),
],
);
assert_eq!(build_system_text(&ctx), "Custom system prompt");
}
#[test]
fn build_system_text_default() {
let ctx = test_runtime_context("gpt-4o", vec![]);
let text = build_system_text(&ctx);
assert!(text.contains("42_story_test"));
assert!(text.contains("/tmp/wt"));
}
}
+8 -54
View File
@@ -4,6 +4,7 @@ use serde_json::{Value, json};
use super::super::super::TokenUsage;
use super::super::RuntimeContext;
use super::super::api_common::build_system_text;
// ── Gemini API types ─────────────────────────────────────────────────
@@ -19,26 +20,8 @@ pub(super) struct GeminiFunctionDeclaration {
/// Build the system instruction content from the RuntimeContext.
pub(super) fn build_system_instruction(ctx: &RuntimeContext) -> Value {
// Use system_prompt from args if provided via --append-system-prompt,
// otherwise use a sensible default.
let system_text = ctx
.args
.iter()
.position(|a| a == "--append-system-prompt")
.and_then(|i| ctx.args.get(i + 1))
.cloned()
.unwrap_or_else(|| {
format!(
"You are an AI coding agent working on story {}. \
You have access to tools via function calling. \
Use them to complete the task. \
Work in the directory: {}",
ctx.story_id, ctx.cwd
)
});
json!({
"parts": [{ "text": system_text }]
"parts": [{ "text": build_system_text(ctx) }]
})
}
@@ -92,34 +75,18 @@ pub(super) fn parse_usage_metadata(response: &Value) -> Option<TokenUsage> {
#[cfg(test)]
mod tests {
use super::super::super::api_common::test_runtime_context;
use super::*;
use crate::http::context::AppContext;
use std::sync::Arc;
fn test_app_ctx() -> Arc<AppContext> {
let tmp = tempfile::tempdir().unwrap();
Arc::new(AppContext::new_test(tmp.path().to_path_buf()))
}
#[test]
fn build_system_instruction_uses_args() {
let ctx = RuntimeContext {
story_id: "42_story_test".to_string(),
agent_name: "coder-1".to_string(),
command: "gemini-2.5-pro".to_string(),
args: vec![
let ctx = test_runtime_context(
"gemini-2.5-pro",
vec![
"--append-system-prompt".to_string(),
"Custom system prompt".to_string(),
],
prompt: "Do the thing".to_string(),
cwd: "/tmp/wt".to_string(),
inactivity_timeout_secs: 300,
app_ctx: Some(test_app_ctx()),
session_id_to_resume: None,
fresh_prompt: None,
project_root: std::path::PathBuf::from("/tmp/project"),
model: None,
};
);
let instruction = build_system_instruction(&ctx);
assert_eq!(instruction["parts"][0]["text"], "Custom system prompt");
@@ -127,20 +94,7 @@ mod tests {
#[test]
fn build_system_instruction_default() {
let ctx = RuntimeContext {
story_id: "42_story_test".to_string(),
agent_name: "coder-1".to_string(),
command: "gemini-2.5-pro".to_string(),
args: vec![],
prompt: "Do the thing".to_string(),
cwd: "/tmp/wt".to_string(),
inactivity_timeout_secs: 300,
app_ctx: Some(test_app_ctx()),
session_id_to_resume: None,
fresh_prompt: None,
project_root: std::path::PathBuf::from("/tmp/project"),
model: None,
};
let ctx = test_runtime_context("gemini-2.5-pro", vec![]);
let instruction = build_system_instruction(&ctx);
let text = instruction["parts"][0]["text"].as_str().unwrap();
+2 -80
View File
@@ -8,6 +8,7 @@ use crate::slog;
use crate::http::mcp::tools_list::list_tools;
use super::super::api_common::clean_schema_properties;
use super::api::GeminiFunctionDeclaration;
// ── MCP tool loading ────────────────────────────────────────────────
@@ -62,7 +63,7 @@ pub(super) fn convert_mcp_schema_to_gemini(schema: Option<&Value>) -> Option<Val
let mut result = json!({
"type": "object",
"properties": clean_schema_properties(properties),
"properties": clean_schema_properties(properties, true),
});
// Preserve required fields if present.
@@ -73,44 +74,6 @@ pub(super) fn convert_mcp_schema_to_gemini(schema: Option<&Value>) -> Option<Val
Some(result)
}
/// Recursively clean schema properties to be Gemini-compatible.
/// Removes unsupported JSON Schema keywords.
fn clean_schema_properties(properties: &Value) -> Value {
let Some(obj) = properties.as_object() else {
return properties.clone();
};
let mut cleaned = serde_json::Map::new();
for (key, value) in obj {
let mut prop = value.clone();
// Remove JSON Schema keywords not supported by Gemini
if let Some(p) = prop.as_object_mut() {
p.remove("$schema");
p.remove("additionalProperties");
// Recursively clean nested object properties
if let Some(nested_props) = p.get("properties").cloned() {
p.insert(
"properties".to_string(),
clean_schema_properties(&nested_props),
);
}
// Clean items schema for arrays
if let Some(items) = p.get("items").cloned()
&& let Some(items_obj) = items.as_object()
{
let mut cleaned_items = items_obj.clone();
cleaned_items.remove("$schema");
cleaned_items.remove("additionalProperties");
p.insert("items".to_string(), Value::Object(cleaned_items));
}
}
cleaned.insert(key.clone(), prop);
}
Value::Object(cleaned)
}
// ── Tests ────────────────────────────────────────────────────────────
#[cfg(test)]
@@ -170,45 +133,4 @@ mod tests {
assert!(name_prop.get("$schema").is_none());
assert_eq!(name_prop["type"], "string");
}
#[test]
fn convert_mcp_schema_with_nested_objects() {
let schema = json!({
"type": "object",
"properties": {
"config": {
"type": "object",
"properties": {
"key": { "type": "string" }
}
}
}
});
let result = convert_mcp_schema_to_gemini(Some(&schema)).unwrap();
assert!(result["properties"]["config"]["properties"]["key"].is_object());
}
#[test]
fn convert_mcp_schema_with_array_items() {
let schema = json!({
"type": "object",
"properties": {
"items": {
"type": "array",
"items": {
"type": "object",
"properties": {
"name": { "type": "string" }
},
"additionalProperties": false
}
}
}
});
let result = convert_mcp_schema_to_gemini(Some(&schema)).unwrap();
let items_schema = &result["properties"]["items"]["items"];
assert!(items_schema.get("additionalProperties").is_none());
}
}
+20 -127
View File
@@ -1,5 +1,5 @@
//! Gemini runtime — drives Google Gemini API sessions as agent backends.
use std::sync::atomic::{AtomicBool, Ordering};
use std::sync::atomic::Ordering;
use std::sync::{Arc, Mutex};
use reqwest::Client;
@@ -10,7 +10,10 @@ use crate::agent_log::AgentLogWriter;
use crate::http::mcp::dispatch::dispatch_tool_call;
use crate::slog;
use super::super::{AgentEvent, TokenUsage};
use super::super::AgentEvent;
use super::api_common::{
CancellationFlag, check_loop_guard, done_result, extract_model, start_conversation_loop,
};
use super::{AgentRuntime, RuntimeContext, RuntimeResult, RuntimeStatus};
mod api;
@@ -40,14 +43,14 @@ struct GeminiFunctionCall {
/// 6. Tracks token usage from the API response metadata.
pub struct GeminiRuntime {
/// Whether a stop has been requested.
cancelled: Arc<AtomicBool>,
cancelled: CancellationFlag,
}
impl GeminiRuntime {
/// Create a new Gemini runtime instance.
pub fn new() -> Self {
Self {
cancelled: Arc::new(AtomicBool::new(false)),
cancelled: CancellationFlag::new(),
}
}
}
@@ -66,19 +69,7 @@ impl AgentRuntime for GeminiRuntime {
.to_string()
})?;
let model = if ctx.command.starts_with("gemini") {
// The pool puts the model into `command` for non-CLI runtimes,
// but also check args for a --model flag.
ctx.command.clone()
} else {
// Fall back to args: look for --model <value>
ctx.args
.iter()
.position(|a| a == "--model")
.and_then(|i| ctx.args.get(i + 1))
.cloned()
.unwrap_or_else(|| "gemini-2.5-pro".to_string())
};
let model = extract_model(&ctx, |c| c.starts_with("gemini"), "gemini-2.5-pro");
let app_ctx = ctx
.app_ctx
@@ -86,7 +77,7 @@ impl AgentRuntime for GeminiRuntime {
.ok_or_else(|| "Gemini runtime requires app_ctx to be set".to_string())?;
let client = Client::new();
let cancelled = Arc::clone(&self.cancelled);
let cancelled = self.cancelled.handle();
// Step 1: Load MCP tool definitions and convert to Gemini format.
let gemini_tools = convert_mcp_tools_to_gemini();
@@ -98,65 +89,14 @@ impl AgentRuntime for GeminiRuntime {
"parts": [{ "text": ctx.prompt }]
})];
let mut total_usage = TokenUsage {
input_tokens: 0,
output_tokens: 0,
cache_creation_input_tokens: 0,
cache_read_input_tokens: 0,
total_cost_usd: 0.0,
};
let emit = |event: AgentEvent| {
super::super::pty::emit_event(
event,
&tx,
&event_log,
log_writer.as_ref().map(|w| w.as_ref()),
);
};
emit(AgentEvent::Status {
story_id: ctx.story_id.clone(),
agent_name: ctx.agent_name.clone(),
status: "running".to_string(),
});
let (emit, mut total_usage, mut turn) =
start_conversation_loop(&ctx, tx, event_log, log_writer);
// Step 3: Conversation loop.
let mut turn = 0u32;
let max_turns = 200; // Safety limit
loop {
if cancelled.load(Ordering::Relaxed) {
emit(AgentEvent::Error {
story_id: ctx.story_id.clone(),
agent_name: ctx.agent_name.clone(),
message: "Agent was stopped by user".to_string(),
});
return Ok(RuntimeResult {
session_id: None,
token_usage: Some(total_usage),
exit_ok: true,
aborted_signal: false,
rate_limit_exit: false,
rate_limit_reset_at: None,
});
}
turn += 1;
if turn > max_turns {
emit(AgentEvent::Error {
story_id: ctx.story_id.clone(),
agent_name: ctx.agent_name.clone(),
message: format!("Exceeded maximum turns ({max_turns})"),
});
return Ok(RuntimeResult {
session_id: None,
token_usage: Some(total_usage),
exit_ok: true,
aborted_signal: false,
rate_limit_exit: false,
rate_limit_reset_at: None,
});
if let Some(result) = check_loop_guard(&ctx, &cancelled, &mut turn, &total_usage, &emit)
{
return Ok(result);
}
slog!(
@@ -248,19 +188,7 @@ impl AgentRuntime for GeminiRuntime {
// If no function calls, the model is done.
if function_calls.is_empty() {
emit(AgentEvent::Done {
story_id: ctx.story_id.clone(),
agent_name: ctx.agent_name.clone(),
session_id: None,
});
return Ok(RuntimeResult {
session_id: None,
token_usage: Some(total_usage),
exit_ok: true,
aborted_signal: false,
rate_limit_exit: false,
rate_limit_reset_at: None,
});
return Ok(done_result(&ctx, &emit, total_usage));
}
// Add the model's response to the conversation.
@@ -333,32 +261,15 @@ impl AgentRuntime for GeminiRuntime {
}
}
emit(AgentEvent::Done {
story_id: ctx.story_id.clone(),
agent_name: ctx.agent_name.clone(),
session_id: None,
});
Ok(RuntimeResult {
session_id: None,
token_usage: Some(total_usage),
exit_ok: true,
aborted_signal: false,
rate_limit_exit: false,
rate_limit_reset_at: None,
})
Ok(done_result(&ctx, &emit, total_usage))
}
fn stop(&self) {
self.cancelled.store(true, Ordering::Relaxed);
self.cancelled.stop();
}
fn get_status(&self) -> RuntimeStatus {
if self.cancelled.load(Ordering::Relaxed) {
RuntimeStatus::Failed
} else {
RuntimeStatus::Idle
}
self.cancelled.status()
}
}
@@ -366,13 +277,8 @@ impl AgentRuntime for GeminiRuntime {
#[cfg(test)]
mod tests {
use super::super::api_common::test_runtime_context;
use super::*;
use crate::http::context::AppContext;
fn test_app_ctx() -> Arc<AppContext> {
let tmp = tempfile::tempdir().unwrap();
Arc::new(AppContext::new_test(tmp.path().to_path_buf()))
}
#[test]
fn gemini_runtime_stop_sets_cancelled() {
@@ -385,20 +291,7 @@ mod tests {
#[test]
fn model_extraction_from_command() {
// When command starts with "gemini", use it as model name
let ctx = RuntimeContext {
story_id: "1".to_string(),
agent_name: "coder".to_string(),
command: "gemini-2.5-pro".to_string(),
args: vec![],
prompt: "test".to_string(),
cwd: "/tmp".to_string(),
inactivity_timeout_secs: 300,
app_ctx: Some(test_app_ctx()),
session_id_to_resume: None,
fresh_prompt: None,
project_root: std::path::PathBuf::from("/tmp/project"),
model: None,
};
let ctx = test_runtime_context("gemini-2.5-pro", vec![]);
// The model extraction logic is inside start(), but we test the
// condition here.

Some files were not shown because too many files have changed in this diff Show More