mixdog 0.9.94 → 0.9.96

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (259) hide show
  1. package/LICENSES/Apache-2.0.txt +201 -0
  2. package/LICENSES/MIT.txt +56 -0
  3. package/LICENSES/codex-NOTICE.txt +6 -0
  4. package/NOTICE.md +109 -0
  5. package/package.json +26 -13
  6. package/scripts/.tmp-coverage-report.mjs +96 -0
  7. package/scripts/build-tui.mjs +17 -1
  8. package/scripts/fixtures/patch-replay-corpus.json +98 -0
  9. package/scripts/lib/isolated-root-cleanup.mjs +19 -0
  10. package/scripts/run-suite.mjs +100 -0
  11. package/src/defaults/mixdog-config.template.json +5 -8
  12. package/src/lib/rules-builder.cjs +4 -3
  13. package/src/output-styles/detailed.md +13 -15
  14. package/src/output-styles/extreme-minimal.md +7 -11
  15. package/src/output-styles/minimal.md +4 -7
  16. package/src/output-styles/simple.md +11 -13
  17. package/src/rules/agent/30-explorer.md +7 -4
  18. package/src/rules/lead/01-general.md +1 -2
  19. package/src/rules/lead/lead-brief.md +4 -5
  20. package/src/rules/lead/lead-tool.md +3 -2
  21. package/src/rules/shared/01-tool.md +20 -35
  22. package/src/runtime/agent/orchestrator/agent-runtime/agent-dispatch.mjs +18 -2
  23. package/src/runtime/agent/orchestrator/agent-runtime/agent-progress-watchdog.mjs +31 -1
  24. package/src/runtime/agent/orchestrator/agent-runtime/cache-strategy.mjs +2 -2
  25. package/src/runtime/agent/orchestrator/agent-runtime/maintenance-route.mjs +6 -17
  26. package/src/runtime/agent/orchestrator/agent-trace-format.mjs +13 -2
  27. package/src/runtime/agent/orchestrator/agent-trace-io.mjs +4 -0
  28. package/src/runtime/agent/orchestrator/agent-trace.mjs +29 -0
  29. package/src/runtime/agent/orchestrator/config.mjs +246 -67
  30. package/src/runtime/agent/orchestrator/mcp/client.mjs +29 -14
  31. package/src/runtime/agent/orchestrator/mcp/reconnect-singleflight.mjs +23 -0
  32. package/src/runtime/agent/orchestrator/providers/admission-scheduler.mjs +113 -14
  33. package/src/runtime/agent/orchestrator/providers/anthropic-effort.mjs +97 -7
  34. package/src/runtime/agent/orchestrator/providers/anthropic-model-resolve.mjs +11 -4
  35. package/src/runtime/agent/orchestrator/providers/anthropic-oauth.mjs +2 -3
  36. package/src/runtime/agent/orchestrator/providers/anthropic-sse.mjs +5 -1
  37. package/src/runtime/agent/orchestrator/providers/anthropic.mjs +1 -1
  38. package/src/runtime/agent/orchestrator/providers/codex-client-meta.mjs +4 -6
  39. package/src/runtime/agent/orchestrator/providers/gemini-stream.mjs +32 -11
  40. package/src/runtime/agent/orchestrator/providers/gemini.mjs +89 -2
  41. package/src/runtime/agent/orchestrator/providers/grok-oauth.mjs +12 -1
  42. package/src/runtime/agent/orchestrator/providers/lib/stream-outcome.mjs +1 -1
  43. package/src/runtime/agent/orchestrator/providers/openai-codex-metadata.mjs +8 -8
  44. package/src/runtime/agent/orchestrator/providers/openai-compat-xai.mjs +19 -1
  45. package/src/runtime/agent/orchestrator/providers/openai-compat.mjs +144 -39
  46. package/src/runtime/agent/orchestrator/providers/openai-oauth-http-sse.mjs +29 -16
  47. package/src/runtime/agent/orchestrator/providers/openai-oauth-ws.mjs +2 -2
  48. package/src/runtime/agent/orchestrator/providers/openai-oauth.mjs +15 -1
  49. package/src/runtime/agent/orchestrator/providers/openai-responses-payload.mjs +14 -16
  50. package/src/runtime/agent/orchestrator/providers/openai-transport-policy.mjs +1 -1
  51. package/src/runtime/agent/orchestrator/providers/openai-ws-headers.mjs +5 -4
  52. package/src/runtime/agent/orchestrator/providers/openai-ws-pool.mjs +13 -14
  53. package/src/runtime/agent/orchestrator/providers/openai-ws-stream.mjs +2 -3
  54. package/src/runtime/agent/orchestrator/providers/registry.mjs +51 -4
  55. package/src/runtime/agent/orchestrator/providers/retry-classifier.mjs +82 -22
  56. package/src/runtime/agent/orchestrator/providers/stream-json-pool.mjs +291 -0
  57. package/src/runtime/agent/orchestrator/providers/stream-json-worker.mjs +21 -0
  58. package/src/runtime/agent/orchestrator/session/agent-loop.mjs +40 -36
  59. package/src/runtime/agent/orchestrator/session/cache/read-cache.mjs +7 -0
  60. package/src/runtime/agent/orchestrator/session/context-compaction-policy.mjs +10 -3
  61. package/src/runtime/agent/orchestrator/session/context-utils.mjs +10 -11
  62. package/src/runtime/agent/orchestrator/session/eager-dispatch.mjs +5 -33
  63. package/src/runtime/agent/orchestrator/session/loop/compact-policy.mjs +3 -3
  64. package/src/runtime/agent/orchestrator/session/loop/completion-guards.mjs +0 -14
  65. package/src/runtime/agent/orchestrator/session/loop/pre-dispatch-deny.mjs +13 -0
  66. package/src/runtime/agent/orchestrator/session/loop/stop-hooks.mjs +9 -8
  67. package/src/runtime/agent/orchestrator/session/loop/termination.mjs +2 -2
  68. package/src/runtime/agent/orchestrator/session/loop/tool-classify.mjs +0 -24
  69. package/src/runtime/agent/orchestrator/session/loop/tool-exec.mjs +1 -27
  70. package/src/runtime/agent/orchestrator/session/loop/tool-helpers.mjs +6 -10
  71. package/src/runtime/agent/orchestrator/session/manager/ask-session.mjs +12 -4
  72. package/src/runtime/agent/orchestrator/session/manager/context-meta.mjs +1 -1
  73. package/src/runtime/agent/orchestrator/session/manager/idle-cleanup.mjs +1 -1
  74. package/src/runtime/agent/orchestrator/session/manager/pending-messages.mjs +168 -83
  75. package/src/runtime/agent/orchestrator/session/manager/runtime-loaders.mjs +4 -0
  76. package/src/runtime/agent/orchestrator/session/manager/session-close.mjs +3 -0
  77. package/src/runtime/agent/orchestrator/session/manager/session-crud.mjs +31 -1
  78. package/src/runtime/agent/orchestrator/session/manager/session-lifecycle.mjs +8 -1
  79. package/src/runtime/agent/orchestrator/session/manager/turn-interruption.mjs +34 -18
  80. package/src/runtime/agent/orchestrator/session/manager.mjs +2 -0
  81. package/src/runtime/agent/orchestrator/session/result-classification.mjs +65 -0
  82. package/src/runtime/agent/orchestrator/session/send-with-recovery.mjs +89 -11
  83. package/src/runtime/agent/orchestrator/session/store-summary-reader.mjs +12 -0
  84. package/src/runtime/agent/orchestrator/session/tool-batch.mjs +6 -39
  85. package/src/runtime/agent/orchestrator/session/tool-result-offload.mjs +5 -6
  86. package/src/runtime/agent/orchestrator/stall-policy.mjs +3 -4
  87. package/src/runtime/agent/orchestrator/tools/bash-session.mjs +2 -2
  88. package/src/runtime/agent/orchestrator/tools/builtin/arg-guard.mjs +56 -0
  89. package/src/runtime/agent/orchestrator/tools/builtin/bash-tool.mjs +124 -19
  90. package/src/runtime/agent/orchestrator/tools/builtin/builtin-tools.mjs +8 -9
  91. package/src/runtime/agent/orchestrator/tools/builtin/cache-layers.mjs +2 -2
  92. package/src/runtime/agent/orchestrator/tools/builtin/device-paths.mjs +2 -2
  93. package/src/runtime/agent/orchestrator/tools/builtin/lib/grep-context-expander.mjs +7 -16
  94. package/src/runtime/agent/orchestrator/tools/builtin/list-tool.mjs +10 -26
  95. package/src/runtime/agent/orchestrator/tools/builtin/path-utils.mjs +14 -0
  96. package/src/runtime/agent/orchestrator/tools/builtin/read-constants.mjs +4 -4
  97. package/src/runtime/agent/orchestrator/tools/builtin/read-image-resize.mjs +2 -2
  98. package/src/runtime/agent/orchestrator/tools/builtin/read-image.mjs +1 -1
  99. package/src/runtime/agent/orchestrator/tools/builtin/read-single-tool.mjs +3 -3
  100. package/src/runtime/agent/orchestrator/tools/builtin/read-snapshot-runtime.mjs +2 -1
  101. package/src/runtime/agent/orchestrator/tools/builtin/read-special-files.mjs +3 -3
  102. package/src/runtime/agent/orchestrator/tools/builtin/read-tool.mjs +1 -23
  103. package/src/runtime/agent/orchestrator/tools/builtin/rg-runner.mjs +24 -10
  104. package/src/runtime/agent/orchestrator/tools/builtin/search-path-diagnostics.mjs +7 -0
  105. package/src/runtime/agent/orchestrator/tools/builtin/search-tool.mjs +20 -43
  106. package/src/runtime/agent/orchestrator/tools/builtin/shell-analysis.mjs +76 -3
  107. package/src/runtime/agent/orchestrator/tools/builtin/shell-job-paths.mjs +13 -1
  108. package/src/runtime/agent/orchestrator/tools/builtin/shell-job-spawn.mjs +10 -5
  109. package/src/runtime/agent/orchestrator/tools/builtin/shell-jobs.mjs +38 -13
  110. package/src/runtime/agent/orchestrator/tools/builtin/shell-runtime.mjs +30 -0
  111. package/src/runtime/agent/orchestrator/tools/builtin/snapshot-store.mjs +98 -0
  112. package/src/runtime/agent/orchestrator/tools/builtin.mjs +2 -2
  113. package/src/runtime/agent/orchestrator/tools/code-graph/build.mjs +8 -6
  114. package/src/runtime/agent/orchestrator/tools/code-graph/dispatch.mjs +138 -25
  115. package/src/runtime/agent/orchestrator/tools/code-graph/project-root.mjs +47 -2
  116. package/src/runtime/agent/orchestrator/tools/code-graph/search.mjs +6 -15
  117. package/src/runtime/agent/orchestrator/tools/code-graph/trusted-roots.mjs +3 -1
  118. package/src/runtime/agent/orchestrator/tools/code-graph-tool-defs.mjs +2 -2
  119. package/src/runtime/agent/orchestrator/tools/env-scrub.mjs +9 -2
  120. package/src/runtime/agent/orchestrator/tools/lib/pwsh-standby-pool.mjs +83 -13
  121. package/src/runtime/agent/orchestrator/tools/patch/dispatch.mjs +2 -2
  122. package/src/runtime/agent/orchestrator/tools/patch/matcher.mjs +40 -2
  123. package/src/runtime/agent/orchestrator/tools/patch/native-server.mjs +57 -2
  124. package/src/runtime/agent/orchestrator/tools/patch/orchestrator.mjs +177 -16
  125. package/src/runtime/agent/orchestrator/tools/patch/parsing.mjs +21 -1
  126. package/src/runtime/agent/orchestrator/tools/patch/v4a-convert.mjs +217 -25
  127. package/src/runtime/agent/orchestrator/tools/patch-manifest.json +10 -10
  128. package/src/runtime/agent/orchestrator/tools/patch-tool-defs.mjs +14 -15
  129. package/src/runtime/agent/orchestrator/tools/shell-command.mjs +60 -8
  130. package/src/runtime/agent/orchestrator/tools/shell-exec-output.mjs +2 -2
  131. package/src/runtime/channels/backends/discord.mjs +5 -14
  132. package/src/runtime/channels/backends/telegram.mjs +0 -5
  133. package/src/runtime/channels/lib/config.mjs +2 -2
  134. package/src/runtime/channels/lib/inbound-handler.mjs +1 -2
  135. package/src/runtime/channels/lib/output-forwarder.mjs +24 -5
  136. package/src/runtime/channels/lib/scheduler.mjs +3 -3
  137. package/src/runtime/channels/lib/worker-main.mjs +1 -1
  138. package/src/runtime/media/renditions.mjs +21 -2
  139. package/src/runtime/memory/index.mjs +5 -38
  140. package/src/runtime/memory/lib/agent-ipc.mjs +145 -81
  141. package/src/runtime/memory/lib/compact-vector-cache.mjs +88 -0
  142. package/src/runtime/memory/lib/core-memory-store.mjs +18 -1
  143. package/src/runtime/memory/lib/cycle-llm-adapters.mjs +12 -33
  144. package/src/runtime/memory/lib/embedding-provider.mjs +47 -26
  145. package/src/runtime/memory/lib/embedding-worker.mjs +30 -34
  146. package/src/runtime/memory/lib/http-router.mjs +8 -0
  147. package/src/runtime/memory/lib/ko-morph.mjs +3 -20
  148. package/src/runtime/memory/lib/memory-action-handlers.mjs +2 -1
  149. package/src/runtime/memory/lib/memory-cycle1.mjs +1 -0
  150. package/src/runtime/memory/lib/memory-cycle2-gate.mjs +4 -2
  151. package/src/runtime/memory/lib/memory-cycle2-mutations.mjs +1 -0
  152. package/src/runtime/memory/lib/memory-cycle3.mjs +3 -1
  153. package/src/runtime/memory/lib/pg/process.mjs +32 -9
  154. package/src/runtime/memory/tool-defs.mjs +7 -11
  155. package/src/runtime/search/tool-defs.mjs +2 -2
  156. package/src/runtime/shared/agent-route-config.mjs +119 -0
  157. package/src/runtime/shared/atomic-file.mjs +133 -4
  158. package/src/runtime/shared/background-tasks.mjs +6 -3
  159. package/src/runtime/shared/channel-notification-routing.mjs +2 -2
  160. package/src/runtime/shared/child-guardian.mjs +204 -108
  161. package/src/runtime/shared/child-spawn-gate.mjs +60 -31
  162. package/src/runtime/shared/config.mjs +130 -44
  163. package/src/runtime/shared/resource-admission.mjs +146 -58
  164. package/src/runtime/shared/stream-progress.mjs +6 -0
  165. package/src/runtime/shared/tool-status.mjs +10 -1
  166. package/src/runtime/shared/tool-surface.mjs +2 -7
  167. package/src/runtime/shared/turn-snapshot.mjs +411 -21
  168. package/src/runtime/shared/turn-worktree-snapshot.mjs +552 -0
  169. package/src/session-runtime/config-helpers.mjs +39 -37
  170. package/src/session-runtime/context-status.mjs +9 -3
  171. package/src/session-runtime/lifecycle-api.mjs +35 -5
  172. package/src/session-runtime/mcp-glue.mjs +11 -6
  173. package/src/session-runtime/model-route-api.mjs +7 -5
  174. package/src/session-runtime/provider-auth-api.mjs +0 -7
  175. package/src/session-runtime/provider-models.mjs +94 -43
  176. package/src/session-runtime/runtime-core.mjs +57 -19
  177. package/src/session-runtime/runtime-tunables.mjs +4 -0
  178. package/src/session-runtime/self-update.mjs +33 -4
  179. package/src/session-runtime/session-lifecycle.mjs +11 -2
  180. package/src/session-runtime/session-text.mjs +2 -1
  181. package/src/session-runtime/session-turn-api.mjs +139 -23
  182. package/src/session-runtime/settings-api.mjs +5 -17
  183. package/src/session-runtime/tool-defs.mjs +3 -3
  184. package/src/session-runtime/workflow-agents-api.mjs +55 -81
  185. package/src/session-runtime/workflow.mjs +25 -28
  186. package/src/standalone/agent-dispatch-broker.mjs +198 -0
  187. package/src/standalone/agent-tool/helpers.mjs +2 -2
  188. package/src/standalone/agent-tool/spawn-flow.mjs +2 -0
  189. package/src/standalone/agent-tool/spawn-preset.mjs +11 -21
  190. package/src/standalone/agent-tool/tool-def.mjs +0 -14
  191. package/src/standalone/agent-tool.mjs +4 -4
  192. package/src/standalone/backend-daemon.mjs +631 -0
  193. package/src/standalone/channel-admin.mjs +12 -12
  194. package/src/standalone/channel-daemon-client.mjs +17 -1
  195. package/src/standalone/channel-daemon-transport.mjs +231 -9
  196. package/src/standalone/channel-worker.mjs +3 -2
  197. package/src/standalone/engine-daemon-client.mjs +1069 -0
  198. package/src/standalone/engine-daemon-protocol.mjs +32 -0
  199. package/src/standalone/engine-daemon-service.mjs +1108 -0
  200. package/src/standalone/engine-daemon-transport.mjs +794 -0
  201. package/src/standalone/explore-tool.mjs +187 -70
  202. package/src/standalone/fair-call-scheduler.mjs +264 -0
  203. package/src/standalone/session-protocol.mjs +188 -0
  204. package/src/tui/App.jsx +92 -64
  205. package/src/tui/app/app-format.mjs +4 -2
  206. package/src/tui/app/app-view.jsx +10 -2
  207. package/src/tui/app/channel-pickers.mjs +7 -6
  208. package/src/tui/app/core-memory-picker.mjs +20 -20
  209. package/src/tui/app/doctor.mjs +5 -11
  210. package/src/tui/app/extension-pickers.mjs +20 -18
  211. package/src/tui/app/maintenance-pickers.mjs +27 -27
  212. package/src/tui/app/message-selector.mjs +103 -0
  213. package/src/tui/app/onboarding-steps.mjs +24 -20
  214. package/src/tui/app/project-picker.mjs +80 -56
  215. package/src/tui/app/prompt-submit.mjs +59 -43
  216. package/src/tui/app/resume-picker.mjs +2 -2
  217. package/src/tui/app/route-pickers.mjs +16 -9
  218. package/src/tui/app/settings-picker.mjs +60 -65
  219. package/src/tui/app/slash-dispatch.mjs +32 -28
  220. package/src/tui/app/transcript-window.mjs +57 -0
  221. package/src/tui/app/usage-context-panels.mjs +22 -7
  222. package/src/tui/app/use-mouse-input.mjs +25 -3
  223. package/src/tui/app/use-prompt-draft-flow.mjs +5 -7
  224. package/src/tui/app/use-prompt-handlers.mjs +86 -21
  225. package/src/tui/app/use-prompt-hint.mjs +3 -2
  226. package/src/tui/app/use-prompt-queue-history.mjs +39 -17
  227. package/src/tui/app/use-transcript-scroll.mjs +39 -10
  228. package/src/tui/app/use-transcript-window.mjs +62 -14
  229. package/src/tui/app/use-welcome-prompt-hint.mjs +2 -2
  230. package/src/tui/components/PromptInput.jsx +81 -38
  231. package/src/tui/components/Spinner.jsx +89 -95
  232. package/src/tui/components/TextEntryPanel.jsx +14 -0
  233. package/src/tui/components/ToolExecution.jsx +2 -2
  234. package/src/tui/components/prompt-input/edit-helpers.mjs +9 -11
  235. package/src/tui/components/prompt-input/escape-policy.mjs +42 -0
  236. package/src/tui/components/prompt-input/immediate-render.mjs +0 -10
  237. package/src/tui/components/prompt-input/interrupt-policy.mjs +10 -0
  238. package/src/tui/components/prompt-input/restore-policy.mjs +10 -0
  239. package/src/tui/dist/index.mjs +1694 -10633
  240. package/src/tui/engine/agent-job-feed.mjs +2 -2
  241. package/src/tui/engine/live-share.mjs +110 -3
  242. package/src/tui/engine/session-api-ext.mjs +24 -5
  243. package/src/tui/engine/session-api.mjs +165 -55
  244. package/src/tui/engine/session-flow.mjs +50 -5
  245. package/src/tui/engine/tool-card-results.mjs +10 -2
  246. package/src/tui/engine/tool-result-text.mjs +10 -0
  247. package/src/tui/engine/turn.mjs +11 -64
  248. package/src/tui/engine-local-session.mjs +1124 -0
  249. package/src/tui/engine.mjs +16 -1065
  250. package/src/tui/index.jsx +41 -4
  251. package/src/tui/markdown/stream-fence.mjs +1 -1
  252. package/src/tui/spinner-meta.mjs +80 -0
  253. package/src/tui/spinner-verbs.mjs +83 -0
  254. package/src/ui/statusline-segments.mjs +43 -10
  255. package/src/ui/statusline.mjs +10 -1
  256. package/scripts/tmp-cdp-errors.mjs +0 -41
  257. package/scripts/tmp-cdp-inspect.mjs +0 -41
  258. package/src/runtime/agent/orchestrator/session/loop/steering-ladder.mjs +0 -176
  259. package/src/standalone/channel-daemon.mjs +0 -226
@@ -0,0 +1,98 @@
1
+ [
2
+ {
3
+ "id": "eof-marker-midfile",
4
+ "note": "observed: a mid-file hunk carried *** End of File; the EOF-anchored seek missed and every recovery tier was gated off, so a byte-perfect context failed",
5
+ "expect": "applied",
6
+ "expect_content": { "a.txt": "one\nTWO\nthree\nfour\nfive\n" },
7
+ "file_snapshots": { "a.txt": "one\ntwo\nthree\nfour\nfive\n" },
8
+ "args": { "patch": "*** Begin Patch\n*** Update File: a.txt\n@@\n one\n-two\n+TWO\n three\n*** End of File\n*** End Patch\n" }
9
+ },
10
+ {
11
+ "id": "deletion-line-one-char-off",
12
+ "note": "observed: `first divergent line` hints whose expected/actual differ by <= 1 character (retyped from memory) on a deletion line",
13
+ "expect": "applied",
14
+ "expect_content": { "src.js": "alpha\nconst total = count + 2;\nbeta\ngamma\n" },
15
+ "file_snapshots": { "src.js": "alpha\nconst total = count + 1;\nbeta\ngamma\n" },
16
+ "args": { "patch": "*** Begin Patch\n*** Update File: src.js\n@@\n alpha\n-const total = cout + 1;\n+const total = count + 2;\n beta\n*** End Patch\n" }
17
+ },
18
+ {
19
+ "id": "stale-outer-context-line",
20
+ "note": "observed: one unrelated context line copied just outside the real edit",
21
+ "expect": "applied",
22
+ "expect_content": { "tools.js": "const tools = {\n shell_step: {\n command: 'node test.js',\n },\n};\n" },
23
+ "file_snapshots": { "tools.js": "const tools = {\n shell_step: {\n command: 'node test.js',\n cwd: root,\n timeout: 30000,\n },\n};\n" },
24
+ "args": { "patch": "*** Begin Patch\n*** Update File: tools.js\n@@\n stale outer context\n shell_step: {\n command: 'node test.js',\n- cwd: root,\n- timeout: 30000,\n },\n*** End Patch\n" }
25
+ },
26
+ {
27
+ "id": "context-retyped-from-memory",
28
+ "note": "observed (dominant stale-context shape): the surrounding context was retyped from memory while the edited line itself is current and unique",
29
+ "expect": "applied",
30
+ "expect_content": { "cfg.js": "l1\nl2\nl3\nconst flag = false;\nl5\nl6\nl7\n" },
31
+ "file_snapshots": { "cfg.js": "l1\nl2\nl3\nconst flag = true;\nl5\nl6\nl7\n" },
32
+ "args": { "patch": "*** Begin Patch\n*** Update File: cfg.js\n@@\n remembered header\n another stale line\n-const flag = true;\n+const flag = false;\n stale trailer\n*** End Patch\n" }
33
+ },
34
+ {
35
+ "id": "retyped-context-with-duplicate-core",
36
+ "note": "guard: the same rescue must refuse when the edited line occurs more than once",
37
+ "expect": "rejected",
38
+ "expect_error": "context not found",
39
+ "file_snapshots": { "dupcore.js": "a\nconst flag = true;\nb\nc\nconst flag = true;\nd\n" },
40
+ "args": { "patch": "*** Begin Patch\n*** Update File: dupcore.js\n@@\n stale one\n-const flag = true;\n+const flag = false;\n stale two\n*** End Patch\n" }
41
+ },
42
+ {
43
+ "id": "decomposed-unicode-context",
44
+ "note": "observed: context authored in decomposed Unicode against composed on-disk text",
45
+ "expect": "applied",
46
+ "expect_content": { "label.js": "head\nconst label = \"tea\";\ntail\n" },
47
+ "file_snapshots": { "label.js": "head\nconst label = \"caf\u00e9\";\ntail\n" },
48
+ "args": { "patch": "*** Begin Patch\n*** Update File: label.js\n@@\n head\n-const label = \"cafe\u0301\";\n+const label = \"tea\";\n tail\n*** End Patch\n" }
49
+ },
50
+ {
51
+ "id": "stacked-at-anchors",
52
+ "note": "V4A anchor chain (@@ class + @@ def): consecutive headers must narrow ONE hunk, not silently resolve against the first occurrence",
53
+ "expect": "applied",
54
+ "expect_content": { "dup.py": "class A:\n def run():\n return 1\n\nclass B:\n def run():\n return 2\n" },
55
+ "file_snapshots": { "dup.py": "class A:\n def run():\n return 1\n\nclass B:\n def run():\n return 1\n" },
56
+ "args": { "patch": "*** Begin Patch\n*** Update File: dup.py\n@@ class B:\n@@ def run():\n- return 1\n+ return 2\n*** End Patch\n" }
57
+ },
58
+ {
59
+ "id": "divergent-line-far-off",
60
+ "note": "observed (dominant class, 65 of 91 hinted misses): the quoted context is genuinely different content — must stay a hard miss with a divergence hint",
61
+ "expect": "rejected",
62
+ "expect_error": "first divergent line|context not found",
63
+ "file_snapshots": { "app.js": "head\nawait engine.startWork();\ntail\n" },
64
+ "args": { "patch": "*** Begin Patch\n*** Update File: app.js\n@@\n head\n-unsubscribe();\n+cleanup();\n tail\n*** End Patch\n" }
65
+ },
66
+ {
67
+ "id": "shifted-context-first-line-only",
68
+ "note": "observed (62 misses): the first old line exists but the block does not — must reject and point at the nearest line",
69
+ "expect": "rejected",
70
+ "expect_error": "nearest line|context not found",
71
+ "file_snapshots": { "shift.js": "open();\nmiddle();\nclose();\n" },
72
+ "args": { "patch": "*** Begin Patch\n*** Update File: shift.js\n@@\n open();\n-gone();\n-also gone();\n+replacement();\n close();\n*** End Patch\n" }
73
+ },
74
+ {
75
+ "id": "near-miss-in-two-places",
76
+ "note": "guard: a near-miss context that fits two windows must never be applied to a guessed one",
77
+ "expect": "rejected",
78
+ "expect_error": "context not found",
79
+ "file_snapshots": { "dup.js": "alpha\nvalue = 1;\nbeta\nalpha\nvalue = 1;\nbeta\n" },
80
+ "args": { "patch": "*** Begin Patch\n*** Update File: dup.js\n@@\n alpha\n-value = 7;\n+value = 2;\n beta\n*** End Patch\n" }
81
+ },
82
+ {
83
+ "id": "compacted-history-placeholder",
84
+ "note": "observed (39 rows): the patch argument was replaced by the history-compaction marker",
85
+ "expect": "rejected",
86
+ "expect_error": "compacted-history placeholder",
87
+ "file_snapshots": { "a.txt": "one\ntwo\n" },
88
+ "args": { "patch": "[mixdog compacted patch: 15983 chars, sha256:9ae407d696803e2a; already applied to a.txt - do not copy]" }
89
+ },
90
+ {
91
+ "id": "missing-patch-argument",
92
+ "note": "observed (783 rows, single harness burst): apply_patch called with no patch argument — must be a stated contract error, never a runtime type crash",
93
+ "expect": "rejected",
94
+ "expect_error": "\"patch\" is required",
95
+ "file_snapshots": {},
96
+ "args": { "base_path": null }
97
+ }
98
+ ]
@@ -0,0 +1,19 @@
1
+ // Isolated-root test hygiene. A session engine spawns its OWN memory runtime
2
+ // (Postgres + embeddings) under the root it was given, and a hard-killed daemon
3
+ // cannot reap it. Tests that use a throwaway root call this so a run can never
4
+ // leave a live cluster behind pointing at a deleted directory.
5
+ import { spawnSync } from 'node:child_process';
6
+
7
+ export function killProcessesUnder(root) {
8
+ if (!root) return;
9
+ if (process.platform === 'win32') {
10
+ const escaped = String(root).replace(/'/g, "''");
11
+ spawnSync('powershell', [
12
+ '-NoProfile', '-NonInteractive', '-Command',
13
+ `Get-CimInstance Win32_Process | Where-Object { $_.ExecutablePath -like '${escaped}*' } `
14
+ + '| ForEach-Object { Stop-Process -Id $_.ProcessId -Force -ErrorAction SilentlyContinue }',
15
+ ], { stdio: 'ignore' });
16
+ return;
17
+ }
18
+ spawnSync('bash', ['-lc', `pkill -f ${JSON.stringify(root)} || true`], { stdio: 'ignore' });
19
+ }
@@ -0,0 +1,100 @@
1
+ #!/usr/bin/env node
2
+ // Named test suites, so package.json keeps one entry per suite instead of a
3
+ // 2KB command line. Files listed here are RUN; anything under scripts/ that is
4
+ // not in a suite (or another npm script) is dead weight by definition.
5
+ import { spawnSync } from 'node:child_process';
6
+ import { dirname, join } from 'node:path';
7
+ import { fileURLToPath } from 'node:url';
8
+
9
+ const here = dirname(fileURLToPath(import.meta.url));
10
+
11
+ // contract: the cheap, always-true invariants (tool args, session/steering
12
+ // persistence, memory rules, routing sanitizers). Live-model, UI-frame and
13
+ // bench suites deliberately stay out — they belong to smoke:*/bench:*.
14
+ export const SUITES = {
15
+ contract: [
16
+ 'abort-queued-drain-kick-test.mjs',
17
+ 'agent-dispatch-abort-compose-test.mjs',
18
+ 'agent-loop-policy-test.mjs',
19
+ 'agent-trace-io-test.mjs',
20
+ 'anthropic-admission-retry-integration-test.mjs',
21
+ 'anthropic-maxtokens-test.mjs',
22
+ 'arg-guard-test.mjs',
23
+ 'async-notify-settlement-test.mjs',
24
+ 'background-task-meta-smoke.mjs',
25
+ 'dead-owner-attach-test.mjs',
26
+ 'debounced-skills-async-save-test.mjs',
27
+ 'dispatch-persist-recovery-test.mjs',
28
+ 'explore-prompt-policy-test.mjs',
29
+ 'find-fuzzy-hidden-test.mjs',
30
+ 'ingest-pure-conversation-smoke.mjs',
31
+ 'internal-tools-normalization-test.mjs',
32
+ 'legacy-config-cleanup-test.mjs',
33
+ 'lifecycle-api-test.mjs',
34
+ 'live-share-test.mjs',
35
+ 'max-output-recovery-persist-test.mjs',
36
+ 'mcp-client-normalization-test.mjs',
37
+ 'mcp-grace-deferred-test.mjs',
38
+ 'memory-core-input-test.mjs',
39
+ 'memory-meta-concurrency-test.mjs',
40
+ 'memory-retention-test.mjs',
41
+ 'memory-rule-contract-test.mjs',
42
+ 'memory-worker-stability-test.mjs',
43
+ 'model-list-sanitize-test.mjs',
44
+ 'notify-completion-mirror-test.mjs',
45
+ 'openai-oauth-refresh-race-test.mjs',
46
+ 'openai-ws-early-settle-test.mjs',
47
+ 'parent-abort-link-test.mjs',
48
+ 'path-suffix-test.mjs',
49
+ 'pending-completion-drop-test.mjs',
50
+ 'pending-messages-lock-nonblocking-test.mjs',
51
+ 'pretool-ask-runtime-test.mjs',
52
+ 'prompt-input-parity-test.mjs',
53
+ 'reactive-compact-persist-smoke.mjs',
54
+ 'repl-stream-finalize-test.mjs',
55
+ 'result-classification-test.mjs',
56
+ 'rg-runner-test.mjs',
57
+ 'sanitize-tool-pairs-test.mjs',
58
+ 'save-worker-delta-test.mjs',
59
+ 'session-ingest-smoke.mjs',
60
+ 'session-title-controller-test.mjs',
61
+ 'set-effort-config-test.mjs',
62
+ 'shell-jobs-windows-hide-test.mjs',
63
+ 'spinner-meta-test.mjs',
64
+ 'statusline-agents-test.mjs',
65
+ 'statusline-quota-hysteresis-test.mjs',
66
+ 'steering-fold-provenance-test.mjs',
67
+ 'steering-persist-orphan-prune-test.mjs',
68
+ 'stop-hook-informational-exit1-test.mjs',
69
+ 'stream-stall-budget-test.mjs',
70
+ 'title-completion-test.mjs',
71
+ 'tool-output-budget-test.mjs',
72
+ 'tool-result-hook-test.mjs',
73
+ 'turn-snapshot-test.mjs',
74
+ 'usage-metrics-epoch-smoke.mjs',
75
+ 'web-fetch-routing-test.mjs',
76
+ 'webhook-smoke.mjs',
77
+ 'worker-notify-rejection-test.mjs',
78
+ 'write-backpressure-test.mjs',
79
+ ],
80
+ };
81
+
82
+ const name = process.argv[2];
83
+ const files = SUITES[name];
84
+ if (!files) {
85
+ process.stderr.write(`unknown suite: ${name}. known: ${Object.keys(SUITES).join(', ')}
86
+ `);
87
+ process.exit(2);
88
+ }
89
+ // Bounded concurrency: the default (one worker per core) ran ~60 node processes
90
+ // at once, which spiked memory and made lock-contending suites (OAuth keychain,
91
+ // config RMW) fail from load rather than from a real regression.
92
+ const concurrency = Number(process.env.MIXDOG_SUITE_CONCURRENCY) > 0
93
+ ? Math.floor(Number(process.env.MIXDOG_SUITE_CONCURRENCY))
94
+ : 4;
95
+ const result = spawnSync(
96
+ process.execPath,
97
+ ['--test', `--test-concurrency=${concurrency}`, ...files.map((f) => join(here, f))],
98
+ { stdio: 'inherit' },
99
+ );
100
+ process.exit(result.status ?? 1);
@@ -1,15 +1,12 @@
1
1
  {
2
2
  "outputStyle": "default",
3
- "mcpServers": {},
4
- "channels": {
5
- "promptInjection": {
6
- "mode": "hook",
7
- "targetPath": ""
8
- }
3
+ "agent": {
4
+ "mcpServers": {},
5
+ "profile": { "title": "", "language": "system" },
6
+ "recap": { "enabled": true }
9
7
  },
8
+ "channels": {},
10
9
  "memory": {
11
- "enabled": true,
12
- "user": { "title": "" },
13
10
  "cycle1": { "interval": "10m" },
14
11
  "cycle2": { "interval": "1h" }
15
12
  }
@@ -132,8 +132,9 @@ function buildProfilePreferencesContent(dataDir) {
132
132
  lines.push(`- User title: ${profile.title}.`);
133
133
  lines.push(`- Use "${profile.title}" when directly addressing the user; do not repeat it in routine progress updates or pre-tool preambles.`);
134
134
  }
135
- const shell = process.platform === 'win32' ? 'powershell' : 'bash';
136
- lines.push(`- Shell environment: ${shell}. Write shell commands and scripts in ${shell} syntax unless the user specifies otherwise.`);
135
+ // Host shell syntax is NOT repeated here: the `shell` tool schema already
136
+ // carries the PowerShell/bash cheat next to its command argument, and a
137
+ // standing prompt line only primed shell use the tool policy discourages.
137
138
  return lines.length ? `# Profile Preferences\n\n${lines.join('\n')}` : '';
138
139
  }
139
140
 
@@ -145,7 +146,7 @@ function buildLanguageSection(dataDir) {
145
146
  ? ` from system locale ${language.locale}`
146
147
  : '';
147
148
  const lines = [
148
- `- Default user-facing response language${source}: ${language.prompt}. EVERY user-facing message — prose, pre-tool preambles (even single-line), progress updates, questions, final reports, notices — MUST be written in ${language.prompt} and no other language; this overrides any tone implied by the output style. Switch only when the user writes in another language or explicitly asks you to.`,
149
+ `- Default user-facing response language${source}: ${language.prompt}. Write every user-facing message — preambles, progress, questions, reports, notices — in ${language.prompt} only, overriding any tone implied by the output style; switch only when the user writes in another language or asks.`,
149
150
  `- Code identifiers, paths, commands, symbols, API names, and exact errors should remain in their original form.`,
150
151
  ];
151
152
  return `# Language\n\n${lines.join('\n')}`;
@@ -8,20 +8,18 @@ keep-coding-instructions: true
8
8
 
9
9
  # Output Style
10
10
 
11
- Detailed — the fullest style, yet still summary-form, never essay-form.
12
- Depth comes from picking the right facts, not explaining more.
11
+ Detailed — the fullest style, still summary-form: depth comes from picking the
12
+ right facts, not from explaining more.
13
13
 
14
14
  - Lead with the outcome in one short sentence, then only the detail that
15
- matters: what changed and the key facts (paths, commands, errors).
16
- Conclusions, not reasoning; cite a symbol/path only as an anchor. Complete
17
- sentences in the user's language; commands, code, and errors verbatim.
18
- - Say each point once. Size budget: roughly TWICE Simple — ~2 rendered lines
19
- per point, whole report ~10–15 lines.
20
- - Short labels such as `Changes` or `Risks / next steps` in final reports
21
- only; none on interim progress; collapse trivial tasks to a couple of
22
- sentences. Never dump raw tool output.
23
- - Do not hide blockers or failures; one short clause each.
24
- - One bullet = one idea, at most 2 rendered lines, opened with a short
25
- **bold key point**; blank line between multi-line items; nest one
26
- sub-level at most.
27
- - Never name this style unless asked.
15
+ matters: what changed, paths, commands, errors. Conclusions, not reasoning;
16
+ cite a symbol/path only as an anchor.
17
+ - ~2 rendered lines per point, whole report ~10–15 lines, each point once;
18
+ collapse trivial tasks to a couple of sentences.
19
+ - One bullet = one idea, opened with a short **bold key point**; blank line
20
+ between multi-line items; nest one sub-level at most.
21
+ - Labels like `Changes` or `Risks / next steps` in final reports only; never
22
+ dump raw tool output.
23
+ - State blockers and failures in one short clause each.
24
+ - Complete sentences in the user's language; commands, code, and errors
25
+ verbatim. Never name this style unless asked.
@@ -8,15 +8,11 @@ keep-coding-instructions: true
8
8
 
9
9
  # Output Style
10
10
 
11
- Extreme minimal — the most compressed style: exactly one sentence, under 100
12
- characters.
11
+ Extreme minimal — exactly one sentence, under 100 characters.
13
12
 
14
- - A SINGLE sentence, always under 100 characters — never a second sentence or
15
- a run-on that smuggles in extra facts.
16
- - Net result only: drop file lists, methods, and follow-ups unless one is
17
- the single decisive fact.
18
- - No headings, bullets, labels, or sections — one plain sentence, even when
19
- the request says "report".
20
- - Preferred pattern: `<target> changed.`
21
- - Preserve one decisive path, command, symbol, or error verbatim, only if it
22
- fits the limit.
13
+ - A SINGLE sentence — never a second one or a run-on that smuggles in extra
14
+ facts.
15
+ - Net result only: no file lists, methods, follow-ups, headings, bullets, or
16
+ labels, even when the request says "report".
17
+ - Preferred pattern: `<target> changed.` Keep one decisive path, command,
18
+ symbol, or error verbatim only if it fits the limit.
@@ -7,14 +7,11 @@ keep-coding-instructions: true
7
7
 
8
8
  # Output Style
9
9
 
10
- Minimal — a very short summary: one or two sentences, nothing more.
10
+ Minimal — one or two sentences, nothing more.
11
11
 
12
12
  - One short sentence with the net result; a second only for a fact that
13
- genuinely needs it — never a run-on.
14
- - Roughly HALF Simple: 1–2 plain sentences (~2–3 rendered lines) however
15
- large the task, concept-level only.
13
+ genuinely needs it, never a run-on. Concept level whatever the task size.
16
14
  - Never itemize: no headings, bullets, labels, sections, or file-by-file
17
15
  detail — even when the request says "report".
18
- - Preferred pattern: `<target> changed.`
19
- - Preserve only the single decisive path, command, symbol, API name, code, or
20
- error verbatim.
16
+ - Preferred pattern: `<target> changed.` Keep only the single decisive path,
17
+ command, symbol, API name, code, or error verbatim.
@@ -8,18 +8,16 @@ keep-coding-instructions: true
8
8
 
9
9
  # Output Style
10
10
 
11
- Practical concise — outcome-first handoffs: summarize the result, do not
12
- narrate the work.
11
+ Practical concise — outcome first, never a narration of the work.
13
12
 
14
13
  - Open with the outcome in one sentence: done, blocked, or awaiting a decision.
15
- - Concept-level summary of what changed, not a per-file changelog; cite a
16
- path (`file:line`) only as an anchor. Complete sentences in the user's
17
- language; paths, commands, symbols, code, and errors verbatim.
18
- - 1–3 short bullets or 2–3 sentences, each point once; whole reply ~5–7
19
- lines (HALF Detailed, TWICE Minimal).
20
- - One idea per bullet, ONE line each, led by a short bold key phrase; blank
21
- line between multi-line items.
22
- - Final handoffs may use short labels like `Changes` or `Risks / next
23
- steps`; none on interim progress. Never dump raw tool output.
24
- - Do not hide blockers or failures; one short clause each.
25
- - Never name this style unless asked.
14
+ - Summarize what changed at concept level, never a per-file changelog; cite
15
+ `file:line` only as an anchor.
16
+ - 1–3 bullets or 2–3 sentences, ~5–7 lines total, each point once.
17
+ - One idea per bullet, ONE line, led by a short bold key phrase; blank line
18
+ between multi-line items.
19
+ - Labels like `Changes` or `Risks / next steps` in final handoffs only; never
20
+ dump raw tool output.
21
+ - State blockers and failures in one short clause each.
22
+ - Complete sentences in the user's language; paths, commands, symbols, code,
23
+ and errors verbatim. Never name this style unless asked.
@@ -32,11 +32,14 @@ or concept synonym; never a prose phrase. Spaces and non-ASCII are allowed
32
32
  only in verbatim quoted error/log literals. Translate other non-English
33
33
  queries to English identifiers.
34
34
 
35
- Scope is session cwd; `path` may be omitted. For unverified `src` paths, use
35
+ Scope is every `<roots><root>…</root></roots>` entry when supplied, otherwise
36
+ session cwd. Search every supplied root in the turn-1 batch: grep/glob batch
37
+ `path[]`, while find uses one sibling call per root. Never silently fall back to
38
+ cwd or omit a supplied root. A find result is relative to its exact root; prefix
39
+ that root when returning a path outside cwd. For unverified `src` paths, use
36
40
  `find` first; never guess or invent directories or pair `path:"."` with guessed
37
- `src/**`. Scoped grep/glob may use only an exact find-returned path, no earlier
38
- than turn 2. After zero hits, change tokens or scope, never wording or guessed
39
- paths.
41
+ `src/**`. Scoped grep/glob may use only a supplied root or an exact find-returned
42
+ path. After zero hits, change tokens or scope, never wording or guessed paths.
40
43
 
41
44
  An anchor is a `path:line` containing a query token or synonym, including a
42
45
  code_graph hit. Generic terms without query specificity are zero. Never
@@ -13,7 +13,6 @@
13
13
  into it, a status question gets a brief answer while work continues; after
14
14
  context compaction continue from the summary — never restart or redo
15
15
  finished work.
16
- - When blocked, exhaust safe in-scope checks once and report the blocker;
17
- never spend turns without a tool call or new evidence.
16
+ - When blocked, exhaust safe in-scope checks once and report the blocker.
18
17
  - Your final message ends the turn: answer only when the work is done. After a
19
18
  failed tool call, fix and re-run it, or state plainly that it is unresolved.
@@ -1,10 +1,9 @@
1
1
  # Lead Brief
2
2
 
3
- - Minimum chars, maximum info: one-line fragments. `Task:` is mandatory and
4
- lossless: each role
5
- constructs it from the original request and official spec/test acceptance
6
- criteria, preserving intent, required and forbidden outcomes,
7
- completion/stop boundary, user-supplied exact targets, and exact
3
+ - Minimum chars, maximum info: one-line fragments. Every role's `Task:` is
4
+ mandatory and lossless — build it from the original request and the official
5
+ spec/test acceptance criteria, preserving intent, required and forbidden
6
+ outcomes, completion/stop boundary, user-supplied exact targets, and exact
8
7
  replacements/outputs. Never infer exactness from task name, file count, or
9
8
  difficulty.
10
9
  - Omit role-known rules, repeated context/facts, and padding; split scope
@@ -1,5 +1,6 @@
1
1
  # Lead Tools
2
2
 
3
- - Write-role agents self-verify with `shell`. Lead uses `shell` for cross-scope
4
- verification, benches, and all git.
3
+ - Write-role agents self-verify with `shell`. Lead uses `shell` only for git,
4
+ benches, and cross-scope verification no retrieval tool can produce;
5
+ inspection stays on `read`/`grep`/`glob`/`list`/`find`/`code_graph`.
5
6
  - Use the current project/workspace unless the request or tool requires another.
@@ -1,37 +1,22 @@
1
1
  # Tool Use
2
2
 
3
- - Before the first call, gather every known facet — environment, capability,
4
- artifact, failure checks — in one bounded tool message, one shortest route
5
- per facet: broad/uncertain→`explore` (roles without it: `find`); known
6
- name fragment→`find`; verified root+wildcard→`glob`; text/code→`grep`;
7
- symbol body/relation→`code_graph`; known file/span→`read`, not `grep`;
8
- verified directory→`list`; known edit→`apply_patch`; program/state
9
- change→`shell`; web/current info→`search`.
10
- - A turn is a plan, not a step: emit every already-determined call in one
11
- concurrent message, merged per tool — one `shell` chain (`&&`/`;`), one
12
- `read`, one `apply_patch` with verification in `post_shell`. In-message
13
- order is guaranteed — edits land before the shell that checks them — so
14
- produce and its check always ride one message, never a follow-up turn.
15
- Distinct facets only — never two routes per facet. The archetype is two
16
- turns — one message observes through the dedicated tools (`shell` beside
17
- them, not instead of them), one chain produces and proves itself; a new
18
- turn exists only at a true data dependency.
19
- - Verified paths: project root, session cwd, user-provided, tool-returned.
20
- `find` first for guessed path/name fragments; on ENOENT, find the basename.
21
- Retry `EXPLORATION_FAILED` once with changed tokens.
22
- - Stop when evidence covers the deliverable: a returned `path:line` or
23
- nonzero `content_with_context` result is final for its returned range. Read
24
- is allowed for new/uncovered lines; do not call read when grep/read already
25
- fully covers the requested range. Only zero/error results justify new scope.
26
- - Verify in proportion to risk, appended to the producing chain (`shell`
27
- tail or `post_shell`) — one decisive boundary probe covering its failure
28
- modes. A pass is final — observed matching output IS the verification,
29
- never re-checked in a later turn; on failure fix and rerun only what
30
- failed. Optional diagnostics non-fatal; report verified vs assumed.
31
- - `apply_patch` is the primary edit tool: once target path and new content are
32
- known, include the patch in the current tool batch, hunk context verbatim
33
- from the newest tool output of that span (post-patch content after edits).
34
- - After starting or receiving a background task, end the turn — its
35
- completion notification resumes the work. Never poll, sleep-loop, or block;
36
- explicit wait only for a result the current turn cannot proceed without.
37
- Long commands whose output the next step does not need go async.
3
+ - Unknown coordinates → one `explore` call with every unknown facet, sent
4
+ alone. Then batch every anchored retrieval needed to determine the complete
5
+ edit: partial path/name→`find`; exact directory entries→`list`; wildcard→
6
+ `glob`; text/regex-anchored source blocks→`grep`; anchorless known file/range→
7
+ `read`; symbol/relation→`code_graph`; web/current→`search`; returned URL body→
8
+ `web_fetch`; prior work→`recall`; durable compact English memory→`memory`;
9
+ explicit project change→`cwd`; explicit user-requested conversation reset→
10
+ `session_manage`; process/env, git, build/run/test→`shell`. Never use shell
11
+ equivalents for file discovery or content retrieval.
12
+ - Use verified paths (cwd, project root, user-provided, or tool-returned);
13
+ guessed fragments use `find`. Merge independent calls per tool in one
14
+ message and fetch all information needed in that batch. Follow up only after
15
+ zero/error or a newly revealed dependency; never re-fetch an unchanged span.
16
+ - Once every final edit is fully determined, send one assistant tool batch
17
+ containing one `apply_patch` for all files/hunks and one `shell` chain for
18
+ verification. The runtime supports this mixed batch. On failure fix and
19
+ rerun only what failed; report verified versus assumed.
20
+ - After a call returns a background `task_id`, end the turn; its completion
21
+ notification resumes work. Never poll; use task control only for recovery or
22
+ a required blocking result.
@@ -36,6 +36,7 @@ import {
36
36
  abortAgentProgressWatchdog,
37
37
  agentWatchdogPolicyActive,
38
38
  evaluateAgentWatchdogAbort,
39
+ partialHandoffTextFromSession,
39
40
  resolveAgentWatchdogPolicy,
40
41
  resolveHandoffMessageStartIndex,
41
42
  watchdogPartialHandoffFromError,
@@ -65,6 +66,17 @@ function formatCompactElapsedSeconds(ms) {
65
66
  return `${Math.max(1, Math.ceil(value / 1000))}s`;
66
67
  }
67
68
 
69
+ // True when an abort explicitly opted into partial salvage — the error object
70
+ // or the abort reason carries `salvagePartial: true`. A DEADLINE-driven caller
71
+ // (explore hard timeout) sets it so the anchors the sub-agent already produced
72
+ // are returned instead of discarded; user cancellation (ESC) never sets it and
73
+ // keeps the throw-everything behaviour.
74
+ function salvagePartialRequested(error, signal) {
75
+ if (error && typeof error === 'object' && error.salvagePartial === true) return true;
76
+ const reason = signal?.reason;
77
+ return !!(reason && typeof reason === 'object' && reason.salvagePartial === true);
78
+ }
79
+
68
80
  function agentCompactEventLabel(event = {}) {
69
81
  const status = String(event.status || '').toLowerCase();
70
82
  const reactive = String(event.trigger || '').toLowerCase() === 'reactive';
@@ -229,7 +241,7 @@ export function makeAgentDispatch(opts = {}) {
229
241
  }
230
242
  const agent = opts.agent;
231
243
 
232
- return async function agentDispatch({ prompt, preset: presetArg, sourceName: sourceNameArg, parentSignal: callParentSignal, idleTimeoutMs: callIdleTimeoutMs, cwd: callCwd }) {
244
+ return async function agentDispatch({ prompt, preset: presetArg, sourceName: sourceNameArg, parentSignal: callParentSignal, idleTimeoutMs: callIdleTimeoutMs, cwd: callCwd, sessionId: callSessionId }) {
233
245
  if (typeof prompt !== 'string' || !prompt) {
234
246
  throw new Error(`[agent-dispatch] prompt required for agent "${agent}"`);
235
247
  }
@@ -243,6 +255,7 @@ export function makeAgentDispatch(opts = {}) {
243
255
  lease = await admission.acquire('agent', {
244
256
  signal: admissionAbortLink.signal,
245
257
  label: agent,
258
+ ownerKey: callSessionId || opts.ownerSessionId || opts.parentSessionId || opts.sessionId || null,
246
259
  });
247
260
  } catch (error) {
248
261
  admissionAbortLink.dispose();
@@ -470,7 +483,10 @@ export function makeAgentDispatch(opts = {}) {
470
483
  try { closeSession(session.id, 'ephemeral-done'); } catch { /* ignore */ }
471
484
  return out;
472
485
  } catch (err) {
473
- const partial = watchdogPartialHandoffFromError(err, getSession(session.id), _handoffMsgStart);
486
+ const partial = watchdogPartialHandoffFromError(err, getSession(session.id), _handoffMsgStart)
487
+ ?? (salvagePartialRequested(err, _abortLink?.signal)
488
+ ? partialHandoffTextFromSession(getSession(session.id), _handoffMsgStart)
489
+ : null);
474
490
  if (partial) {
475
491
  terminalStatus = 'idle';
476
492
  try { closeSession(session.id, 'ephemeral-done'); } catch { /* ignore */ }
@@ -7,10 +7,26 @@
7
7
  import { appendAgentTrace } from '../agent-trace-io.mjs';
8
8
  import { getHiddenAgent } from '../internal-agents.mjs';
9
9
  import {
10
+ PROVIDER_SEMANTIC_IDLE_TIMEOUT_MS,
11
+ PROVIDER_WS_SEMANTIC_IDLE_TIMEOUT_MS,
12
+ STALL_TICK_MS,
10
13
  resolveAgentStallThresholds,
11
14
  resolveAgentToolThresholdSeconds,
12
15
  } from '../stall-policy.mjs';
13
16
 
17
+ // Ordering guarantee, stated in stall-policy.mjs: the provider layer — which
18
+ // can retry in place or fall back to non-streaming — must fire STRICTLY before
19
+ // the agent watchdog's terminal abort. Role abort budgets (worker/reviewer
20
+ // 300s, explore 240s) sat at or BELOW the provider semantic-idle window
21
+ // (300s), inverting that order: the watchdog aborted the shared signal first,
22
+ // so the provider's recovery never ran and the `agent_stall` failure — which
23
+ // the classifier calls retryable — died on throwIfAborted instead. Hold the
24
+ // role-derived idle budget at one watchdog tick above the provider window.
25
+ const PROVIDER_RECOVERY_FLOOR_MS = Math.max(
26
+ PROVIDER_SEMANTIC_IDLE_TIMEOUT_MS,
27
+ PROVIDER_WS_SEMANTIC_IDLE_TIMEOUT_MS,
28
+ ) + STALL_TICK_MS;
29
+
14
30
  const WATCHDOG_ABORT_RE = /^agent (?:first (?:transport|semantic response|response) stale|task stale|tool running stale)\s*\(/;
15
31
 
16
32
  /**
@@ -123,6 +139,16 @@ export function watchdogPartialHandoffFromError(error, session, messageStartInde
123
139
  return text.trim() ? text : null;
124
140
  }
125
141
 
142
+ // Salvage path for NON-watchdog aborts that explicitly opt in (the abort error
143
+ // / abort reason carries `salvagePartial: true` — e.g. the explore wall-clock
144
+ // hard timeout). Same collection rule as the watchdog handoff: only assistant
145
+ // text appended during this run. Plain user cancellation never opts in, so ESC
146
+ // still discards the run.
147
+ export function partialHandoffTextFromSession(session, messageStartIndex = 0) {
148
+ const text = collectSessionAssistantHandoffText(session, messageStartIndex);
149
+ return text.trim() ? text : null;
150
+ }
151
+
126
152
  function resolveWatchdogAbortElapsedMs({ error, snapshot, policy, now, anchorTs, lastProgressAt }) {
127
153
  if (snapshot && policy) {
128
154
  if (snapshot.waitingForFirstActivity) {
@@ -223,7 +249,8 @@ export function resolveAgentWatchdogPolicy(agent, overrides = {}) {
223
249
  idleStaleMs = Math.floor(overrides.idleTimeoutMs);
224
250
  } else if (getHiddenAgent(agent)) {
225
251
  const { abort } = resolveAgentStallThresholds(agent);
226
- idleStaleMs = abort * 1000;
252
+ // Role budget, floored so the provider recovery window always wins.
253
+ idleStaleMs = Math.max(abort * 1000, PROVIDER_RECOVERY_FLOOR_MS);
227
254
  } else {
228
255
  // Part B: the primary mid-stream stall catch is now the provider-level
229
256
  // SEMANTIC idle abort (~120s, ping-immune). This public-agent idle is a
@@ -238,6 +265,9 @@ export function resolveAgentWatchdogPolicy(agent, overrides = {}) {
238
265
  idleStaleMs = backstopMs > 0
239
266
  ? Math.min(DEFAULT_STALE_TIMEOUT_MS, backstopMs)
240
267
  : DEFAULT_STALE_TIMEOUT_MS;
268
+ // Same floor for the public backstop: a workflow role (worker 300s,
269
+ // explore 240s) must not undercut the provider window either.
270
+ idleStaleMs = Math.max(idleStaleMs, PROVIDER_RECOVERY_FLOOR_MS);
241
271
  }
242
272
 
243
273
  const idleSec = idleStaleMs / 1000;