@monoes/monomindcli 2.13.0 → 2.14.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (251) hide show
  1. package/.claude/skills/monodesign/scripts/context.mjs +1 -1
  2. package/.claude/skills/monodesign/scripts/detector/detect-antipatterns-browser.js +2 -2
  3. package/.claude/skills/monodesign/scripts/detector/rules/checks.mjs +2 -2
  4. package/.claude/skills/monodesign/scripts/live-browser.js +1 -1
  5. package/bin/cli.js +21 -1
  6. package/bin/mcp-server.js +21 -1
  7. package/dist/src/capabilities/cap-documents.d.ts.map +1 -1
  8. package/dist/src/capabilities/cap-documents.js +23 -0
  9. package/dist/src/capabilities/cap-documents.js.map +1 -1
  10. package/dist/src/commands/doc-filters.d.ts +19 -0
  11. package/dist/src/commands/doc-filters.d.ts.map +1 -0
  12. package/dist/src/commands/doc-filters.js +58 -0
  13. package/dist/src/commands/doc-filters.js.map +1 -0
  14. package/dist/src/commands/doc-library.d.ts +21 -0
  15. package/dist/src/commands/doc-library.d.ts.map +1 -0
  16. package/dist/src/commands/doc-library.js +390 -0
  17. package/dist/src/commands/doc-library.js.map +1 -0
  18. package/dist/src/commands/doc-list.d.ts +9 -0
  19. package/dist/src/commands/doc-list.d.ts.map +1 -0
  20. package/dist/src/commands/doc-list.js +99 -0
  21. package/dist/src/commands/doc-list.js.map +1 -0
  22. package/dist/src/commands/doc.d.ts.map +1 -1
  23. package/dist/src/commands/doc.js +49 -37
  24. package/dist/src/commands/doc.js.map +1 -1
  25. package/dist/src/commands/doctor-mcp-probe.d.ts +48 -0
  26. package/dist/src/commands/doctor-mcp-probe.d.ts.map +1 -0
  27. package/dist/src/commands/doctor-mcp-probe.js +205 -0
  28. package/dist/src/commands/doctor-mcp-probe.js.map +1 -0
  29. package/dist/src/commands/doctor-project-checks.d.ts +12 -1
  30. package/dist/src/commands/doctor-project-checks.d.ts.map +1 -1
  31. package/dist/src/commands/doctor-project-checks.js +55 -4
  32. package/dist/src/commands/doctor-project-checks.js.map +1 -1
  33. package/dist/src/commands/doctor.d.ts.map +1 -1
  34. package/dist/src/commands/doctor.js +10 -1
  35. package/dist/src/commands/doctor.js.map +1 -1
  36. package/dist/src/commands/init.d.ts.map +1 -1
  37. package/dist/src/commands/init.js +17 -0
  38. package/dist/src/commands/init.js.map +1 -1
  39. package/dist/src/commands/mcp.d.ts.map +1 -1
  40. package/dist/src/commands/mcp.js +4 -3
  41. package/dist/src/commands/mcp.js.map +1 -1
  42. package/dist/src/commands/org-observe.d.ts.map +1 -1
  43. package/dist/src/commands/org-observe.js +14 -0
  44. package/dist/src/commands/org-observe.js.map +1 -1
  45. package/dist/src/commands/org.d.ts.map +1 -1
  46. package/dist/src/commands/org.js +12 -2
  47. package/dist/src/commands/org.js.map +1 -1
  48. package/dist/src/init/claudemd-generator.d.ts.map +1 -1
  49. package/dist/src/init/claudemd-generator.js +2 -1
  50. package/dist/src/init/claudemd-generator.js.map +1 -1
  51. package/dist/src/init/mcp-generator.d.ts.map +1 -1
  52. package/dist/src/init/mcp-generator.js +3 -3
  53. package/dist/src/init/mcp-generator.js.map +1 -1
  54. package/dist/src/init/types.d.ts +6 -0
  55. package/dist/src/init/types.d.ts.map +1 -1
  56. package/dist/src/init/types.js.map +1 -1
  57. package/dist/src/init/write-runtime-config.d.ts.map +1 -1
  58. package/dist/src/init/write-runtime-config.js +12 -2
  59. package/dist/src/init/write-runtime-config.js.map +1 -1
  60. package/dist/src/knowledge/capture-envelope.d.ts +80 -0
  61. package/dist/src/knowledge/capture-envelope.d.ts.map +1 -0
  62. package/dist/src/knowledge/capture-envelope.js +165 -0
  63. package/dist/src/knowledge/capture-envelope.js.map +1 -0
  64. package/dist/src/knowledge/capture-text.d.ts +23 -0
  65. package/dist/src/knowledge/capture-text.d.ts.map +1 -0
  66. package/dist/src/knowledge/capture-text.js +49 -0
  67. package/dist/src/knowledge/capture-text.js.map +1 -0
  68. package/dist/src/knowledge/citation.d.ts +121 -0
  69. package/dist/src/knowledge/citation.d.ts.map +1 -0
  70. package/dist/src/knowledge/citation.js +252 -0
  71. package/dist/src/knowledge/citation.js.map +1 -0
  72. package/dist/src/knowledge/document-pipeline.d.ts +60 -0
  73. package/dist/src/knowledge/document-pipeline.d.ts.map +1 -1
  74. package/dist/src/knowledge/document-pipeline.js +188 -4
  75. package/dist/src/knowledge/document-pipeline.js.map +1 -1
  76. package/dist/src/knowledge/html-extract.d.ts +43 -0
  77. package/dist/src/knowledge/html-extract.d.ts.map +1 -0
  78. package/dist/src/knowledge/html-extract.js +286 -0
  79. package/dist/src/knowledge/html-extract.js.map +1 -0
  80. package/dist/src/knowledge/html-tags.d.ts +30 -0
  81. package/dist/src/knowledge/html-tags.d.ts.map +1 -0
  82. package/dist/src/knowledge/html-tags.js +229 -0
  83. package/dist/src/knowledge/html-tags.js.map +1 -0
  84. package/dist/src/knowledge/library.d.ts +106 -0
  85. package/dist/src/knowledge/library.d.ts.map +1 -0
  86. package/dist/src/knowledge/library.js +195 -0
  87. package/dist/src/knowledge/library.js.map +1 -0
  88. package/dist/src/knowledge/mhtml.d.ts +60 -0
  89. package/dist/src/knowledge/mhtml.d.ts.map +1 -0
  90. package/dist/src/knowledge/mhtml.js +154 -0
  91. package/dist/src/knowledge/mhtml.js.map +1 -0
  92. package/dist/src/knowledge/related.d.ts +58 -0
  93. package/dist/src/knowledge/related.d.ts.map +1 -0
  94. package/dist/src/knowledge/related.js +222 -0
  95. package/dist/src/knowledge/related.js.map +1 -0
  96. package/dist/src/knowledge/section-diff.d.ts +48 -0
  97. package/dist/src/knowledge/section-diff.d.ts.map +1 -0
  98. package/dist/src/knowledge/section-diff.js +151 -0
  99. package/dist/src/knowledge/section-diff.js.map +1 -0
  100. package/dist/src/knowledge/watch.d.ts +92 -0
  101. package/dist/src/knowledge/watch.d.ts.map +1 -0
  102. package/dist/src/knowledge/watch.js +221 -0
  103. package/dist/src/knowledge/watch.js.map +1 -0
  104. package/dist/src/mcp-server.d.ts.map +1 -1
  105. package/dist/src/mcp-server.js +44 -87
  106. package/dist/src/mcp-server.js.map +1 -1
  107. package/dist/src/mcp-tools/browser-instrument-tools.d.ts +35 -0
  108. package/dist/src/mcp-tools/browser-instrument-tools.d.ts.map +1 -0
  109. package/dist/src/mcp-tools/browser-instrument-tools.js +359 -0
  110. package/dist/src/mcp-tools/browser-instrument-tools.js.map +1 -0
  111. package/dist/src/mcp-tools/browser-metrics.d.ts +42 -0
  112. package/dist/src/mcp-tools/browser-metrics.d.ts.map +1 -0
  113. package/dist/src/mcp-tools/browser-metrics.js +91 -0
  114. package/dist/src/mcp-tools/browser-metrics.js.map +1 -0
  115. package/dist/src/mcp-tools/browser-profile-tools.d.ts +16 -0
  116. package/dist/src/mcp-tools/browser-profile-tools.d.ts.map +1 -0
  117. package/dist/src/mcp-tools/browser-profile-tools.js +324 -0
  118. package/dist/src/mcp-tools/browser-profile-tools.js.map +1 -0
  119. package/dist/src/mcp-tools/browser-session.d.ts +68 -0
  120. package/dist/src/mcp-tools/browser-session.d.ts.map +1 -0
  121. package/dist/src/mcp-tools/browser-session.js +224 -0
  122. package/dist/src/mcp-tools/browser-session.js.map +1 -0
  123. package/dist/src/mcp-tools/browser-tools.d.ts.map +1 -1
  124. package/dist/src/mcp-tools/browser-tools.js +10 -176
  125. package/dist/src/mcp-tools/browser-tools.js.map +1 -1
  126. package/dist/src/mcp-tools/capture-resource-read.d.ts +115 -0
  127. package/dist/src/mcp-tools/capture-resource-read.d.ts.map +1 -0
  128. package/dist/src/mcp-tools/capture-resource-read.js +296 -0
  129. package/dist/src/mcp-tools/capture-resource-read.js.map +1 -0
  130. package/dist/src/mcp-tools/capture-resource-tools.d.ts +22 -0
  131. package/dist/src/mcp-tools/capture-resource-tools.d.ts.map +1 -0
  132. package/dist/src/mcp-tools/capture-resource-tools.js +182 -0
  133. package/dist/src/mcp-tools/capture-resource-tools.js.map +1 -0
  134. package/dist/src/mcp-tools/capture-resources.d.ts +142 -0
  135. package/dist/src/mcp-tools/capture-resources.d.ts.map +1 -0
  136. package/dist/src/mcp-tools/capture-resources.js +289 -0
  137. package/dist/src/mcp-tools/capture-resources.js.map +1 -0
  138. package/dist/src/mcp-tools/index.d.ts +3 -0
  139. package/dist/src/mcp-tools/index.d.ts.map +1 -1
  140. package/dist/src/mcp-tools/index.js +6 -0
  141. package/dist/src/mcp-tools/index.js.map +1 -1
  142. package/dist/src/mcp-tools/knowledge-tools.d.ts.map +1 -1
  143. package/dist/src/mcp-tools/knowledge-tools.js +9 -1
  144. package/dist/src/mcp-tools/knowledge-tools.js.map +1 -1
  145. package/dist/src/mcp-tools/resource-router.d.ts +86 -0
  146. package/dist/src/mcp-tools/resource-router.d.ts.map +1 -0
  147. package/dist/src/mcp-tools/resource-router.js +181 -0
  148. package/dist/src/mcp-tools/resource-router.js.map +1 -0
  149. package/dist/src/memory/memory-bridge.js +1 -1
  150. package/dist/src/memory/memory-bridge.js.map +1 -1
  151. package/dist/src/orgrt/agent-runner.d.ts +38 -0
  152. package/dist/src/orgrt/agent-runner.d.ts.map +1 -1
  153. package/dist/src/orgrt/agent-runner.js +47 -0
  154. package/dist/src/orgrt/agent-runner.js.map +1 -1
  155. package/dist/src/orgrt/checkpoint.d.ts +12 -1
  156. package/dist/src/orgrt/checkpoint.d.ts.map +1 -1
  157. package/dist/src/orgrt/checkpoint.js +15 -6
  158. package/dist/src/orgrt/checkpoint.js.map +1 -1
  159. package/dist/src/orgrt/completion-gate.d.ts +74 -0
  160. package/dist/src/orgrt/completion-gate.d.ts.map +1 -1
  161. package/dist/src/orgrt/completion-gate.js +45 -0
  162. package/dist/src/orgrt/completion-gate.js.map +1 -1
  163. package/dist/src/orgrt/cost-tier.d.ts +144 -0
  164. package/dist/src/orgrt/cost-tier.d.ts.map +1 -0
  165. package/dist/src/orgrt/cost-tier.js +183 -0
  166. package/dist/src/orgrt/cost-tier.js.map +1 -0
  167. package/dist/src/orgrt/cross-org.d.ts.map +1 -1
  168. package/dist/src/orgrt/cross-org.js +14 -0
  169. package/dist/src/orgrt/cross-org.js.map +1 -1
  170. package/dist/src/orgrt/daemon.d.ts +12 -1
  171. package/dist/src/orgrt/daemon.d.ts.map +1 -1
  172. package/dist/src/orgrt/daemon.js +216 -49
  173. package/dist/src/orgrt/daemon.js.map +1 -1
  174. package/dist/src/orgrt/decisions.d.ts +14 -2
  175. package/dist/src/orgrt/decisions.d.ts.map +1 -1
  176. package/dist/src/orgrt/decisions.js +216 -10
  177. package/dist/src/orgrt/decisions.js.map +1 -1
  178. package/dist/src/orgrt/file-roots.d.ts +7 -0
  179. package/dist/src/orgrt/file-roots.d.ts.map +1 -1
  180. package/dist/src/orgrt/file-roots.js +38 -1
  181. package/dist/src/orgrt/file-roots.js.map +1 -1
  182. package/dist/src/orgrt/forwarder.d.ts.map +1 -1
  183. package/dist/src/orgrt/forwarder.js +16 -2
  184. package/dist/src/orgrt/forwarder.js.map +1 -1
  185. package/dist/src/orgrt/idle-deadline.d.ts +74 -1
  186. package/dist/src/orgrt/idle-deadline.d.ts.map +1 -1
  187. package/dist/src/orgrt/idle-deadline.js +52 -2
  188. package/dist/src/orgrt/idle-deadline.js.map +1 -1
  189. package/dist/src/orgrt/loadouts.d.ts +38 -0
  190. package/dist/src/orgrt/loadouts.d.ts.map +1 -0
  191. package/dist/src/orgrt/loadouts.js +129 -0
  192. package/dist/src/orgrt/loadouts.js.map +1 -0
  193. package/dist/src/orgrt/mailbox.d.ts +17 -1
  194. package/dist/src/orgrt/mailbox.d.ts.map +1 -1
  195. package/dist/src/orgrt/mailbox.js +50 -3
  196. package/dist/src/orgrt/mailbox.js.map +1 -1
  197. package/dist/src/orgrt/policy.d.ts +42 -2
  198. package/dist/src/orgrt/policy.d.ts.map +1 -1
  199. package/dist/src/orgrt/policy.js +53 -8
  200. package/dist/src/orgrt/policy.js.map +1 -1
  201. package/dist/src/orgrt/prompt-vars.d.ts +11 -0
  202. package/dist/src/orgrt/prompt-vars.d.ts.map +1 -0
  203. package/dist/src/orgrt/prompt-vars.js +49 -0
  204. package/dist/src/orgrt/prompt-vars.js.map +1 -0
  205. package/dist/src/orgrt/questions.d.ts +22 -1
  206. package/dist/src/orgrt/questions.d.ts.map +1 -1
  207. package/dist/src/orgrt/questions.js +24 -3
  208. package/dist/src/orgrt/questions.js.map +1 -1
  209. package/dist/src/orgrt/review-packet.d.ts +29 -0
  210. package/dist/src/orgrt/review-packet.d.ts.map +1 -0
  211. package/dist/src/orgrt/review-packet.js +67 -0
  212. package/dist/src/orgrt/review-packet.js.map +1 -0
  213. package/dist/src/orgrt/role-sandbox.d.ts.map +1 -1
  214. package/dist/src/orgrt/role-sandbox.js +16 -3
  215. package/dist/src/orgrt/role-sandbox.js.map +1 -1
  216. package/dist/src/orgrt/session-ledger.d.ts +68 -0
  217. package/dist/src/orgrt/session-ledger.d.ts.map +1 -0
  218. package/dist/src/orgrt/session-ledger.js +128 -0
  219. package/dist/src/orgrt/session-ledger.js.map +1 -0
  220. package/dist/src/orgrt/session.d.ts +45 -5
  221. package/dist/src/orgrt/session.d.ts.map +1 -1
  222. package/dist/src/orgrt/session.js +397 -27
  223. package/dist/src/orgrt/session.js.map +1 -1
  224. package/dist/src/orgrt/task-dag.d.ts +34 -1
  225. package/dist/src/orgrt/task-dag.d.ts.map +1 -1
  226. package/dist/src/orgrt/task-dag.js +58 -7
  227. package/dist/src/orgrt/task-dag.js.map +1 -1
  228. package/dist/src/orgrt/tool-spill.d.ts +94 -0
  229. package/dist/src/orgrt/tool-spill.d.ts.map +1 -0
  230. package/dist/src/orgrt/tool-spill.js +180 -0
  231. package/dist/src/orgrt/tool-spill.js.map +1 -0
  232. package/dist/src/orgrt/types.d.ts +138 -0
  233. package/dist/src/orgrt/types.d.ts.map +1 -1
  234. package/dist/src/orgrt/types.js +148 -0
  235. package/dist/src/orgrt/types.js.map +1 -1
  236. package/dist/src/platform-adapters/renderers/mcp.d.ts +20 -2
  237. package/dist/src/platform-adapters/renderers/mcp.d.ts.map +1 -1
  238. package/dist/src/platform-adapters/renderers/mcp.js +25 -5
  239. package/dist/src/platform-adapters/renderers/mcp.js.map +1 -1
  240. package/dist/src/protocol-capabilities.d.ts +2 -1
  241. package/dist/src/protocol-capabilities.d.ts.map +1 -1
  242. package/dist/src/protocol-capabilities.js +2 -1
  243. package/dist/src/protocol-capabilities.js.map +1 -1
  244. package/dist/src/ui/dashboard.html +450 -164
  245. package/dist/src/ui/org-hil.mjs +276 -0
  246. package/dist/src/ui/org-runtime.mjs +350 -0
  247. package/dist/src/ui/orgs.html +15 -4
  248. package/dist/src/ui/routes-org.mjs +290 -541
  249. package/dist/src/ui/server.mjs +25 -5
  250. package/dist/tsconfig.tsbuildinfo +1 -1
  251. package/package.json +9 -8
@@ -12,10 +12,15 @@ import { StateDetector } from './state-detector.js';
12
12
  * "boss appears hung". */
13
13
  const SILENT_SESSION_MS = 4 * 60_000;
14
14
  const CONTEXT_LIMIT_RE = /context.window.limit|context.length.exceeded|maximum.context/i;
15
+ import { createHash } from 'node:crypto';
15
16
  import { readFileSync } from 'node:fs';
17
+ import { join } from 'node:path';
18
+ import { resolveRoleCostTier } from './cost-tier.js';
19
+ import { expandRolePromptVars, promptVarsFor } from './prompt-vars.js';
16
20
  import { resolveProviderEnv, resolveRoleProvider } from './provider.js';
17
21
  import { resolveRoleGitEnforcement } from './role-sandbox.js';
18
22
  import { loadBuiltinRoleSkill } from './role-skills.js';
23
+ import { mailRouteKey, ROLE_SESSION_KEY, resolveSessionScope, SessionLedger, } from './session-ledger.js';
19
24
  import { DEFAULT_CLAUDE_MODEL, VERCEL_PROVIDERS } from './vercel-providers.js';
20
25
  /**
21
26
  * Resolves the extra system-prompt block for a role: built-in archetype
@@ -189,6 +194,14 @@ endpointBriefing) {
189
194
  .filter(Boolean)
190
195
  .join('\n\n');
191
196
  }
197
+ /** The system prompt one session of this role is built with. */
198
+ function rolePromptFor(opts) {
199
+ return buildRolePrompt(expandRolePromptVars(opts.role, promptVarsFor(opts.orgRoot ?? opts.cwd)), (opts.def ?? { name: opts.org, goal: '' }), opts.def?.roles.map((r) => r.id) ?? [opts.role.id], opts.glossary,
200
+ // D7: the loadout's text follows the role's own guidance. With no
201
+ // loadout this is exactly resolveRoleExtraGuidance(role), as before.
202
+ [resolveRoleExtraGuidance(opts.role), opts.loadout?.guidance].filter(Boolean).join('\n\n') ||
203
+ undefined, opts.onComplete ? endpointBriefingLines(opts.def) : undefined);
204
+ }
192
205
  /**
193
206
  * Runs a role for the life of the org, transparently restarting the
194
207
  * underlying SDK session whenever it ends on its own (`maxTurns` reached)
@@ -248,23 +261,151 @@ async function runAgentSessionLoop(opts) {
248
261
  // SDK session id here (it must outlive individual runOneSession calls,
249
262
  // since resume continues the same billing session) and emit only deltas.
250
263
  const sessionCostTotals = new Map();
264
+ // ADR-O001 D1: modelUsage is cumulative per session exactly like
265
+ // total_cost_usd, so it needs the same prev-value map to become a delta.
266
+ const sessionTokenTotals = new Map();
267
+ // ADR-O001 D3. 'role' scope (the default) keeps the pre-D3 loop exactly: one
268
+ // model session for the role's life, resumed across maxTurns restarts via
269
+ // resumeSessionId. 'task' scope keys model sessions by the task a message
270
+ // belongs to: the process exits at a task boundary (or on idle) and the next
271
+ // one resumes that task's session from the ledger. Either way every session
272
+ // run is recorded with its session id before and after.
273
+ const scope = resolveSessionScope(opts.role, opts.def);
274
+ const idleExitMs = opts.def?.run_config
275
+ ?.session_idle_exit_ms;
276
+ const ledger = opts.sessionLedger ?? new SessionLedger();
277
+ const runtimeKey = opts.role.runtime ?? opts.def?.runtime ?? 'claude';
278
+ let taskKey = ROLE_SESSION_KEY;
279
+ // Why the next fresh session for a key is fresh, when the loop itself threw
280
+ // the record away (stale resume, turn-limit error) — recorded, not guessed.
281
+ const droppedBecause = new Map();
282
+ const staleTried = new Set();
283
+ // The options THIS session is built from — the role's own, except that a
284
+ // task-scoped session carries its task's loadout (D7).
285
+ let sessionOpts = opts;
286
+ let promptHash = '*';
287
+ // Task scope: which task last wrote to each correspondent, so their
288
+ // untagged reply goes back to that task's session (see mailRouteKey).
289
+ const correspondents = new Map();
251
290
  // Always run at least once: a mailbox can be closed with queued items still
252
291
  // pending (stream() drains the queue before honoring `closed`), which is a
253
292
  // normal, valid starting state - checking isClosed before the first run
254
293
  // would skip that drain entirely.
255
294
  while (true) {
295
+ // Opt-in only: keep the process DOWN until there is mail, instead of
296
+ // starting a query() that parks on an empty mailbox. waitForMessage()
297
+ // still returns true for a closed mailbox with queued items.
298
+ if (scope !== 'role' || idleExitMs !== undefined) {
299
+ if (!(await mailbox.waitForMessage()))
300
+ return;
301
+ }
302
+ let startReason;
303
+ if (scope === 'cold') {
304
+ // D6: nothing carries over — not a checkpointed session, not the last
305
+ // message's. Paying the cache miss here is the point.
306
+ resumeSessionId = undefined;
307
+ startReason = 'fresh-cold';
308
+ }
309
+ else if (scope === 'task') {
310
+ // An untagged message (mail, an answer, a continuation) belongs to the
311
+ // session the role is already in.
312
+ taskKey = mailRouteKey(mailbox.peek() ?? '', correspondents) ?? taskKey;
313
+ const key = taskKey;
314
+ sessionOpts = {
315
+ ...opts,
316
+ // D7: this task's session is built with this task's loadout.
317
+ ...(key !== ROLE_SESSION_KEY && opts.loadoutFor ? { loadout: opts.loadoutFor(key) } : {}),
318
+ // Mail sent from a task's session names the task in its subject, so
319
+ // the reader knows what it is about and a reply can find its way back.
320
+ deliver: (from, to, subject, body) => {
321
+ if (key === ROLE_SESSION_KEY)
322
+ return opts.deliver(from, to, subject, body);
323
+ correspondents.set(to, key);
324
+ const tagged = subject.includes('[task:') ? subject : `[task:${key}] ${subject}`;
325
+ return opts.deliver(from, to, tagged, body);
326
+ },
327
+ };
328
+ promptHash = createHash('sha256')
329
+ .update(rolePromptFor(sessionOpts))
330
+ .digest('hex')
331
+ .slice(0, 16);
332
+ const pick = ledger.resumeFor({
333
+ role: opts.role.id,
334
+ runtime: runtimeKey,
335
+ taskKey,
336
+ cwd: opts.cwd,
337
+ promptHash,
338
+ });
339
+ resumeSessionId = pick.sessionId;
340
+ startReason =
341
+ pick.reason === 'fresh-no-record'
342
+ ? (droppedBecause.get(taskKey) ?? pick.reason)
343
+ : pick.reason;
344
+ }
345
+ else {
346
+ startReason = resumeSessionId ? 'resumed' : 'fresh-no-record';
347
+ }
348
+ const sessionKey = taskKey;
349
+ const streamOpts = scope === 'cold'
350
+ ? { stopBefore: () => true, idleExitMs }
351
+ : scope === 'task'
352
+ ? {
353
+ stopBefore: (next) => {
354
+ const k = mailRouteKey(next, correspondents);
355
+ return k !== undefined && k !== sessionKey;
356
+ },
357
+ idleExitMs,
358
+ }
359
+ : idleExitMs !== undefined
360
+ ? { idleExitMs }
361
+ : undefined;
362
+ const sessionIdBefore = resumeSessionId;
363
+ const startedAt = Date.now();
364
+ const recordRun = (after, error) => {
365
+ const run = ledger.recordRun({
366
+ role: opts.role.id,
367
+ runtime: runtimeKey,
368
+ taskKey: sessionKey,
369
+ sessionIdBefore,
370
+ sessionIdAfter: after,
371
+ reason: startReason,
372
+ startedAt,
373
+ endedAt: Date.now(),
374
+ ...(error ? { error } : {}),
375
+ });
376
+ opts.bus.emit({
377
+ type: 'audit',
378
+ from: opts.role.id,
379
+ reason: 'session-run',
380
+ msg: `session ${run.resumed ? 'resumed' : 'started fresh'} (${startReason}) for ${sessionKey}`,
381
+ data: run,
382
+ });
383
+ };
256
384
  const realBefore = mailbox.consumedRealCount;
257
385
  let sessionId;
258
386
  let hitTurnLimit = false;
259
387
  const attempt = { replied: false };
260
388
  try {
261
- const res = await runOneSession(opts, resumeSessionId, sessionCostTotals, attempt);
389
+ const res = await runOneSession(sessionOpts, resumeSessionId, sessionCostTotals, attempt, sessionTokenTotals, streamOpts);
262
390
  sessionId = res.sessionId;
263
391
  hitTurnLimit = res.hitTurnLimit;
264
392
  resumeSessionId = sessionId;
393
+ recordRun(sessionId);
394
+ if (scope === 'task' && sessionId) {
395
+ droppedBecause.delete(sessionKey);
396
+ ledger.set({
397
+ role: opts.role.id,
398
+ runtime: runtimeKey,
399
+ taskKey: sessionKey,
400
+ cwd: opts.cwd,
401
+ promptHash,
402
+ sessionId,
403
+ });
404
+ }
265
405
  }
266
406
  catch (err) {
267
407
  const errMsg = err instanceof Error ? err.message : String(err);
408
+ recordRun(undefined, errMsg);
268
409
  // #304 (review round 3): an org's own stop aborts whatever this attempt
269
410
  // was doing — max-turns and stale-resume below are diagnoses for a
270
411
  // genuinely failed attempt, not for one the org itself just cut off.
@@ -285,8 +426,36 @@ async function runAgentSessionLoop(opts) {
285
426
  sessionId = undefined;
286
427
  resumeSessionId = undefined;
287
428
  hitTurnLimit = true;
429
+ if (scope === 'task') {
430
+ ledger.drop({ role: opts.role.id, runtime: runtimeKey, taskKey: sessionKey });
431
+ droppedBecause.set(sessionKey, 'fresh-after-turn-limit');
432
+ }
433
+ }
434
+ else if (!stopping &&
435
+ scope === 'task' &&
436
+ sessionIdBefore !== undefined &&
437
+ !staleTried.has(sessionKey) &&
438
+ !attempt.replied) {
439
+ // D3's per-key form of #149 below: a recorded session that fails
440
+ // before replying is treated as expired — forget it and retry that
441
+ // task fresh once. Anything it had already pulled goes back first.
442
+ staleTried.add(sessionKey);
443
+ ledger.drop({ role: opts.role.id, runtime: runtimeKey, taskKey: sessionKey });
444
+ droppedBecause.set(sessionKey, 'fresh-after-stale-resume');
445
+ mailbox.reclaimInFlight();
446
+ sessionId = undefined;
447
+ resumeSessionId = undefined;
448
+ hitTurnLimit = false;
449
+ opts.bus.emit({
450
+ type: 'status',
451
+ from: opts.role.id,
452
+ reason: 'resume-session-stale',
453
+ msg: `agent "${opts.role.id}" could not resume its session for ${sessionKey} — retrying with a fresh session`,
454
+ data: { error: errMsg },
455
+ });
288
456
  }
289
457
  else if (!stopping &&
458
+ scope === 'role' &&
290
459
  resumeSessionId &&
291
460
  resumeSessionId === initialResumeSessionId &&
292
461
  !triedFreshAfterResumeFailure &&
@@ -357,6 +526,17 @@ async function runAgentSessionLoop(opts) {
357
526
  });
358
527
  }
359
528
  }
529
+ else if (mailbox.lastStreamEnd) {
530
+ // D3: the process ended on purpose — a task boundary or idle — and the
531
+ // model session is kept for the next wake.
532
+ opts.bus.emit({
533
+ type: 'status',
534
+ from: opts.role.id,
535
+ reason: 'session-cycled',
536
+ msg: `process cycled (${mailbox.lastStreamEnd}); model session kept for ${sessionKey}`,
537
+ data: { taskKey: sessionKey, end: mailbox.lastStreamEnd, sessionId },
538
+ });
539
+ }
360
540
  else {
361
541
  opts.bus.emit({
362
542
  type: 'status',
@@ -366,11 +546,68 @@ async function runAgentSessionLoop(opts) {
366
546
  }
367
547
  }
368
548
  }
549
+ /** ADR-O001 D1 — token-metering helpers.
550
+ *
551
+ * `cache_read_input_tokens` and `cache_creation_input_tokens` are siblings
552
+ * of `input_tokens` in the Anthropic API, not subsets of it, and both are
553
+ * billable. Everything below therefore sums all four. */
554
+ function totalTokens(u) {
555
+ return u.input + u.output + u.cacheRead + u.cacheCreation;
556
+ }
557
+ function addTo(target, add) {
558
+ target.input += add.input;
559
+ target.output += add.output;
560
+ target.cacheRead += add.cacheRead;
561
+ target.cacheCreation += add.cacheCreation;
562
+ }
563
+ /** One model turn's own usage, off an 'assistant' (or per-turn 'result')
564
+ * message. */
565
+ function turnBreakdown(m) {
566
+ return {
567
+ input: m.input_tokens ?? 0,
568
+ output: m.output_tokens ?? 0,
569
+ cacheRead: m.cache_read_input_tokens ?? 0,
570
+ cacheCreation: m.cache_creation_input_tokens ?? 0,
571
+ };
572
+ }
573
+ /** What a 'result' message says this mailbox message consumed.
574
+ *
575
+ * When the runner reports `cumulative_tokens` (the Claude SDK's whole-pipeline
576
+ * `modelUsage`, which unlike `usage` includes Task subagents and sidechains),
577
+ * that value is CUMULATIVE per session — the same lifecycle as
578
+ * `total_cost_usd` — so it is converted to a delta against the previous value
579
+ * for the same session_id. A fresh/restarted session has no prior entry and
580
+ * correctly yields its full value; a value that ticks down (a provider-side
581
+ * correction) floors at 0 rather than re-adding the whole cumulative total.
582
+ * Without `cumulative_tokens` the per-turn fields are used as before. */
583
+ function resultBreakdown(m, tokenTotals, sid) {
584
+ const cum = m.cumulative_tokens;
585
+ if (!cum)
586
+ return turnBreakdown(m);
587
+ const now = {
588
+ input: cum.input,
589
+ output: cum.output,
590
+ cacheRead: cum.cache_read,
591
+ cacheCreation: cum.cache_creation,
592
+ };
593
+ if (!tokenTotals)
594
+ return now;
595
+ const prev = tokenTotals.get(sid);
596
+ tokenTotals.set(sid, now);
597
+ if (!prev)
598
+ return now;
599
+ return {
600
+ input: Math.max(0, now.input - prev.input),
601
+ output: Math.max(0, now.output - prev.output),
602
+ cacheRead: Math.max(0, now.cacheRead - prev.cacheRead),
603
+ cacheCreation: Math.max(0, now.cacheCreation - prev.cacheCreation),
604
+ };
605
+ }
369
606
  /** One bounded SDK session for a role; resolves with the SDK's session_id (for
370
607
  * resuming on restart) and whether it ended by hitting the turn limit (so the
371
608
  * caller can push a continuation) when the stream ends (mailbox closed or
372
609
  * maxTurns reached). */
373
- async function runOneSession(opts, resume, costTotals, progress) {
610
+ async function runOneSession(opts, resume, costTotals, progress, tokenTotals, streamOpts) {
374
611
  const { org, role, bus, policy, mailbox, cwd } = opts;
375
612
  // Read lastMessageId live from opts instead of capturing at session start
376
613
  // This ensures chat responses link to the most recent message delivered
@@ -389,7 +626,24 @@ async function runOneSession(opts, resume, costTotals, progress) {
389
626
  // The named provider's default model fills in adapter_config.model when the
390
627
  // role didn't pin one.
391
628
  const prov = resolveRoleProvider(role, opts.orgRoot ?? opts.cwd);
629
+ // ADR-O001 D8: the role's cost tier, when the org declares one. Resolved
630
+ // here — the single choke point where a role's model is decided — so the
631
+ // documented precedence holds in exactly one place:
632
+ // explicit adapter_config.model > tier > named-provider default > runtime
633
+ // The tier's EFFORT is applied even when the model came from an explicit
634
+ // pin: which model to run and how hard to think are separate axes, and
635
+ // silently dropping the effort because a model was pinned would be the
636
+ // "silent downgrade" this decision exists to prevent.
637
+ // Throws (fails the session) rather than guessing when the tier has no
638
+ // entry for this role's provider — daemon.ts validates the whole roster
639
+ // up front so that is normally caught before any token is spent.
640
+ const tier = resolveRoleCostTier({
641
+ role,
642
+ def: opts.def,
643
+ vendor: role.provider?.vendor ?? prov.cfg?.vendor,
644
+ });
392
645
  const model = role.adapter_config?.model ??
646
+ tier?.model ??
393
647
  prov.defaultModel ??
394
648
  resolveModel(role, role.runtime, role.provider?.vendor ?? prov.cfg?.vendor);
395
649
  bus.emit({ type: 'status', from: role.id, msg: 'session starting' });
@@ -401,7 +655,7 @@ async function runOneSession(opts, resume, costTotals, progress) {
401
655
  // time a 'result' message ends one mailbox message and the next one starts.
402
656
  // Exists purely so the 'result' branch never re-adds what this branch
403
657
  // already added (see there for why it can't just always add).
404
- let messageTurnTokens = 0;
658
+ let messageTurnTokens = { input: 0, output: 0, cacheRead: 0, cacheCreation: 0 };
405
659
  // Abort hook for the runner (AgentRunArgs.signal): the silent-stream
406
660
  // abort below used to call iterator.return() only, which queues behind a
407
661
  // subprocess runner blocked in `for await (child.stdout)` — the child was
@@ -438,12 +692,18 @@ async function runOneSession(opts, resume, costTotals, progress) {
438
692
  });
439
693
  const stream = runner.run({
440
694
  tools,
441
- prompt: mailbox.stream(),
442
- systemPrompt: buildRolePrompt(role, (opts.def ?? { name: org, goal: '' }), opts.def?.roles.map((r) => r.id) ?? [role.id], opts.glossary, resolveRoleExtraGuidance(role), opts.onComplete ? endpointBriefingLines(opts.def) : undefined),
695
+ // No options = the pre-D3 stream, exactly.
696
+ prompt: streamOpts ? mailbox.stream('', streamOpts) : mailbox.stream(),
697
+ systemPrompt: rolePromptFor(opts),
443
698
  model,
444
699
  cwd,
700
+ effort: tier?.effort,
445
701
  env: {
446
702
  ...resolveProviderEnv(prov.cfg),
703
+ // D8: how a NON-Claude provider expresses the tier's effort level.
704
+ // Empty for Claude (handled natively by ClaudeAgentRunner) and for a
705
+ // provider that declares no mechanism — which simply ignores effort.
706
+ ...(tier?.env ?? {}),
447
707
  // Custom-endpoint providers (named-provider path): pin the engine's
448
708
  // model env so background/haiku tasks also route to the endpoint's
449
709
  // model instead of erroring on an Anthropic-only default.
@@ -475,6 +735,13 @@ async function runOneSession(opts, resume, costTotals, progress) {
475
735
  maxTurns: opts.maxTurns ?? 30,
476
736
  resume,
477
737
  claudeRestrictions: gitEnforcement.claudeRestrictions,
738
+ // ADR-O001 D2: tool results are 76% of a role's context mass and nothing
739
+ // bounded them. Under the ORG STATE dir (never the workspace cwd, which
740
+ // may be the repo), and under orgRoot — which file-roots.ts already
741
+ // makes readable to the role's file tools and role-sandbox.ts already
742
+ // makes readable to Bash — so the path in the digest actually resolves
743
+ // when the role decides it needs the full text.
744
+ toolSpillDir: join(opts.orgDir ?? opts.cwd, 'tool-results', role.id.replace(/[^a-zA-Z0-9_.-]/g, '_')),
478
745
  canUseTool: gatedCanUseTool(policy, opts.beforeTool, role.id, opts.fence, opts.onDecision
479
746
  ? (toolName, _input, decision, kind) => opts.onDecision?.(role.id, toolName, decision.message ?? 'denied', kind)
480
747
  : undefined, opts.hasPendingGate),
@@ -624,10 +891,18 @@ async function runOneSession(opts, resume, costTotals, progress) {
624
891
  // .d.ts, which puts `usage`/token counts on BetaMessage but cost only
625
892
  // on SDKResultSuccess.total_cost_usd/modelUsage. overBudgetUsd is
626
893
  // still checked below, once per message, same as before this fix.)
627
- const turnTokens = (m.input_tokens ?? 0) + (m.output_tokens ?? 0);
894
+ //
895
+ // ADR-O001 D1: the sum must include BOTH cache fields. They are
896
+ // siblings of input_tokens in the Anthropic API, not subsets of it —
897
+ // `input_tokens` is the uncached remainder — and both are billable
898
+ // (~0.1x and ~1.25x input). Omitting them meant the better the cache
899
+ // worked the less the meter saw: on one measured run, 2,765M tokens
900
+ // billed against 8.1M recorded, with input_tokens at 0.0M.
901
+ const turn = turnBreakdown(m);
902
+ const turnTokens = totalTokens(turn);
628
903
  if (turnTokens > 0) {
629
- messageTurnTokens += turnTokens;
630
- policy.addUsage(turnTokens);
904
+ addTo(messageTurnTokens, turn);
905
+ policy.addTokenUsage(turn);
631
906
  if (policy.overBudget) {
632
907
  bus.emit({
633
908
  type: 'status',
@@ -664,20 +939,48 @@ async function runOneSession(opts, resume, costTotals, progress) {
664
939
  });
665
940
  }
666
941
  else if (m.type === 'result') {
667
- const tokens = (m.input_tokens ?? 0) + (m.output_tokens ?? 0);
942
+ // ADR-O001 D1: prefer the SDK's `modelUsage` over `usage`. The SDK
943
+ // documents `usage` as "MAIN AGENT LOOP ONLY — excludes Task
944
+ // subagent, sidechain, and auxiliary model calls ... Prefer
945
+ // modelUsage for token/cost accounting"; the measured run made 46
946
+ // subagent calls this counter never saw. modelUsage is CUMULATIVE per
947
+ // session (same lifecycle as total_cost_usd, per its own type doc),
948
+ // so it is converted to a delta here rather than added, exactly as
949
+ // cost is below. A runner that reports no modelUsage falls back to
950
+ // the per-turn `usage` fields, which keep their old semantics.
951
+ const resultTokens = resultBreakdown(m, tokenTotals, m.session_id ?? sessionId ?? '');
668
952
  // Per the SDK's own type docs, a 'result' message's usage is that
669
953
  // message's own (effectively last-turn) usage in streaming-input mode,
670
954
  // NOT a cumulative total across every turn of the mailbox message —
671
955
  // and that last turn was already counted above via its own 'assistant'
672
956
  // message, specifically so overBudget could trip mid-message. Adding
673
- // `tokens` again here unconditionally would double-count it. Only make
957
+ // the result's own usage again unconditionally would double-count it.
958
+ // (A modelUsage-derived delta is per-session-cumulative, so the same
959
+ // subtraction is exactly right there too: it removes what the
960
+ // assistant turns of THIS message already contributed and leaves the
961
+ // subagent/auxiliary volume the main loop never reported.) Only make
674
962
  // up the shortfall (never negative) so a turn whose usage somehow
675
963
  // never reached the 'assistant' branch (e.g. a runner/test double that
676
964
  // doesn't emit per-turn usage) still gets counted at least once.
677
- const shortfall = Math.max(0, tokens - messageTurnTokens);
678
- if (shortfall > 0)
679
- policy.addUsage(shortfall);
680
- messageTurnTokens = 0;
965
+ const shortfall = {
966
+ input: Math.max(0, resultTokens.input - messageTurnTokens.input),
967
+ output: Math.max(0, resultTokens.output - messageTurnTokens.output),
968
+ cacheRead: Math.max(0, resultTokens.cacheRead - messageTurnTokens.cacheRead),
969
+ cacheCreation: Math.max(0, resultTokens.cacheCreation - messageTurnTokens.cacheCreation),
970
+ };
971
+ if (totalTokens(shortfall) > 0)
972
+ policy.addTokenUsage(shortfall);
973
+ // What this whole mailbox message actually added to the meter: the
974
+ // per-turn accounting above plus whatever the result topped up. This
975
+ // is what the 'usage' event reports, so a consumer summing events
976
+ // lands on the same number as policy.usage.
977
+ const messageTokens = {
978
+ input: messageTurnTokens.input + shortfall.input,
979
+ output: messageTurnTokens.output + shortfall.output,
980
+ cacheRead: messageTurnTokens.cacheRead + shortfall.cacheRead,
981
+ cacheCreation: messageTurnTokens.cacheCreation + shortfall.cacheCreation,
982
+ };
983
+ messageTurnTokens = { input: 0, output: 0, cacheRead: 0, cacheCreation: 0 };
681
984
  // Convert the SDK's cumulative-per-session total_cost_usd into a
682
985
  // per-result delta before emitting - downstream sums usage events.
683
986
  // costTotals is keyed by session_id, so a genuinely new/restarted
@@ -701,7 +1004,19 @@ async function runOneSession(opts, resume, costTotals, progress) {
701
1004
  bus.emit({
702
1005
  type: 'usage',
703
1006
  from: role.id,
704
- data: { tokens, cost_usd: costDelta, subtype: m.subtype },
1007
+ // ADR-O001 D1: the four quantities travel separately so every
1008
+ // downstream consumer (forwarder → dashboard state.json, reporting,
1009
+ // `org costs`) can record real values instead of the 0s they used
1010
+ // to persist. `tokens` stays the single billable total.
1011
+ data: {
1012
+ tokens: totalTokens(messageTokens),
1013
+ cost_usd: costDelta,
1014
+ subtype: m.subtype,
1015
+ tokens_in: messageTokens.input,
1016
+ tokens_out: messageTokens.output,
1017
+ cache_read: messageTokens.cacheRead,
1018
+ cache_creation: messageTokens.cacheCreation,
1019
+ },
705
1020
  });
706
1021
  if (m.subtype && m.subtype !== 'success') {
707
1022
  if (m.subtype === 'error_max_turns')
@@ -787,6 +1102,13 @@ async function runOneSession(opts, resume, costTotals, progress) {
787
1102
  providerSet?.close();
788
1103
  }
789
1104
  }
1105
+ /** ADR-O001 D7: org_task's description suffix for an org with a catalog. */
1106
+ function loadoutHelp(catalog) {
1107
+ const list = catalog
1108
+ .map((l) => (l.description ? `${l.name} (${l.description})` : l.name))
1109
+ .join(', ');
1110
+ return ` Optionally select a "loadout" — the named, stable specialisation the assignee's session is built with: ${list}. Select by kind of work; put everything specific to this task (which diff, criteria, what failed last time) in the title or a message, not in the choice of loadout. The selection is recorded on the task and reused on every retry.`;
1111
+ }
790
1112
  /** Build the org tool surface as platform-agnostic OrgToolDef[]. The handlers
791
1113
  * close over sessionOpts callbacks (deliver, recall, remember, …) — same
792
1114
  * wiring as the previous inline createSdkMcpServer block, just decoupled from
@@ -912,22 +1234,64 @@ export function buildOrgTools(opts) {
912
1234
  handler: async (args) => text(await onGate(role.id, args.name, args.description)),
913
1235
  });
914
1236
  }
1237
+ // ADR-O001 D7: the `loadout` argument exists only for an org with a
1238
+ // catalog, so every other org's tool list stays byte-identical.
1239
+ const catalog = opts.loadoutCatalog?.length ? opts.loadoutCatalog : undefined;
1240
+ const loadoutArg = catalog
1241
+ ? { loadout: z.enum(catalog.map((l) => l.name)).optional() }
1242
+ : {};
915
1243
  const createTask = opts.createTask;
916
1244
  if (createTask) {
917
1245
  tools.push({
918
1246
  name: 'org_task',
919
- description: 'Create a task in the DAG with optional dependencies. Dependencies must be existing task IDs. Tasks become ready when all deps are done, then get dispatched to the assignee.',
920
- schema: { title: z.string(), assignee: z.string(), deps: z.array(z.string()).default([]) },
921
- handler: async (args) => text(createTask(role.id, args.title, args.assignee, args.deps ?? [])),
1247
+ description: 'Create a task in the DAG with optional dependencies. Dependencies must be existing task IDs. Tasks become ready when all deps are done, then get dispatched to the assignee.' +
1248
+ (catalog ? loadoutHelp(catalog) : ''),
1249
+ schema: {
1250
+ title: z.string(),
1251
+ assignee: z.string(),
1252
+ deps: z.array(z.string()).default([]),
1253
+ ...loadoutArg,
1254
+ },
1255
+ handler: async (args) => text(createTask(role.id, args.title, args.assignee, args.deps ?? [], args.loadout)),
922
1256
  });
923
1257
  }
924
1258
  const completeTask = opts.completeTask;
925
1259
  if (completeTask) {
926
1260
  tools.push({
927
1261
  name: 'org_task_done',
928
- description: 'Mark a task as completed and optionally provide a result summary. Any downstream tasks whose deps are now all done will become ready and be dispatched.',
929
- schema: { taskId: z.string(), result: z.string().optional() },
930
- handler: async (args) => text(completeTask(role.id, args.taskId, args.result)),
1262
+ description: opts.requireTaskEvidence
1263
+ ? 'Mark a task as completed. This org requires EVIDENCE (run_config.completion_evidence): pass `evidence` with the current commit sha and one entry per acceptance criterion — the command you actually ran, its real exit code, and its output. Evidence pinned to an older commit is stale and will be refused, and a refused completion puts the task back in your queue with the reason — but only up to run_config.max_evidence_attempts times (default 3), after which the task is recorded as failed and escalated to the boss instead of returned to you. Any downstream tasks whose deps are now all done become ready and are dispatched.'
1264
+ : 'Mark a task as completed and optionally provide a result summary. Any downstream tasks whose deps are now all done will become ready and be dispatched.',
1265
+ schema: {
1266
+ taskId: z.string(),
1267
+ result: z.string().optional(),
1268
+ evidence: z
1269
+ .object({
1270
+ headSha: z.string(),
1271
+ checks: z
1272
+ .array(z.object({
1273
+ command: z.string(),
1274
+ exitCode: z.number().int(),
1275
+ output: z.string().optional(),
1276
+ }))
1277
+ .default([]),
1278
+ })
1279
+ .optional(),
1280
+ },
1281
+ handler: async (args) => text(completeTask(role.id, args.taskId, args.result, args.evidence)),
1282
+ });
1283
+ }
1284
+ const requestReview = opts.requestReview;
1285
+ if (requestReview) {
1286
+ tools.push({
1287
+ name: 'org_review',
1288
+ description: "Ask an artifact-only reviewer for a verdict on a task. You pass ids only: the runtime builds the review from the task's text, its assignee's latest org_task_done evidence (commands, exit codes, output) and its own git diff of base...headSha — nothing you write is added, so there is no summary to give. The reviewer starts cold every time and replies to you with org_send. Refused if the task has no evidence yet.",
1289
+ schema: {
1290
+ taskId: z.string(),
1291
+ reviewer: z.string(),
1292
+ base: z.string().optional().describe("git ref to diff against (default 'main')"),
1293
+ },
1294
+ handler: async (args) => text(requestReview(role.id, args.taskId, args.reviewer, args.base)),
931
1295
  });
932
1296
  }
933
1297
  const listTasks = opts.listTasks;
@@ -982,7 +1346,8 @@ export function buildOrgTools(opts) {
982
1346
  if (planGraph) {
983
1347
  tools.push({
984
1348
  name: 'org_plan_graph',
985
- description: 'Propose a full work graph in one call. Each task spec uses a local "name" and references other specs by name in "after".',
1349
+ description: 'Propose a full work graph in one call. Each task spec uses a local "name" and references other specs by name in "after".' +
1350
+ (catalog ? ' Each spec may select a "loadout" exactly as org_task does.' : ''),
986
1351
  schema: {
987
1352
  tasks: z
988
1353
  .array(z.object({
@@ -990,11 +1355,11 @@ export function buildOrgTools(opts) {
990
1355
  title: z.string(),
991
1356
  assignee: z.string(),
992
1357
  after: z.array(z.string()).default([]),
1358
+ ...loadoutArg,
993
1359
  }))
994
1360
  .min(1),
995
1361
  },
996
- handler: async (args) => text(planGraph(role.id, args.tasks ??
997
- [])),
1362
+ handler: async (args) => text(planGraph(role.id, args.tasks ?? [])),
998
1363
  });
999
1364
  }
1000
1365
  tools.push({
@@ -1015,12 +1380,17 @@ export function buildOrgTools(opts) {
1015
1380
  });
1016
1381
  tools.push({
1017
1382
  name: 'ask_human',
1018
- description: 'Ask a human a free-form question and pause for their answer. Use only when you genuinely need human judgment.',
1019
- schema: { question: z.string() },
1383
+ description: 'Ask a human a free-form question. Use only when you genuinely need human judgment. ' +
1384
+ 'Set blocking: true ONLY if you cannot continue until it is answered — a blocking question pauses the ' +
1385
+ "org's idle watchdog (for up to an hour; after that the run resumes its normal idle checks either way). " +
1386
+ 'If you can keep working while you wait — an FYI, a preference, anything you would describe as "not blocking on this" — ' +
1387
+ 'pass blocking: false and carry on; the question is still recorded and answered, it just does not freeze the run. ' +
1388
+ 'Defaults to blocking.',
1389
+ schema: { question: z.string(), blocking: z.boolean().optional() },
1020
1390
  handler: async (args) => {
1021
1391
  if (!opts.askHuman)
1022
1392
  return text('ask_human is not available in this session');
1023
- const receipt = await opts.askHuman(role.id, args.question);
1393
+ const receipt = await opts.askHuman(role.id, args.question, args.blocking);
1024
1394
  return text(receipt);
1025
1395
  },
1026
1396
  });