@monoes/monomindcli 2.13.0 → 2.14.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude/skills/monodesign/scripts/context.mjs +1 -1
- package/.claude/skills/monodesign/scripts/detector/detect-antipatterns-browser.js +2 -2
- package/.claude/skills/monodesign/scripts/detector/rules/checks.mjs +2 -2
- package/.claude/skills/monodesign/scripts/live-browser.js +1 -1
- package/bin/cli.js +21 -1
- package/bin/mcp-server.js +21 -1
- package/dist/src/capabilities/cap-documents.d.ts.map +1 -1
- package/dist/src/capabilities/cap-documents.js +23 -0
- package/dist/src/capabilities/cap-documents.js.map +1 -1
- package/dist/src/commands/doc-filters.d.ts +19 -0
- package/dist/src/commands/doc-filters.d.ts.map +1 -0
- package/dist/src/commands/doc-filters.js +58 -0
- package/dist/src/commands/doc-filters.js.map +1 -0
- package/dist/src/commands/doc-library.d.ts +21 -0
- package/dist/src/commands/doc-library.d.ts.map +1 -0
- package/dist/src/commands/doc-library.js +390 -0
- package/dist/src/commands/doc-library.js.map +1 -0
- package/dist/src/commands/doc-list.d.ts +9 -0
- package/dist/src/commands/doc-list.d.ts.map +1 -0
- package/dist/src/commands/doc-list.js +99 -0
- package/dist/src/commands/doc-list.js.map +1 -0
- package/dist/src/commands/doc.d.ts.map +1 -1
- package/dist/src/commands/doc.js +49 -37
- package/dist/src/commands/doc.js.map +1 -1
- package/dist/src/commands/doctor-mcp-probe.d.ts +48 -0
- package/dist/src/commands/doctor-mcp-probe.d.ts.map +1 -0
- package/dist/src/commands/doctor-mcp-probe.js +205 -0
- package/dist/src/commands/doctor-mcp-probe.js.map +1 -0
- package/dist/src/commands/doctor-project-checks.d.ts +12 -1
- package/dist/src/commands/doctor-project-checks.d.ts.map +1 -1
- package/dist/src/commands/doctor-project-checks.js +55 -4
- package/dist/src/commands/doctor-project-checks.js.map +1 -1
- package/dist/src/commands/doctor.d.ts.map +1 -1
- package/dist/src/commands/doctor.js +10 -1
- package/dist/src/commands/doctor.js.map +1 -1
- package/dist/src/commands/init.d.ts.map +1 -1
- package/dist/src/commands/init.js +17 -0
- package/dist/src/commands/init.js.map +1 -1
- package/dist/src/commands/mcp.d.ts.map +1 -1
- package/dist/src/commands/mcp.js +4 -3
- package/dist/src/commands/mcp.js.map +1 -1
- package/dist/src/commands/org-observe.d.ts.map +1 -1
- package/dist/src/commands/org-observe.js +14 -0
- package/dist/src/commands/org-observe.js.map +1 -1
- package/dist/src/commands/org.d.ts.map +1 -1
- package/dist/src/commands/org.js +12 -2
- package/dist/src/commands/org.js.map +1 -1
- package/dist/src/init/claudemd-generator.d.ts.map +1 -1
- package/dist/src/init/claudemd-generator.js +2 -1
- package/dist/src/init/claudemd-generator.js.map +1 -1
- package/dist/src/init/mcp-generator.d.ts.map +1 -1
- package/dist/src/init/mcp-generator.js +3 -3
- package/dist/src/init/mcp-generator.js.map +1 -1
- package/dist/src/init/types.d.ts +6 -0
- package/dist/src/init/types.d.ts.map +1 -1
- package/dist/src/init/types.js.map +1 -1
- package/dist/src/init/write-runtime-config.d.ts.map +1 -1
- package/dist/src/init/write-runtime-config.js +12 -2
- package/dist/src/init/write-runtime-config.js.map +1 -1
- package/dist/src/knowledge/capture-envelope.d.ts +80 -0
- package/dist/src/knowledge/capture-envelope.d.ts.map +1 -0
- package/dist/src/knowledge/capture-envelope.js +165 -0
- package/dist/src/knowledge/capture-envelope.js.map +1 -0
- package/dist/src/knowledge/capture-text.d.ts +23 -0
- package/dist/src/knowledge/capture-text.d.ts.map +1 -0
- package/dist/src/knowledge/capture-text.js +49 -0
- package/dist/src/knowledge/capture-text.js.map +1 -0
- package/dist/src/knowledge/citation.d.ts +121 -0
- package/dist/src/knowledge/citation.d.ts.map +1 -0
- package/dist/src/knowledge/citation.js +252 -0
- package/dist/src/knowledge/citation.js.map +1 -0
- package/dist/src/knowledge/document-pipeline.d.ts +60 -0
- package/dist/src/knowledge/document-pipeline.d.ts.map +1 -1
- package/dist/src/knowledge/document-pipeline.js +188 -4
- package/dist/src/knowledge/document-pipeline.js.map +1 -1
- package/dist/src/knowledge/html-extract.d.ts +43 -0
- package/dist/src/knowledge/html-extract.d.ts.map +1 -0
- package/dist/src/knowledge/html-extract.js +286 -0
- package/dist/src/knowledge/html-extract.js.map +1 -0
- package/dist/src/knowledge/html-tags.d.ts +30 -0
- package/dist/src/knowledge/html-tags.d.ts.map +1 -0
- package/dist/src/knowledge/html-tags.js +229 -0
- package/dist/src/knowledge/html-tags.js.map +1 -0
- package/dist/src/knowledge/library.d.ts +106 -0
- package/dist/src/knowledge/library.d.ts.map +1 -0
- package/dist/src/knowledge/library.js +195 -0
- package/dist/src/knowledge/library.js.map +1 -0
- package/dist/src/knowledge/mhtml.d.ts +60 -0
- package/dist/src/knowledge/mhtml.d.ts.map +1 -0
- package/dist/src/knowledge/mhtml.js +154 -0
- package/dist/src/knowledge/mhtml.js.map +1 -0
- package/dist/src/knowledge/related.d.ts +58 -0
- package/dist/src/knowledge/related.d.ts.map +1 -0
- package/dist/src/knowledge/related.js +222 -0
- package/dist/src/knowledge/related.js.map +1 -0
- package/dist/src/knowledge/section-diff.d.ts +48 -0
- package/dist/src/knowledge/section-diff.d.ts.map +1 -0
- package/dist/src/knowledge/section-diff.js +151 -0
- package/dist/src/knowledge/section-diff.js.map +1 -0
- package/dist/src/knowledge/watch.d.ts +92 -0
- package/dist/src/knowledge/watch.d.ts.map +1 -0
- package/dist/src/knowledge/watch.js +221 -0
- package/dist/src/knowledge/watch.js.map +1 -0
- package/dist/src/mcp-server.d.ts.map +1 -1
- package/dist/src/mcp-server.js +44 -87
- package/dist/src/mcp-server.js.map +1 -1
- package/dist/src/mcp-tools/browser-instrument-tools.d.ts +35 -0
- package/dist/src/mcp-tools/browser-instrument-tools.d.ts.map +1 -0
- package/dist/src/mcp-tools/browser-instrument-tools.js +359 -0
- package/dist/src/mcp-tools/browser-instrument-tools.js.map +1 -0
- package/dist/src/mcp-tools/browser-metrics.d.ts +42 -0
- package/dist/src/mcp-tools/browser-metrics.d.ts.map +1 -0
- package/dist/src/mcp-tools/browser-metrics.js +91 -0
- package/dist/src/mcp-tools/browser-metrics.js.map +1 -0
- package/dist/src/mcp-tools/browser-profile-tools.d.ts +16 -0
- package/dist/src/mcp-tools/browser-profile-tools.d.ts.map +1 -0
- package/dist/src/mcp-tools/browser-profile-tools.js +324 -0
- package/dist/src/mcp-tools/browser-profile-tools.js.map +1 -0
- package/dist/src/mcp-tools/browser-session.d.ts +68 -0
- package/dist/src/mcp-tools/browser-session.d.ts.map +1 -0
- package/dist/src/mcp-tools/browser-session.js +224 -0
- package/dist/src/mcp-tools/browser-session.js.map +1 -0
- package/dist/src/mcp-tools/browser-tools.d.ts.map +1 -1
- package/dist/src/mcp-tools/browser-tools.js +10 -176
- package/dist/src/mcp-tools/browser-tools.js.map +1 -1
- package/dist/src/mcp-tools/capture-resource-read.d.ts +115 -0
- package/dist/src/mcp-tools/capture-resource-read.d.ts.map +1 -0
- package/dist/src/mcp-tools/capture-resource-read.js +296 -0
- package/dist/src/mcp-tools/capture-resource-read.js.map +1 -0
- package/dist/src/mcp-tools/capture-resource-tools.d.ts +22 -0
- package/dist/src/mcp-tools/capture-resource-tools.d.ts.map +1 -0
- package/dist/src/mcp-tools/capture-resource-tools.js +182 -0
- package/dist/src/mcp-tools/capture-resource-tools.js.map +1 -0
- package/dist/src/mcp-tools/capture-resources.d.ts +142 -0
- package/dist/src/mcp-tools/capture-resources.d.ts.map +1 -0
- package/dist/src/mcp-tools/capture-resources.js +289 -0
- package/dist/src/mcp-tools/capture-resources.js.map +1 -0
- package/dist/src/mcp-tools/index.d.ts +3 -0
- package/dist/src/mcp-tools/index.d.ts.map +1 -1
- package/dist/src/mcp-tools/index.js +6 -0
- package/dist/src/mcp-tools/index.js.map +1 -1
- package/dist/src/mcp-tools/knowledge-tools.d.ts.map +1 -1
- package/dist/src/mcp-tools/knowledge-tools.js +9 -1
- package/dist/src/mcp-tools/knowledge-tools.js.map +1 -1
- package/dist/src/mcp-tools/resource-router.d.ts +86 -0
- package/dist/src/mcp-tools/resource-router.d.ts.map +1 -0
- package/dist/src/mcp-tools/resource-router.js +181 -0
- package/dist/src/mcp-tools/resource-router.js.map +1 -0
- package/dist/src/memory/memory-bridge.js +1 -1
- package/dist/src/memory/memory-bridge.js.map +1 -1
- package/dist/src/orgrt/agent-runner.d.ts +38 -0
- package/dist/src/orgrt/agent-runner.d.ts.map +1 -1
- package/dist/src/orgrt/agent-runner.js +47 -0
- package/dist/src/orgrt/agent-runner.js.map +1 -1
- package/dist/src/orgrt/checkpoint.d.ts +12 -1
- package/dist/src/orgrt/checkpoint.d.ts.map +1 -1
- package/dist/src/orgrt/checkpoint.js +15 -6
- package/dist/src/orgrt/checkpoint.js.map +1 -1
- package/dist/src/orgrt/completion-gate.d.ts +74 -0
- package/dist/src/orgrt/completion-gate.d.ts.map +1 -1
- package/dist/src/orgrt/completion-gate.js +45 -0
- package/dist/src/orgrt/completion-gate.js.map +1 -1
- package/dist/src/orgrt/cost-tier.d.ts +144 -0
- package/dist/src/orgrt/cost-tier.d.ts.map +1 -0
- package/dist/src/orgrt/cost-tier.js +183 -0
- package/dist/src/orgrt/cost-tier.js.map +1 -0
- package/dist/src/orgrt/cross-org.d.ts.map +1 -1
- package/dist/src/orgrt/cross-org.js +14 -0
- package/dist/src/orgrt/cross-org.js.map +1 -1
- package/dist/src/orgrt/daemon.d.ts +12 -1
- package/dist/src/orgrt/daemon.d.ts.map +1 -1
- package/dist/src/orgrt/daemon.js +216 -49
- package/dist/src/orgrt/daemon.js.map +1 -1
- package/dist/src/orgrt/decisions.d.ts +14 -2
- package/dist/src/orgrt/decisions.d.ts.map +1 -1
- package/dist/src/orgrt/decisions.js +216 -10
- package/dist/src/orgrt/decisions.js.map +1 -1
- package/dist/src/orgrt/file-roots.d.ts +7 -0
- package/dist/src/orgrt/file-roots.d.ts.map +1 -1
- package/dist/src/orgrt/file-roots.js +38 -1
- package/dist/src/orgrt/file-roots.js.map +1 -1
- package/dist/src/orgrt/forwarder.d.ts.map +1 -1
- package/dist/src/orgrt/forwarder.js +16 -2
- package/dist/src/orgrt/forwarder.js.map +1 -1
- package/dist/src/orgrt/idle-deadline.d.ts +74 -1
- package/dist/src/orgrt/idle-deadline.d.ts.map +1 -1
- package/dist/src/orgrt/idle-deadline.js +52 -2
- package/dist/src/orgrt/idle-deadline.js.map +1 -1
- package/dist/src/orgrt/loadouts.d.ts +38 -0
- package/dist/src/orgrt/loadouts.d.ts.map +1 -0
- package/dist/src/orgrt/loadouts.js +129 -0
- package/dist/src/orgrt/loadouts.js.map +1 -0
- package/dist/src/orgrt/mailbox.d.ts +17 -1
- package/dist/src/orgrt/mailbox.d.ts.map +1 -1
- package/dist/src/orgrt/mailbox.js +50 -3
- package/dist/src/orgrt/mailbox.js.map +1 -1
- package/dist/src/orgrt/policy.d.ts +42 -2
- package/dist/src/orgrt/policy.d.ts.map +1 -1
- package/dist/src/orgrt/policy.js +53 -8
- package/dist/src/orgrt/policy.js.map +1 -1
- package/dist/src/orgrt/prompt-vars.d.ts +11 -0
- package/dist/src/orgrt/prompt-vars.d.ts.map +1 -0
- package/dist/src/orgrt/prompt-vars.js +49 -0
- package/dist/src/orgrt/prompt-vars.js.map +1 -0
- package/dist/src/orgrt/questions.d.ts +22 -1
- package/dist/src/orgrt/questions.d.ts.map +1 -1
- package/dist/src/orgrt/questions.js +24 -3
- package/dist/src/orgrt/questions.js.map +1 -1
- package/dist/src/orgrt/review-packet.d.ts +29 -0
- package/dist/src/orgrt/review-packet.d.ts.map +1 -0
- package/dist/src/orgrt/review-packet.js +67 -0
- package/dist/src/orgrt/review-packet.js.map +1 -0
- package/dist/src/orgrt/role-sandbox.d.ts.map +1 -1
- package/dist/src/orgrt/role-sandbox.js +16 -3
- package/dist/src/orgrt/role-sandbox.js.map +1 -1
- package/dist/src/orgrt/session-ledger.d.ts +68 -0
- package/dist/src/orgrt/session-ledger.d.ts.map +1 -0
- package/dist/src/orgrt/session-ledger.js +128 -0
- package/dist/src/orgrt/session-ledger.js.map +1 -0
- package/dist/src/orgrt/session.d.ts +45 -5
- package/dist/src/orgrt/session.d.ts.map +1 -1
- package/dist/src/orgrt/session.js +397 -27
- package/dist/src/orgrt/session.js.map +1 -1
- package/dist/src/orgrt/task-dag.d.ts +34 -1
- package/dist/src/orgrt/task-dag.d.ts.map +1 -1
- package/dist/src/orgrt/task-dag.js +58 -7
- package/dist/src/orgrt/task-dag.js.map +1 -1
- package/dist/src/orgrt/tool-spill.d.ts +94 -0
- package/dist/src/orgrt/tool-spill.d.ts.map +1 -0
- package/dist/src/orgrt/tool-spill.js +180 -0
- package/dist/src/orgrt/tool-spill.js.map +1 -0
- package/dist/src/orgrt/types.d.ts +138 -0
- package/dist/src/orgrt/types.d.ts.map +1 -1
- package/dist/src/orgrt/types.js +148 -0
- package/dist/src/orgrt/types.js.map +1 -1
- package/dist/src/platform-adapters/renderers/mcp.d.ts +20 -2
- package/dist/src/platform-adapters/renderers/mcp.d.ts.map +1 -1
- package/dist/src/platform-adapters/renderers/mcp.js +25 -5
- package/dist/src/platform-adapters/renderers/mcp.js.map +1 -1
- package/dist/src/protocol-capabilities.d.ts +2 -1
- package/dist/src/protocol-capabilities.d.ts.map +1 -1
- package/dist/src/protocol-capabilities.js +2 -1
- package/dist/src/protocol-capabilities.js.map +1 -1
- package/dist/src/ui/dashboard.html +450 -164
- package/dist/src/ui/org-hil.mjs +276 -0
- package/dist/src/ui/org-runtime.mjs +350 -0
- package/dist/src/ui/orgs.html +15 -4
- package/dist/src/ui/routes-org.mjs +290 -541
- package/dist/src/ui/server.mjs +25 -5
- package/dist/tsconfig.tsbuildinfo +1 -1
- package/package.json +9 -8
|
@@ -12,10 +12,15 @@ import { StateDetector } from './state-detector.js';
|
|
|
12
12
|
* "boss appears hung". */
|
|
13
13
|
const SILENT_SESSION_MS = 4 * 60_000;
|
|
14
14
|
const CONTEXT_LIMIT_RE = /context.window.limit|context.length.exceeded|maximum.context/i;
|
|
15
|
+
import { createHash } from 'node:crypto';
|
|
15
16
|
import { readFileSync } from 'node:fs';
|
|
17
|
+
import { join } from 'node:path';
|
|
18
|
+
import { resolveRoleCostTier } from './cost-tier.js';
|
|
19
|
+
import { expandRolePromptVars, promptVarsFor } from './prompt-vars.js';
|
|
16
20
|
import { resolveProviderEnv, resolveRoleProvider } from './provider.js';
|
|
17
21
|
import { resolveRoleGitEnforcement } from './role-sandbox.js';
|
|
18
22
|
import { loadBuiltinRoleSkill } from './role-skills.js';
|
|
23
|
+
import { mailRouteKey, ROLE_SESSION_KEY, resolveSessionScope, SessionLedger, } from './session-ledger.js';
|
|
19
24
|
import { DEFAULT_CLAUDE_MODEL, VERCEL_PROVIDERS } from './vercel-providers.js';
|
|
20
25
|
/**
|
|
21
26
|
* Resolves the extra system-prompt block for a role: built-in archetype
|
|
@@ -189,6 +194,14 @@ endpointBriefing) {
|
|
|
189
194
|
.filter(Boolean)
|
|
190
195
|
.join('\n\n');
|
|
191
196
|
}
|
|
197
|
+
/** The system prompt one session of this role is built with. */
|
|
198
|
+
function rolePromptFor(opts) {
|
|
199
|
+
return buildRolePrompt(expandRolePromptVars(opts.role, promptVarsFor(opts.orgRoot ?? opts.cwd)), (opts.def ?? { name: opts.org, goal: '' }), opts.def?.roles.map((r) => r.id) ?? [opts.role.id], opts.glossary,
|
|
200
|
+
// D7: the loadout's text follows the role's own guidance. With no
|
|
201
|
+
// loadout this is exactly resolveRoleExtraGuidance(role), as before.
|
|
202
|
+
[resolveRoleExtraGuidance(opts.role), opts.loadout?.guidance].filter(Boolean).join('\n\n') ||
|
|
203
|
+
undefined, opts.onComplete ? endpointBriefingLines(opts.def) : undefined);
|
|
204
|
+
}
|
|
192
205
|
/**
|
|
193
206
|
* Runs a role for the life of the org, transparently restarting the
|
|
194
207
|
* underlying SDK session whenever it ends on its own (`maxTurns` reached)
|
|
@@ -248,23 +261,151 @@ async function runAgentSessionLoop(opts) {
|
|
|
248
261
|
// SDK session id here (it must outlive individual runOneSession calls,
|
|
249
262
|
// since resume continues the same billing session) and emit only deltas.
|
|
250
263
|
const sessionCostTotals = new Map();
|
|
264
|
+
// ADR-O001 D1: modelUsage is cumulative per session exactly like
|
|
265
|
+
// total_cost_usd, so it needs the same prev-value map to become a delta.
|
|
266
|
+
const sessionTokenTotals = new Map();
|
|
267
|
+
// ADR-O001 D3. 'role' scope (the default) keeps the pre-D3 loop exactly: one
|
|
268
|
+
// model session for the role's life, resumed across maxTurns restarts via
|
|
269
|
+
// resumeSessionId. 'task' scope keys model sessions by the task a message
|
|
270
|
+
// belongs to: the process exits at a task boundary (or on idle) and the next
|
|
271
|
+
// one resumes that task's session from the ledger. Either way every session
|
|
272
|
+
// run is recorded with its session id before and after.
|
|
273
|
+
const scope = resolveSessionScope(opts.role, opts.def);
|
|
274
|
+
const idleExitMs = opts.def?.run_config
|
|
275
|
+
?.session_idle_exit_ms;
|
|
276
|
+
const ledger = opts.sessionLedger ?? new SessionLedger();
|
|
277
|
+
const runtimeKey = opts.role.runtime ?? opts.def?.runtime ?? 'claude';
|
|
278
|
+
let taskKey = ROLE_SESSION_KEY;
|
|
279
|
+
// Why the next fresh session for a key is fresh, when the loop itself threw
|
|
280
|
+
// the record away (stale resume, turn-limit error) — recorded, not guessed.
|
|
281
|
+
const droppedBecause = new Map();
|
|
282
|
+
const staleTried = new Set();
|
|
283
|
+
// The options THIS session is built from — the role's own, except that a
|
|
284
|
+
// task-scoped session carries its task's loadout (D7).
|
|
285
|
+
let sessionOpts = opts;
|
|
286
|
+
let promptHash = '*';
|
|
287
|
+
// Task scope: which task last wrote to each correspondent, so their
|
|
288
|
+
// untagged reply goes back to that task's session (see mailRouteKey).
|
|
289
|
+
const correspondents = new Map();
|
|
251
290
|
// Always run at least once: a mailbox can be closed with queued items still
|
|
252
291
|
// pending (stream() drains the queue before honoring `closed`), which is a
|
|
253
292
|
// normal, valid starting state - checking isClosed before the first run
|
|
254
293
|
// would skip that drain entirely.
|
|
255
294
|
while (true) {
|
|
295
|
+
// Opt-in only: keep the process DOWN until there is mail, instead of
|
|
296
|
+
// starting a query() that parks on an empty mailbox. waitForMessage()
|
|
297
|
+
// still returns true for a closed mailbox with queued items.
|
|
298
|
+
if (scope !== 'role' || idleExitMs !== undefined) {
|
|
299
|
+
if (!(await mailbox.waitForMessage()))
|
|
300
|
+
return;
|
|
301
|
+
}
|
|
302
|
+
let startReason;
|
|
303
|
+
if (scope === 'cold') {
|
|
304
|
+
// D6: nothing carries over — not a checkpointed session, not the last
|
|
305
|
+
// message's. Paying the cache miss here is the point.
|
|
306
|
+
resumeSessionId = undefined;
|
|
307
|
+
startReason = 'fresh-cold';
|
|
308
|
+
}
|
|
309
|
+
else if (scope === 'task') {
|
|
310
|
+
// An untagged message (mail, an answer, a continuation) belongs to the
|
|
311
|
+
// session the role is already in.
|
|
312
|
+
taskKey = mailRouteKey(mailbox.peek() ?? '', correspondents) ?? taskKey;
|
|
313
|
+
const key = taskKey;
|
|
314
|
+
sessionOpts = {
|
|
315
|
+
...opts,
|
|
316
|
+
// D7: this task's session is built with this task's loadout.
|
|
317
|
+
...(key !== ROLE_SESSION_KEY && opts.loadoutFor ? { loadout: opts.loadoutFor(key) } : {}),
|
|
318
|
+
// Mail sent from a task's session names the task in its subject, so
|
|
319
|
+
// the reader knows what it is about and a reply can find its way back.
|
|
320
|
+
deliver: (from, to, subject, body) => {
|
|
321
|
+
if (key === ROLE_SESSION_KEY)
|
|
322
|
+
return opts.deliver(from, to, subject, body);
|
|
323
|
+
correspondents.set(to, key);
|
|
324
|
+
const tagged = subject.includes('[task:') ? subject : `[task:${key}] ${subject}`;
|
|
325
|
+
return opts.deliver(from, to, tagged, body);
|
|
326
|
+
},
|
|
327
|
+
};
|
|
328
|
+
promptHash = createHash('sha256')
|
|
329
|
+
.update(rolePromptFor(sessionOpts))
|
|
330
|
+
.digest('hex')
|
|
331
|
+
.slice(0, 16);
|
|
332
|
+
const pick = ledger.resumeFor({
|
|
333
|
+
role: opts.role.id,
|
|
334
|
+
runtime: runtimeKey,
|
|
335
|
+
taskKey,
|
|
336
|
+
cwd: opts.cwd,
|
|
337
|
+
promptHash,
|
|
338
|
+
});
|
|
339
|
+
resumeSessionId = pick.sessionId;
|
|
340
|
+
startReason =
|
|
341
|
+
pick.reason === 'fresh-no-record'
|
|
342
|
+
? (droppedBecause.get(taskKey) ?? pick.reason)
|
|
343
|
+
: pick.reason;
|
|
344
|
+
}
|
|
345
|
+
else {
|
|
346
|
+
startReason = resumeSessionId ? 'resumed' : 'fresh-no-record';
|
|
347
|
+
}
|
|
348
|
+
const sessionKey = taskKey;
|
|
349
|
+
const streamOpts = scope === 'cold'
|
|
350
|
+
? { stopBefore: () => true, idleExitMs }
|
|
351
|
+
: scope === 'task'
|
|
352
|
+
? {
|
|
353
|
+
stopBefore: (next) => {
|
|
354
|
+
const k = mailRouteKey(next, correspondents);
|
|
355
|
+
return k !== undefined && k !== sessionKey;
|
|
356
|
+
},
|
|
357
|
+
idleExitMs,
|
|
358
|
+
}
|
|
359
|
+
: idleExitMs !== undefined
|
|
360
|
+
? { idleExitMs }
|
|
361
|
+
: undefined;
|
|
362
|
+
const sessionIdBefore = resumeSessionId;
|
|
363
|
+
const startedAt = Date.now();
|
|
364
|
+
const recordRun = (after, error) => {
|
|
365
|
+
const run = ledger.recordRun({
|
|
366
|
+
role: opts.role.id,
|
|
367
|
+
runtime: runtimeKey,
|
|
368
|
+
taskKey: sessionKey,
|
|
369
|
+
sessionIdBefore,
|
|
370
|
+
sessionIdAfter: after,
|
|
371
|
+
reason: startReason,
|
|
372
|
+
startedAt,
|
|
373
|
+
endedAt: Date.now(),
|
|
374
|
+
...(error ? { error } : {}),
|
|
375
|
+
});
|
|
376
|
+
opts.bus.emit({
|
|
377
|
+
type: 'audit',
|
|
378
|
+
from: opts.role.id,
|
|
379
|
+
reason: 'session-run',
|
|
380
|
+
msg: `session ${run.resumed ? 'resumed' : 'started fresh'} (${startReason}) for ${sessionKey}`,
|
|
381
|
+
data: run,
|
|
382
|
+
});
|
|
383
|
+
};
|
|
256
384
|
const realBefore = mailbox.consumedRealCount;
|
|
257
385
|
let sessionId;
|
|
258
386
|
let hitTurnLimit = false;
|
|
259
387
|
const attempt = { replied: false };
|
|
260
388
|
try {
|
|
261
|
-
const res = await runOneSession(
|
|
389
|
+
const res = await runOneSession(sessionOpts, resumeSessionId, sessionCostTotals, attempt, sessionTokenTotals, streamOpts);
|
|
262
390
|
sessionId = res.sessionId;
|
|
263
391
|
hitTurnLimit = res.hitTurnLimit;
|
|
264
392
|
resumeSessionId = sessionId;
|
|
393
|
+
recordRun(sessionId);
|
|
394
|
+
if (scope === 'task' && sessionId) {
|
|
395
|
+
droppedBecause.delete(sessionKey);
|
|
396
|
+
ledger.set({
|
|
397
|
+
role: opts.role.id,
|
|
398
|
+
runtime: runtimeKey,
|
|
399
|
+
taskKey: sessionKey,
|
|
400
|
+
cwd: opts.cwd,
|
|
401
|
+
promptHash,
|
|
402
|
+
sessionId,
|
|
403
|
+
});
|
|
404
|
+
}
|
|
265
405
|
}
|
|
266
406
|
catch (err) {
|
|
267
407
|
const errMsg = err instanceof Error ? err.message : String(err);
|
|
408
|
+
recordRun(undefined, errMsg);
|
|
268
409
|
// #304 (review round 3): an org's own stop aborts whatever this attempt
|
|
269
410
|
// was doing — max-turns and stale-resume below are diagnoses for a
|
|
270
411
|
// genuinely failed attempt, not for one the org itself just cut off.
|
|
@@ -285,8 +426,36 @@ async function runAgentSessionLoop(opts) {
|
|
|
285
426
|
sessionId = undefined;
|
|
286
427
|
resumeSessionId = undefined;
|
|
287
428
|
hitTurnLimit = true;
|
|
429
|
+
if (scope === 'task') {
|
|
430
|
+
ledger.drop({ role: opts.role.id, runtime: runtimeKey, taskKey: sessionKey });
|
|
431
|
+
droppedBecause.set(sessionKey, 'fresh-after-turn-limit');
|
|
432
|
+
}
|
|
433
|
+
}
|
|
434
|
+
else if (!stopping &&
|
|
435
|
+
scope === 'task' &&
|
|
436
|
+
sessionIdBefore !== undefined &&
|
|
437
|
+
!staleTried.has(sessionKey) &&
|
|
438
|
+
!attempt.replied) {
|
|
439
|
+
// D3's per-key form of #149 below: a recorded session that fails
|
|
440
|
+
// before replying is treated as expired — forget it and retry that
|
|
441
|
+
// task fresh once. Anything it had already pulled goes back first.
|
|
442
|
+
staleTried.add(sessionKey);
|
|
443
|
+
ledger.drop({ role: opts.role.id, runtime: runtimeKey, taskKey: sessionKey });
|
|
444
|
+
droppedBecause.set(sessionKey, 'fresh-after-stale-resume');
|
|
445
|
+
mailbox.reclaimInFlight();
|
|
446
|
+
sessionId = undefined;
|
|
447
|
+
resumeSessionId = undefined;
|
|
448
|
+
hitTurnLimit = false;
|
|
449
|
+
opts.bus.emit({
|
|
450
|
+
type: 'status',
|
|
451
|
+
from: opts.role.id,
|
|
452
|
+
reason: 'resume-session-stale',
|
|
453
|
+
msg: `agent "${opts.role.id}" could not resume its session for ${sessionKey} — retrying with a fresh session`,
|
|
454
|
+
data: { error: errMsg },
|
|
455
|
+
});
|
|
288
456
|
}
|
|
289
457
|
else if (!stopping &&
|
|
458
|
+
scope === 'role' &&
|
|
290
459
|
resumeSessionId &&
|
|
291
460
|
resumeSessionId === initialResumeSessionId &&
|
|
292
461
|
!triedFreshAfterResumeFailure &&
|
|
@@ -357,6 +526,17 @@ async function runAgentSessionLoop(opts) {
|
|
|
357
526
|
});
|
|
358
527
|
}
|
|
359
528
|
}
|
|
529
|
+
else if (mailbox.lastStreamEnd) {
|
|
530
|
+
// D3: the process ended on purpose — a task boundary or idle — and the
|
|
531
|
+
// model session is kept for the next wake.
|
|
532
|
+
opts.bus.emit({
|
|
533
|
+
type: 'status',
|
|
534
|
+
from: opts.role.id,
|
|
535
|
+
reason: 'session-cycled',
|
|
536
|
+
msg: `process cycled (${mailbox.lastStreamEnd}); model session kept for ${sessionKey}`,
|
|
537
|
+
data: { taskKey: sessionKey, end: mailbox.lastStreamEnd, sessionId },
|
|
538
|
+
});
|
|
539
|
+
}
|
|
360
540
|
else {
|
|
361
541
|
opts.bus.emit({
|
|
362
542
|
type: 'status',
|
|
@@ -366,11 +546,68 @@ async function runAgentSessionLoop(opts) {
|
|
|
366
546
|
}
|
|
367
547
|
}
|
|
368
548
|
}
|
|
549
|
+
/** ADR-O001 D1 — token-metering helpers.
|
|
550
|
+
*
|
|
551
|
+
* `cache_read_input_tokens` and `cache_creation_input_tokens` are siblings
|
|
552
|
+
* of `input_tokens` in the Anthropic API, not subsets of it, and both are
|
|
553
|
+
* billable. Everything below therefore sums all four. */
|
|
554
|
+
function totalTokens(u) {
|
|
555
|
+
return u.input + u.output + u.cacheRead + u.cacheCreation;
|
|
556
|
+
}
|
|
557
|
+
function addTo(target, add) {
|
|
558
|
+
target.input += add.input;
|
|
559
|
+
target.output += add.output;
|
|
560
|
+
target.cacheRead += add.cacheRead;
|
|
561
|
+
target.cacheCreation += add.cacheCreation;
|
|
562
|
+
}
|
|
563
|
+
/** One model turn's own usage, off an 'assistant' (or per-turn 'result')
|
|
564
|
+
* message. */
|
|
565
|
+
function turnBreakdown(m) {
|
|
566
|
+
return {
|
|
567
|
+
input: m.input_tokens ?? 0,
|
|
568
|
+
output: m.output_tokens ?? 0,
|
|
569
|
+
cacheRead: m.cache_read_input_tokens ?? 0,
|
|
570
|
+
cacheCreation: m.cache_creation_input_tokens ?? 0,
|
|
571
|
+
};
|
|
572
|
+
}
|
|
573
|
+
/** What a 'result' message says this mailbox message consumed.
|
|
574
|
+
*
|
|
575
|
+
* When the runner reports `cumulative_tokens` (the Claude SDK's whole-pipeline
|
|
576
|
+
* `modelUsage`, which unlike `usage` includes Task subagents and sidechains),
|
|
577
|
+
* that value is CUMULATIVE per session — the same lifecycle as
|
|
578
|
+
* `total_cost_usd` — so it is converted to a delta against the previous value
|
|
579
|
+
* for the same session_id. A fresh/restarted session has no prior entry and
|
|
580
|
+
* correctly yields its full value; a value that ticks down (a provider-side
|
|
581
|
+
* correction) floors at 0 rather than re-adding the whole cumulative total.
|
|
582
|
+
* Without `cumulative_tokens` the per-turn fields are used as before. */
|
|
583
|
+
function resultBreakdown(m, tokenTotals, sid) {
|
|
584
|
+
const cum = m.cumulative_tokens;
|
|
585
|
+
if (!cum)
|
|
586
|
+
return turnBreakdown(m);
|
|
587
|
+
const now = {
|
|
588
|
+
input: cum.input,
|
|
589
|
+
output: cum.output,
|
|
590
|
+
cacheRead: cum.cache_read,
|
|
591
|
+
cacheCreation: cum.cache_creation,
|
|
592
|
+
};
|
|
593
|
+
if (!tokenTotals)
|
|
594
|
+
return now;
|
|
595
|
+
const prev = tokenTotals.get(sid);
|
|
596
|
+
tokenTotals.set(sid, now);
|
|
597
|
+
if (!prev)
|
|
598
|
+
return now;
|
|
599
|
+
return {
|
|
600
|
+
input: Math.max(0, now.input - prev.input),
|
|
601
|
+
output: Math.max(0, now.output - prev.output),
|
|
602
|
+
cacheRead: Math.max(0, now.cacheRead - prev.cacheRead),
|
|
603
|
+
cacheCreation: Math.max(0, now.cacheCreation - prev.cacheCreation),
|
|
604
|
+
};
|
|
605
|
+
}
|
|
369
606
|
/** One bounded SDK session for a role; resolves with the SDK's session_id (for
|
|
370
607
|
* resuming on restart) and whether it ended by hitting the turn limit (so the
|
|
371
608
|
* caller can push a continuation) when the stream ends (mailbox closed or
|
|
372
609
|
* maxTurns reached). */
|
|
373
|
-
async function runOneSession(opts, resume, costTotals, progress) {
|
|
610
|
+
async function runOneSession(opts, resume, costTotals, progress, tokenTotals, streamOpts) {
|
|
374
611
|
const { org, role, bus, policy, mailbox, cwd } = opts;
|
|
375
612
|
// Read lastMessageId live from opts instead of capturing at session start
|
|
376
613
|
// This ensures chat responses link to the most recent message delivered
|
|
@@ -389,7 +626,24 @@ async function runOneSession(opts, resume, costTotals, progress) {
|
|
|
389
626
|
// The named provider's default model fills in adapter_config.model when the
|
|
390
627
|
// role didn't pin one.
|
|
391
628
|
const prov = resolveRoleProvider(role, opts.orgRoot ?? opts.cwd);
|
|
629
|
+
// ADR-O001 D8: the role's cost tier, when the org declares one. Resolved
|
|
630
|
+
// here — the single choke point where a role's model is decided — so the
|
|
631
|
+
// documented precedence holds in exactly one place:
|
|
632
|
+
// explicit adapter_config.model > tier > named-provider default > runtime
|
|
633
|
+
// The tier's EFFORT is applied even when the model came from an explicit
|
|
634
|
+
// pin: which model to run and how hard to think are separate axes, and
|
|
635
|
+
// silently dropping the effort because a model was pinned would be the
|
|
636
|
+
// "silent downgrade" this decision exists to prevent.
|
|
637
|
+
// Throws (fails the session) rather than guessing when the tier has no
|
|
638
|
+
// entry for this role's provider — daemon.ts validates the whole roster
|
|
639
|
+
// up front so that is normally caught before any token is spent.
|
|
640
|
+
const tier = resolveRoleCostTier({
|
|
641
|
+
role,
|
|
642
|
+
def: opts.def,
|
|
643
|
+
vendor: role.provider?.vendor ?? prov.cfg?.vendor,
|
|
644
|
+
});
|
|
392
645
|
const model = role.adapter_config?.model ??
|
|
646
|
+
tier?.model ??
|
|
393
647
|
prov.defaultModel ??
|
|
394
648
|
resolveModel(role, role.runtime, role.provider?.vendor ?? prov.cfg?.vendor);
|
|
395
649
|
bus.emit({ type: 'status', from: role.id, msg: 'session starting' });
|
|
@@ -401,7 +655,7 @@ async function runOneSession(opts, resume, costTotals, progress) {
|
|
|
401
655
|
// time a 'result' message ends one mailbox message and the next one starts.
|
|
402
656
|
// Exists purely so the 'result' branch never re-adds what this branch
|
|
403
657
|
// already added (see there for why it can't just always add).
|
|
404
|
-
let messageTurnTokens = 0;
|
|
658
|
+
let messageTurnTokens = { input: 0, output: 0, cacheRead: 0, cacheCreation: 0 };
|
|
405
659
|
// Abort hook for the runner (AgentRunArgs.signal): the silent-stream
|
|
406
660
|
// abort below used to call iterator.return() only, which queues behind a
|
|
407
661
|
// subprocess runner blocked in `for await (child.stdout)` — the child was
|
|
@@ -438,12 +692,18 @@ async function runOneSession(opts, resume, costTotals, progress) {
|
|
|
438
692
|
});
|
|
439
693
|
const stream = runner.run({
|
|
440
694
|
tools,
|
|
441
|
-
|
|
442
|
-
|
|
695
|
+
// No options = the pre-D3 stream, exactly.
|
|
696
|
+
prompt: streamOpts ? mailbox.stream('', streamOpts) : mailbox.stream(),
|
|
697
|
+
systemPrompt: rolePromptFor(opts),
|
|
443
698
|
model,
|
|
444
699
|
cwd,
|
|
700
|
+
effort: tier?.effort,
|
|
445
701
|
env: {
|
|
446
702
|
...resolveProviderEnv(prov.cfg),
|
|
703
|
+
// D8: how a NON-Claude provider expresses the tier's effort level.
|
|
704
|
+
// Empty for Claude (handled natively by ClaudeAgentRunner) and for a
|
|
705
|
+
// provider that declares no mechanism — which simply ignores effort.
|
|
706
|
+
...(tier?.env ?? {}),
|
|
447
707
|
// Custom-endpoint providers (named-provider path): pin the engine's
|
|
448
708
|
// model env so background/haiku tasks also route to the endpoint's
|
|
449
709
|
// model instead of erroring on an Anthropic-only default.
|
|
@@ -475,6 +735,13 @@ async function runOneSession(opts, resume, costTotals, progress) {
|
|
|
475
735
|
maxTurns: opts.maxTurns ?? 30,
|
|
476
736
|
resume,
|
|
477
737
|
claudeRestrictions: gitEnforcement.claudeRestrictions,
|
|
738
|
+
// ADR-O001 D2: tool results are 76% of a role's context mass and nothing
|
|
739
|
+
// bounded them. Under the ORG STATE dir (never the workspace cwd, which
|
|
740
|
+
// may be the repo), and under orgRoot — which file-roots.ts already
|
|
741
|
+
// makes readable to the role's file tools and role-sandbox.ts already
|
|
742
|
+
// makes readable to Bash — so the path in the digest actually resolves
|
|
743
|
+
// when the role decides it needs the full text.
|
|
744
|
+
toolSpillDir: join(opts.orgDir ?? opts.cwd, 'tool-results', role.id.replace(/[^a-zA-Z0-9_.-]/g, '_')),
|
|
478
745
|
canUseTool: gatedCanUseTool(policy, opts.beforeTool, role.id, opts.fence, opts.onDecision
|
|
479
746
|
? (toolName, _input, decision, kind) => opts.onDecision?.(role.id, toolName, decision.message ?? 'denied', kind)
|
|
480
747
|
: undefined, opts.hasPendingGate),
|
|
@@ -624,10 +891,18 @@ async function runOneSession(opts, resume, costTotals, progress) {
|
|
|
624
891
|
// .d.ts, which puts `usage`/token counts on BetaMessage but cost only
|
|
625
892
|
// on SDKResultSuccess.total_cost_usd/modelUsage. overBudgetUsd is
|
|
626
893
|
// still checked below, once per message, same as before this fix.)
|
|
627
|
-
|
|
894
|
+
//
|
|
895
|
+
// ADR-O001 D1: the sum must include BOTH cache fields. They are
|
|
896
|
+
// siblings of input_tokens in the Anthropic API, not subsets of it —
|
|
897
|
+
// `input_tokens` is the uncached remainder — and both are billable
|
|
898
|
+
// (~0.1x and ~1.25x input). Omitting them meant the better the cache
|
|
899
|
+
// worked the less the meter saw: on one measured run, 2,765M tokens
|
|
900
|
+
// billed against 8.1M recorded, with input_tokens at 0.0M.
|
|
901
|
+
const turn = turnBreakdown(m);
|
|
902
|
+
const turnTokens = totalTokens(turn);
|
|
628
903
|
if (turnTokens > 0) {
|
|
629
|
-
messageTurnTokens
|
|
630
|
-
policy.
|
|
904
|
+
addTo(messageTurnTokens, turn);
|
|
905
|
+
policy.addTokenUsage(turn);
|
|
631
906
|
if (policy.overBudget) {
|
|
632
907
|
bus.emit({
|
|
633
908
|
type: 'status',
|
|
@@ -664,20 +939,48 @@ async function runOneSession(opts, resume, costTotals, progress) {
|
|
|
664
939
|
});
|
|
665
940
|
}
|
|
666
941
|
else if (m.type === 'result') {
|
|
667
|
-
|
|
942
|
+
// ADR-O001 D1: prefer the SDK's `modelUsage` over `usage`. The SDK
|
|
943
|
+
// documents `usage` as "MAIN AGENT LOOP ONLY — excludes Task
|
|
944
|
+
// subagent, sidechain, and auxiliary model calls ... Prefer
|
|
945
|
+
// modelUsage for token/cost accounting"; the measured run made 46
|
|
946
|
+
// subagent calls this counter never saw. modelUsage is CUMULATIVE per
|
|
947
|
+
// session (same lifecycle as total_cost_usd, per its own type doc),
|
|
948
|
+
// so it is converted to a delta here rather than added, exactly as
|
|
949
|
+
// cost is below. A runner that reports no modelUsage falls back to
|
|
950
|
+
// the per-turn `usage` fields, which keep their old semantics.
|
|
951
|
+
const resultTokens = resultBreakdown(m, tokenTotals, m.session_id ?? sessionId ?? '');
|
|
668
952
|
// Per the SDK's own type docs, a 'result' message's usage is that
|
|
669
953
|
// message's own (effectively last-turn) usage in streaming-input mode,
|
|
670
954
|
// NOT a cumulative total across every turn of the mailbox message —
|
|
671
955
|
// and that last turn was already counted above via its own 'assistant'
|
|
672
956
|
// message, specifically so overBudget could trip mid-message. Adding
|
|
673
|
-
//
|
|
957
|
+
// the result's own usage again unconditionally would double-count it.
|
|
958
|
+
// (A modelUsage-derived delta is per-session-cumulative, so the same
|
|
959
|
+
// subtraction is exactly right there too: it removes what the
|
|
960
|
+
// assistant turns of THIS message already contributed and leaves the
|
|
961
|
+
// subagent/auxiliary volume the main loop never reported.) Only make
|
|
674
962
|
// up the shortfall (never negative) so a turn whose usage somehow
|
|
675
963
|
// never reached the 'assistant' branch (e.g. a runner/test double that
|
|
676
964
|
// doesn't emit per-turn usage) still gets counted at least once.
|
|
677
|
-
const shortfall =
|
|
678
|
-
|
|
679
|
-
|
|
680
|
-
|
|
965
|
+
const shortfall = {
|
|
966
|
+
input: Math.max(0, resultTokens.input - messageTurnTokens.input),
|
|
967
|
+
output: Math.max(0, resultTokens.output - messageTurnTokens.output),
|
|
968
|
+
cacheRead: Math.max(0, resultTokens.cacheRead - messageTurnTokens.cacheRead),
|
|
969
|
+
cacheCreation: Math.max(0, resultTokens.cacheCreation - messageTurnTokens.cacheCreation),
|
|
970
|
+
};
|
|
971
|
+
if (totalTokens(shortfall) > 0)
|
|
972
|
+
policy.addTokenUsage(shortfall);
|
|
973
|
+
// What this whole mailbox message actually added to the meter: the
|
|
974
|
+
// per-turn accounting above plus whatever the result topped up. This
|
|
975
|
+
// is what the 'usage' event reports, so a consumer summing events
|
|
976
|
+
// lands on the same number as policy.usage.
|
|
977
|
+
const messageTokens = {
|
|
978
|
+
input: messageTurnTokens.input + shortfall.input,
|
|
979
|
+
output: messageTurnTokens.output + shortfall.output,
|
|
980
|
+
cacheRead: messageTurnTokens.cacheRead + shortfall.cacheRead,
|
|
981
|
+
cacheCreation: messageTurnTokens.cacheCreation + shortfall.cacheCreation,
|
|
982
|
+
};
|
|
983
|
+
messageTurnTokens = { input: 0, output: 0, cacheRead: 0, cacheCreation: 0 };
|
|
681
984
|
// Convert the SDK's cumulative-per-session total_cost_usd into a
|
|
682
985
|
// per-result delta before emitting - downstream sums usage events.
|
|
683
986
|
// costTotals is keyed by session_id, so a genuinely new/restarted
|
|
@@ -701,7 +1004,19 @@ async function runOneSession(opts, resume, costTotals, progress) {
|
|
|
701
1004
|
bus.emit({
|
|
702
1005
|
type: 'usage',
|
|
703
1006
|
from: role.id,
|
|
704
|
-
|
|
1007
|
+
// ADR-O001 D1: the four quantities travel separately so every
|
|
1008
|
+
// downstream consumer (forwarder → dashboard state.json, reporting,
|
|
1009
|
+
// `org costs`) can record real values instead of the 0s they used
|
|
1010
|
+
// to persist. `tokens` stays the single billable total.
|
|
1011
|
+
data: {
|
|
1012
|
+
tokens: totalTokens(messageTokens),
|
|
1013
|
+
cost_usd: costDelta,
|
|
1014
|
+
subtype: m.subtype,
|
|
1015
|
+
tokens_in: messageTokens.input,
|
|
1016
|
+
tokens_out: messageTokens.output,
|
|
1017
|
+
cache_read: messageTokens.cacheRead,
|
|
1018
|
+
cache_creation: messageTokens.cacheCreation,
|
|
1019
|
+
},
|
|
705
1020
|
});
|
|
706
1021
|
if (m.subtype && m.subtype !== 'success') {
|
|
707
1022
|
if (m.subtype === 'error_max_turns')
|
|
@@ -787,6 +1102,13 @@ async function runOneSession(opts, resume, costTotals, progress) {
|
|
|
787
1102
|
providerSet?.close();
|
|
788
1103
|
}
|
|
789
1104
|
}
|
|
1105
|
+
/** ADR-O001 D7: org_task's description suffix for an org with a catalog. */
|
|
1106
|
+
function loadoutHelp(catalog) {
|
|
1107
|
+
const list = catalog
|
|
1108
|
+
.map((l) => (l.description ? `${l.name} (${l.description})` : l.name))
|
|
1109
|
+
.join(', ');
|
|
1110
|
+
return ` Optionally select a "loadout" — the named, stable specialisation the assignee's session is built with: ${list}. Select by kind of work; put everything specific to this task (which diff, criteria, what failed last time) in the title or a message, not in the choice of loadout. The selection is recorded on the task and reused on every retry.`;
|
|
1111
|
+
}
|
|
790
1112
|
/** Build the org tool surface as platform-agnostic OrgToolDef[]. The handlers
|
|
791
1113
|
* close over sessionOpts callbacks (deliver, recall, remember, …) — same
|
|
792
1114
|
* wiring as the previous inline createSdkMcpServer block, just decoupled from
|
|
@@ -912,22 +1234,64 @@ export function buildOrgTools(opts) {
|
|
|
912
1234
|
handler: async (args) => text(await onGate(role.id, args.name, args.description)),
|
|
913
1235
|
});
|
|
914
1236
|
}
|
|
1237
|
+
// ADR-O001 D7: the `loadout` argument exists only for an org with a
|
|
1238
|
+
// catalog, so every other org's tool list stays byte-identical.
|
|
1239
|
+
const catalog = opts.loadoutCatalog?.length ? opts.loadoutCatalog : undefined;
|
|
1240
|
+
const loadoutArg = catalog
|
|
1241
|
+
? { loadout: z.enum(catalog.map((l) => l.name)).optional() }
|
|
1242
|
+
: {};
|
|
915
1243
|
const createTask = opts.createTask;
|
|
916
1244
|
if (createTask) {
|
|
917
1245
|
tools.push({
|
|
918
1246
|
name: 'org_task',
|
|
919
|
-
description: 'Create a task in the DAG with optional dependencies. Dependencies must be existing task IDs. Tasks become ready when all deps are done, then get dispatched to the assignee.'
|
|
920
|
-
|
|
921
|
-
|
|
1247
|
+
description: 'Create a task in the DAG with optional dependencies. Dependencies must be existing task IDs. Tasks become ready when all deps are done, then get dispatched to the assignee.' +
|
|
1248
|
+
(catalog ? loadoutHelp(catalog) : ''),
|
|
1249
|
+
schema: {
|
|
1250
|
+
title: z.string(),
|
|
1251
|
+
assignee: z.string(),
|
|
1252
|
+
deps: z.array(z.string()).default([]),
|
|
1253
|
+
...loadoutArg,
|
|
1254
|
+
},
|
|
1255
|
+
handler: async (args) => text(createTask(role.id, args.title, args.assignee, args.deps ?? [], args.loadout)),
|
|
922
1256
|
});
|
|
923
1257
|
}
|
|
924
1258
|
const completeTask = opts.completeTask;
|
|
925
1259
|
if (completeTask) {
|
|
926
1260
|
tools.push({
|
|
927
1261
|
name: 'org_task_done',
|
|
928
|
-
description:
|
|
929
|
-
|
|
930
|
-
|
|
1262
|
+
description: opts.requireTaskEvidence
|
|
1263
|
+
? 'Mark a task as completed. This org requires EVIDENCE (run_config.completion_evidence): pass `evidence` with the current commit sha and one entry per acceptance criterion — the command you actually ran, its real exit code, and its output. Evidence pinned to an older commit is stale and will be refused, and a refused completion puts the task back in your queue with the reason — but only up to run_config.max_evidence_attempts times (default 3), after which the task is recorded as failed and escalated to the boss instead of returned to you. Any downstream tasks whose deps are now all done become ready and are dispatched.'
|
|
1264
|
+
: 'Mark a task as completed and optionally provide a result summary. Any downstream tasks whose deps are now all done will become ready and be dispatched.',
|
|
1265
|
+
schema: {
|
|
1266
|
+
taskId: z.string(),
|
|
1267
|
+
result: z.string().optional(),
|
|
1268
|
+
evidence: z
|
|
1269
|
+
.object({
|
|
1270
|
+
headSha: z.string(),
|
|
1271
|
+
checks: z
|
|
1272
|
+
.array(z.object({
|
|
1273
|
+
command: z.string(),
|
|
1274
|
+
exitCode: z.number().int(),
|
|
1275
|
+
output: z.string().optional(),
|
|
1276
|
+
}))
|
|
1277
|
+
.default([]),
|
|
1278
|
+
})
|
|
1279
|
+
.optional(),
|
|
1280
|
+
},
|
|
1281
|
+
handler: async (args) => text(completeTask(role.id, args.taskId, args.result, args.evidence)),
|
|
1282
|
+
});
|
|
1283
|
+
}
|
|
1284
|
+
const requestReview = opts.requestReview;
|
|
1285
|
+
if (requestReview) {
|
|
1286
|
+
tools.push({
|
|
1287
|
+
name: 'org_review',
|
|
1288
|
+
description: "Ask an artifact-only reviewer for a verdict on a task. You pass ids only: the runtime builds the review from the task's text, its assignee's latest org_task_done evidence (commands, exit codes, output) and its own git diff of base...headSha — nothing you write is added, so there is no summary to give. The reviewer starts cold every time and replies to you with org_send. Refused if the task has no evidence yet.",
|
|
1289
|
+
schema: {
|
|
1290
|
+
taskId: z.string(),
|
|
1291
|
+
reviewer: z.string(),
|
|
1292
|
+
base: z.string().optional().describe("git ref to diff against (default 'main')"),
|
|
1293
|
+
},
|
|
1294
|
+
handler: async (args) => text(requestReview(role.id, args.taskId, args.reviewer, args.base)),
|
|
931
1295
|
});
|
|
932
1296
|
}
|
|
933
1297
|
const listTasks = opts.listTasks;
|
|
@@ -982,7 +1346,8 @@ export function buildOrgTools(opts) {
|
|
|
982
1346
|
if (planGraph) {
|
|
983
1347
|
tools.push({
|
|
984
1348
|
name: 'org_plan_graph',
|
|
985
|
-
description: 'Propose a full work graph in one call. Each task spec uses a local "name" and references other specs by name in "after".'
|
|
1349
|
+
description: 'Propose a full work graph in one call. Each task spec uses a local "name" and references other specs by name in "after".' +
|
|
1350
|
+
(catalog ? ' Each spec may select a "loadout" exactly as org_task does.' : ''),
|
|
986
1351
|
schema: {
|
|
987
1352
|
tasks: z
|
|
988
1353
|
.array(z.object({
|
|
@@ -990,11 +1355,11 @@ export function buildOrgTools(opts) {
|
|
|
990
1355
|
title: z.string(),
|
|
991
1356
|
assignee: z.string(),
|
|
992
1357
|
after: z.array(z.string()).default([]),
|
|
1358
|
+
...loadoutArg,
|
|
993
1359
|
}))
|
|
994
1360
|
.min(1),
|
|
995
1361
|
},
|
|
996
|
-
handler: async (args) => text(planGraph(role.id, args.tasks ??
|
|
997
|
-
[])),
|
|
1362
|
+
handler: async (args) => text(planGraph(role.id, args.tasks ?? [])),
|
|
998
1363
|
});
|
|
999
1364
|
}
|
|
1000
1365
|
tools.push({
|
|
@@ -1015,12 +1380,17 @@ export function buildOrgTools(opts) {
|
|
|
1015
1380
|
});
|
|
1016
1381
|
tools.push({
|
|
1017
1382
|
name: 'ask_human',
|
|
1018
|
-
description: 'Ask a human a free-form question
|
|
1019
|
-
|
|
1383
|
+
description: 'Ask a human a free-form question. Use only when you genuinely need human judgment. ' +
|
|
1384
|
+
'Set blocking: true ONLY if you cannot continue until it is answered — a blocking question pauses the ' +
|
|
1385
|
+
"org's idle watchdog (for up to an hour; after that the run resumes its normal idle checks either way). " +
|
|
1386
|
+
'If you can keep working while you wait — an FYI, a preference, anything you would describe as "not blocking on this" — ' +
|
|
1387
|
+
'pass blocking: false and carry on; the question is still recorded and answered, it just does not freeze the run. ' +
|
|
1388
|
+
'Defaults to blocking.',
|
|
1389
|
+
schema: { question: z.string(), blocking: z.boolean().optional() },
|
|
1020
1390
|
handler: async (args) => {
|
|
1021
1391
|
if (!opts.askHuman)
|
|
1022
1392
|
return text('ask_human is not available in this session');
|
|
1023
|
-
const receipt = await opts.askHuman(role.id, args.question);
|
|
1393
|
+
const receipt = await opts.askHuman(role.id, args.question, args.blocking);
|
|
1024
1394
|
return text(receipt);
|
|
1025
1395
|
},
|
|
1026
1396
|
});
|