@ashlr/hub 2.2.0 → 3.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (177) hide show
  1. package/CHANGELOG.md +55 -0
  2. package/README.md +59 -2
  3. package/dist/api/plugin.d.ts +1 -0
  4. package/dist/cli/backlog.js +2 -0
  5. package/dist/cli/backlog.js.map +1 -1
  6. package/dist/cli/digest.d.ts +6 -0
  7. package/dist/cli/digest.js +127 -2
  8. package/dist/cli/digest.js.map +1 -1
  9. package/dist/cli/eval-fixtures.d.ts +18 -0
  10. package/dist/cli/eval-fixtures.js +38 -0
  11. package/dist/cli/eval-fixtures.js.map +1 -0
  12. package/dist/cli/eval.d.ts +56 -0
  13. package/dist/cli/eval.js +283 -0
  14. package/dist/cli/eval.js.map +1 -0
  15. package/dist/cli/fleet.d.ts +27 -0
  16. package/dist/cli/fleet.js +222 -0
  17. package/dist/cli/fleet.js.map +1 -0
  18. package/dist/cli/goal.d.ts +15 -0
  19. package/dist/cli/goal.js +91 -0
  20. package/dist/cli/goal.js.map +1 -0
  21. package/dist/cli/help.js +10 -3
  22. package/dist/cli/help.js.map +1 -1
  23. package/dist/cli/inbox.d.ts +6 -0
  24. package/dist/cli/inbox.js +183 -8
  25. package/dist/cli/inbox.js.map +1 -1
  26. package/dist/cli/index.js +42 -0
  27. package/dist/cli/index.js.map +1 -1
  28. package/dist/cli/loop.d.ts +16 -0
  29. package/dist/cli/loop.js +66 -0
  30. package/dist/cli/loop.js.map +1 -0
  31. package/dist/cli/mcp.d.ts +12 -0
  32. package/dist/cli/mcp.js +148 -0
  33. package/dist/cli/mcp.js.map +1 -1
  34. package/dist/cli/onboard.d.ts +16 -0
  35. package/dist/cli/onboard.js +71 -0
  36. package/dist/cli/onboard.js.map +1 -1
  37. package/dist/cli/plugins.js +9 -8
  38. package/dist/cli/plugins.js.map +1 -1
  39. package/dist/cli/pulse.d.ts +10 -0
  40. package/dist/cli/pulse.js +261 -0
  41. package/dist/cli/pulse.js.map +1 -1
  42. package/dist/cli/run.js +21 -3
  43. package/dist/cli/run.js.map +1 -1
  44. package/dist/cli/stack.d.ts +34 -0
  45. package/dist/cli/stack.js +221 -0
  46. package/dist/cli/stack.js.map +1 -0
  47. package/dist/cli/update.js +12 -5
  48. package/dist/cli/update.js.map +1 -1
  49. package/dist/core/daemon/loop.d.ts +10 -5
  50. package/dist/core/daemon/loop.js +167 -37
  51. package/dist/core/daemon/loop.js.map +1 -1
  52. package/dist/core/env-bridge.js +3 -2
  53. package/dist/core/env-bridge.js.map +1 -1
  54. package/dist/core/fleet/automerge-pass.d.ts +33 -0
  55. package/dist/core/fleet/automerge-pass.js +61 -0
  56. package/dist/core/fleet/automerge-pass.js.map +1 -0
  57. package/dist/core/fleet/quota.d.ts +70 -0
  58. package/dist/core/fleet/quota.js +198 -0
  59. package/dist/core/fleet/quota.js.map +1 -0
  60. package/dist/core/fleet/router.d.ts +46 -0
  61. package/dist/core/fleet/router.js +121 -0
  62. package/dist/core/fleet/router.js.map +1 -0
  63. package/dist/core/fleet/self.d.ts +58 -0
  64. package/dist/core/fleet/self.js +187 -0
  65. package/dist/core/fleet/self.js.map +1 -0
  66. package/dist/core/fleet/status.d.ts +54 -0
  67. package/dist/core/fleet/status.js +157 -0
  68. package/dist/core/fleet/status.js.map +1 -0
  69. package/dist/core/foundry/provenance.d.ts +78 -0
  70. package/dist/core/foundry/provenance.js +172 -0
  71. package/dist/core/foundry/provenance.js.map +1 -0
  72. package/dist/core/git.d.ts +13 -0
  73. package/dist/core/git.js +26 -0
  74. package/dist/core/git.js.map +1 -1
  75. package/dist/core/inbox/merge.d.ts +151 -0
  76. package/dist/core/inbox/merge.js +858 -0
  77. package/dist/core/inbox/merge.js.map +1 -0
  78. package/dist/core/inbox/store.js +10 -1
  79. package/dist/core/inbox/store.js.map +1 -1
  80. package/dist/core/integrations/markdown.d.ts +66 -0
  81. package/dist/core/integrations/markdown.js +132 -0
  82. package/dist/core/integrations/markdown.js.map +1 -0
  83. package/dist/core/integrations/secrets.d.ts +37 -0
  84. package/dist/core/integrations/secrets.js +67 -0
  85. package/dist/core/integrations/secrets.js.map +1 -0
  86. package/dist/core/integrations/stack.d.ts +47 -0
  87. package/dist/core/integrations/stack.js +107 -0
  88. package/dist/core/integrations/stack.js.map +1 -0
  89. package/dist/core/mcp-native-engineer.d.ts +123 -0
  90. package/dist/core/mcp-native-engineer.js +936 -0
  91. package/dist/core/mcp-native-engineer.js.map +1 -0
  92. package/dist/core/mcp-native.js +20 -42
  93. package/dist/core/mcp-native.js.map +1 -1
  94. package/dist/core/mcp-registry.js +2 -0
  95. package/dist/core/mcp-registry.js.map +1 -1
  96. package/dist/core/observability/codex-source.d.ts +53 -0
  97. package/dist/core/observability/codex-source.js +390 -0
  98. package/dist/core/observability/codex-source.js.map +1 -0
  99. package/dist/core/observability/estimate.d.ts +12 -0
  100. package/dist/core/observability/estimate.js +26 -0
  101. package/dist/core/observability/estimate.js.map +1 -1
  102. package/dist/core/observability/limits.d.ts +52 -0
  103. package/dist/core/observability/limits.js +316 -0
  104. package/dist/core/observability/limits.js.map +1 -0
  105. package/dist/core/observability/usage-source.js +3 -1
  106. package/dist/core/observability/usage-source.js.map +1 -1
  107. package/dist/core/portfolio/scanners.d.ts +1 -0
  108. package/dist/core/portfolio/scanners.js +40 -1
  109. package/dist/core/portfolio/scanners.js.map +1 -1
  110. package/dist/core/providers.js +4 -0
  111. package/dist/core/providers.js.map +1 -1
  112. package/dist/core/quality/health.js +1 -0
  113. package/dist/core/quality/health.js.map +1 -1
  114. package/dist/core/run/agent-loop.d.ts +8 -0
  115. package/dist/core/run/agent-loop.js +24 -14
  116. package/dist/core/run/agent-loop.js.map +1 -1
  117. package/dist/core/run/engine-registry.d.ts +48 -0
  118. package/dist/core/run/engine-registry.js +190 -0
  119. package/dist/core/run/engine-registry.js.map +1 -0
  120. package/dist/core/run/engines.d.ts +17 -3
  121. package/dist/core/run/engines.js +63 -44
  122. package/dist/core/run/engines.js.map +1 -1
  123. package/dist/core/run/learned-router.d.ts +121 -0
  124. package/dist/core/run/learned-router.js +374 -0
  125. package/dist/core/run/learned-router.js.map +1 -0
  126. package/dist/core/run/model-manager.js +3 -3
  127. package/dist/core/run/model-manager.js.map +1 -1
  128. package/dist/core/run/model-profile.d.ts +67 -0
  129. package/dist/core/run/model-profile.js +132 -0
  130. package/dist/core/run/model-profile.js.map +1 -0
  131. package/dist/core/run/orchestrator.d.ts +1 -1
  132. package/dist/core/run/orchestrator.js +642 -510
  133. package/dist/core/run/orchestrator.js.map +1 -1
  134. package/dist/core/run/prompts/budget.d.ts +25 -0
  135. package/dist/core/run/prompts/budget.js +51 -0
  136. package/dist/core/run/prompts/budget.js.map +1 -0
  137. package/dist/core/run/prompts/index.d.ts +14 -0
  138. package/dist/core/run/prompts/index.js +59 -0
  139. package/dist/core/run/prompts/index.js.map +1 -0
  140. package/dist/core/run/prompts/layers.d.ts +26 -0
  141. package/dist/core/run/prompts/layers.js +73 -0
  142. package/dist/core/run/prompts/layers.js.map +1 -0
  143. package/dist/core/run/prompts/roles.d.ts +23 -0
  144. package/dist/core/run/prompts/roles.js +48 -0
  145. package/dist/core/run/prompts/roles.js.map +1 -0
  146. package/dist/core/run/prompts/types.d.ts +38 -0
  147. package/dist/core/run/prompts/types.js +5 -0
  148. package/dist/core/run/prompts/types.js.map +1 -0
  149. package/dist/core/run/provider-client.d.ts +6 -24
  150. package/dist/core/run/provider-client.js +125 -61
  151. package/dist/core/run/provider-client.js.map +1 -1
  152. package/dist/core/run/router.js +8 -0
  153. package/dist/core/run/router.js.map +1 -1
  154. package/dist/core/run/sandboxed-engine.d.ts +66 -0
  155. package/dist/core/run/sandboxed-engine.js +262 -0
  156. package/dist/core/run/sandboxed-engine.js.map +1 -0
  157. package/dist/core/run/verify-commands.d.ts +83 -0
  158. package/dist/core/run/verify-commands.js +234 -0
  159. package/dist/core/run/verify-commands.js.map +1 -0
  160. package/dist/core/run/verify.d.ts +31 -1
  161. package/dist/core/run/verify.js +48 -0
  162. package/dist/core/run/verify.js.map +1 -1
  163. package/dist/core/sandbox/audit.d.ts +27 -0
  164. package/dist/core/sandbox/audit.js +34 -0
  165. package/dist/core/sandbox/audit.js.map +1 -1
  166. package/dist/core/sandbox/confine.d.ts +134 -0
  167. package/dist/core/sandbox/confine.js +370 -0
  168. package/dist/core/sandbox/confine.js.map +1 -0
  169. package/dist/core/types.d.ts +227 -5
  170. package/dist/core/web/api.js +77 -28
  171. package/dist/core/web/api.js.map +1 -1
  172. package/dist/core/web/control.d.ts +116 -0
  173. package/dist/core/web/control.js +307 -0
  174. package/dist/core/web/control.js.map +1 -0
  175. package/dist/core/web/public/app.js +512 -6
  176. package/dist/core/web/public/styles.css +468 -0
  177. package/package.json +1 -1
@@ -67,8 +67,19 @@ import { withToolEnv } from '../env-bridge.js';
67
67
  import { buildEngineCommand, engineInstalled, spawnEngine } from './engines.js';
68
68
  import { nullSink } from './streaming.js';
69
69
  import { withRetry } from './retry.js';
70
- import { verifyTask } from './verify.js';
70
+ import { verifyTaskStructured } from './verify.js';
71
71
  import { withHeal, defaultHealPolicy } from './self-heal.js';
72
+ import { PLANNER_ROLE, SYNTHESIZER_ROLE } from './prompts/roles.js';
73
+ import { systemPromptFor } from './prompts/index.js';
74
+ import { resolveModelProfile, adaptivePromptsEnabled } from './model-profile.js';
75
+ // M42: executable, sandboxed engineering tool surface.
76
+ import { buildEngineerToolSpecs, buildNativeToolSpecsWithFn, } from '../mcp-native-engineer.js';
77
+ import { listNativeTools } from '../mcp-native.js';
78
+ import { selectInboxStore } from '../seams/inbox.js';
79
+ import { scrubSecrets } from '../knowledge/index.js';
80
+ // NOTE: sandbox/worktree.js is imported DYNAMICALLY inside runGoal (matching the
81
+ // swarm runner) so its absence degrades gracefully at runtime instead of
82
+ // becoming a hard load-time dependency (H4 simulates the module being absent).
72
83
  // ---------------------------------------------------------------------------
73
84
  // Constants / defaults
74
85
  // ---------------------------------------------------------------------------
@@ -340,25 +351,12 @@ async function routeTask(taskGoal, cfg, opts, fallbackClient) {
340
351
  // ---------------------------------------------------------------------------
341
352
  // Planning
342
353
  // ---------------------------------------------------------------------------
343
- /** Prompt template for decomposing a goal into a task DAG. */
344
- const PLANNING_SYSTEM = `You are a task planner. Decompose the user's goal into 1-6 subtasks that together accomplish it.
345
- Respond ONLY with a JSON array. Each element must have:
346
- "id": string (unique short slug, e.g. "t1", "t2"),
347
- "goal": string (clear sub-goal for this task),
348
- "deps": string[] (ids of tasks that must complete before this one; empty for root tasks)
349
-
350
- Rules:
351
- - deps must reference earlier ids only (no cycles).
352
- - Keep tasks focused and independently executable.
353
- - Use a minimal number of tasks (don't over-decompose).
354
-
355
- Example:
356
- [
357
- {"id":"t1","goal":"Research the topic","deps":[]},
358
- {"id":"t2","goal":"Summarize findings","deps":["t1"]}
359
- ]
360
-
361
- Return ONLY the JSON array — no prose, no markdown fences.`;
354
+ /**
355
+ * Prompt template for decomposing a goal into a task DAG.
356
+ * M41: single-sourced from prompts/roles.PLANNER_ROLE (verbatim) so the legacy
357
+ * (flag-off) and adaptive (flag-on) paths share identical planner text.
358
+ */
359
+ const PLANNING_SYSTEM = PLANNER_ROLE;
362
360
  /**
363
361
  * Parse a RunTask[] from model output, tolerating prose wrapped around JSON.
364
362
  * Returns null if no valid JSON array of tasks is found.
@@ -457,11 +455,19 @@ function hasCycle(tasks) {
457
455
  * When non-empty, injects "Relevant project memory:" context so the planner
458
456
  * benefits from cross-project knowledge. Kept bounded upstream (GENOME_INJECT_CHAR_CAP).
459
457
  */
460
- export async function planGoal(goal, client, onUsage, memoryContext) {
461
- // Prepend memory block when present (bounded by caller)
462
- const systemContent = memoryContext && memoryContext.length > 0
463
- ? `${memoryContext}\n\n${PLANNING_SYSTEM}`
464
- : PLANNING_SYSTEM;
458
+ export async function planGoal(goal, client, onUsage, memoryContext, adaptive) {
459
+ // M41: adaptive path budgets the memory block via the prompt suite; the legacy
460
+ // path keeps the original prepend behavior byte-for-byte.
461
+ const systemContent = adaptive
462
+ ? systemPromptFor({
463
+ role: 'planner',
464
+ useTools: false,
465
+ profile: resolveModelProfile(client.model),
466
+ memory: memoryContext,
467
+ })
468
+ : memoryContext && memoryContext.length > 0
469
+ ? `${memoryContext}\n\n${PLANNING_SYSTEM}`
470
+ : PLANNING_SYSTEM;
465
471
  const messages = [
466
472
  { role: 'system', content: systemContent },
467
473
  { role: 'user', content: goal },
@@ -490,8 +496,8 @@ export async function planGoal(goal, client, onUsage, memoryContext) {
490
496
  // ---------------------------------------------------------------------------
491
497
  // Synthesis
492
498
  // ---------------------------------------------------------------------------
493
- const SYNTHESIS_SYSTEM = `You are a helpful assistant. The user asked a goal and several subtasks were executed to answer it.
494
- Combine the results into a single, coherent final answer. Be concise and accurate.`;
499
+ // M41: single-sourced from prompts/roles.SYNTHESIZER_ROLE (verbatim).
500
+ const SYNTHESIS_SYSTEM = SYNTHESIZER_ROLE;
495
501
  /**
496
502
  * Synthesize a final answer from completed task results.
497
503
  * Returns a best-effort string even if the model call fails.
@@ -625,7 +631,7 @@ async function checkGovernance(cfg, overBudgetFlag) {
625
631
  */
626
632
  function isBinaryInstalled(name) {
627
633
  try {
628
- execFileSync('which', [name], { stdio: 'ignore' });
634
+ execFileSync(process.platform === 'win32' ? 'where' : 'which', [name], { stdio: 'ignore' });
629
635
  return true;
630
636
  }
631
637
  catch {
@@ -644,7 +650,19 @@ function emit(sink, event) {
644
650
  }
645
651
  }
646
652
  /** Known engine ids (typed subset). */
647
- const KNOWN_ENGINE_IDS = new Set(['builtin', 'ashlrcode', 'aw', 'claude']);
653
+ const KNOWN_ENGINE_IDS = new Set(['builtin', 'ashlrcode', 'aw', 'claude', 'codex']);
654
+ /**
655
+ * M45: whether an external engine should run SANDBOXED (worktree + diff→inbox)
656
+ * instead of raw on the live tree. True only when cfg.foundry opts in; default
657
+ * (absent) keeps the raw delegation path → today's behavior unchanged.
658
+ */
659
+ function foundryWantsSandbox(cfg, engine) {
660
+ if (engine === 'builtin')
661
+ return false;
662
+ if (!cfg.foundry)
663
+ return false;
664
+ return cfg.foundry.sandboxExternal ?? true;
665
+ }
648
666
  // ---------------------------------------------------------------------------
649
667
  // Main: runGoal
650
668
  // ---------------------------------------------------------------------------
@@ -765,6 +783,28 @@ export async function runGoal(goal, cfg, opts) {
765
783
  else {
766
784
  process.stderr.write(`[ashlr run] delegating to engine "${engine}" (${goal.slice(0, 60)}…)\n`);
767
785
  emit(sink, { kind: 'log', text: `delegating to engine "${engine}"` });
786
+ // M45: sandboxed-external path — run the agent CLI confined to a throwaway
787
+ // worktree and capture its diff as a PENDING proposal. Gated by
788
+ // opts.sandboxEngine or cfg.foundry; when neither is set the raw delegation
789
+ // below runs unchanged (today's behavior). No raw fallback here: defeating
790
+ // the sandbox would break the no-outward guarantee for autonomous runs.
791
+ if (opts.sandboxEngine === true || foundryWantsSandbox(cfg, engineId)) {
792
+ const { runEngineSandboxed } = await import('./sandboxed-engine.js');
793
+ const r = await runEngineSandboxed(engineId, goal, cfg, {
794
+ sourceRepo: cwd,
795
+ model: modelEnv,
796
+ budget: opts.budget,
797
+ propose: true,
798
+ });
799
+ emit(sink, {
800
+ kind: r.state.status === 'done' ? 'task-done' : 'log',
801
+ text: r.proposalId
802
+ ? `engine "${engine}" → inbox proposal ${r.proposalId}`
803
+ : `engine "${engine}" ${r.state.status}`,
804
+ });
805
+ saveRun(r.state);
806
+ return r.state;
807
+ }
768
808
  const id = generateRunId();
769
809
  const now = new Date().toISOString();
770
810
  const delegatedState = {
@@ -906,468 +946,633 @@ export async function runGoal(goal, cfg, opts) {
906
946
  saveRun(state);
907
947
  }
908
948
  // -- Tool wiring (optional) --------------------------------------------------
909
- // When opts.tools !== false, attempt to connect to the MCP gateway as a client.
910
- // On any failure, continue tool-free with a warning.
949
+ // Default: spec-only gateway tools (exactly today's behavior). With
950
+ // opts.engineer, the hub loop gets an EXECUTABLE, sandboxed surface: native
951
+ // tools (with fn) + downstream specs + engineering tools (read/glob/grep +
952
+ // sandboxed write/edit, and with allowBash also bash). All writes/exec are
953
+ // confined to a throwaway git worktree; the captured diff is routed to the
954
+ // approval inbox at the end — nothing reaches the live tree unapproved.
911
955
  let tools;
956
+ let activeSandbox = null;
957
+ let sandboxModule = null;
958
+ let engCtx;
912
959
  if (opts.tools !== false && client.supportsTools) {
913
- try {
914
- tools = await loadGatewayTools(cfg);
960
+ if (opts.engineer === true) {
961
+ const sourceRepo = opts.cwd && path.isAbsolute(opts.cwd) && fs.existsSync(opts.cwd)
962
+ ? opts.cwd
963
+ : process.cwd();
964
+ try {
965
+ sandboxModule = await import('../sandbox/worktree.js');
966
+ activeSandbox = sandboxModule.createSandbox(sourceRepo);
967
+ engCtx = {
968
+ workspaceRoot: activeSandbox.worktreePath,
969
+ sourceRepo,
970
+ allowWrite: true,
971
+ allowExec: opts.allowBash === true,
972
+ };
973
+ const gateway = await loadGatewayTools(cfg).catch(() => []);
974
+ const nativeNames = new Set(listNativeTools().map((t) => t.name));
975
+ const downstreamOnly = gateway.filter((t) => !nativeNames.has(t.function?.name ?? ''));
976
+ tools = [
977
+ ...buildNativeToolSpecsWithFn(),
978
+ ...downstreamOnly,
979
+ ...buildEngineerToolSpecs(engCtx),
980
+ ];
981
+ process.stderr.write(`[ashlr run] --engineer: sandboxed tools active in ${activeSandbox.worktreePath}` +
982
+ `${engCtx.allowExec ? ' (bash enabled)' : ''}\n`);
983
+ }
984
+ catch (err) {
985
+ const msg = err instanceof Error ? err.message : String(err);
986
+ process.stderr.write(`[ashlr run] --engineer unavailable (${msg}) — enroll the repo (ashlr enroll) ` +
987
+ `and clear the kill switch; continuing with read-only gateway tools\n`);
988
+ activeSandbox = null;
989
+ sandboxModule = null;
990
+ engCtx = undefined;
991
+ try {
992
+ tools = await loadGatewayTools(cfg);
993
+ }
994
+ catch {
995
+ tools = undefined;
996
+ }
997
+ }
915
998
  }
916
- catch (err) {
917
- const msg = err instanceof Error ? err.message : String(err);
918
- process.stderr.write(`[ashlr run] tool gateway unavailable (${msg}) — continuing tool-free\n`);
919
- tools = undefined;
999
+ else {
1000
+ try {
1001
+ tools = await loadGatewayTools(cfg);
1002
+ }
1003
+ catch (err) {
1004
+ const msg = err instanceof Error ? err.message : String(err);
1005
+ process.stderr.write(`[ashlr run] tool gateway unavailable (${msg}) — continuing tool-free\n`);
1006
+ tools = undefined;
1007
+ }
920
1008
  }
921
1009
  }
922
- // -- M16/M7: Genome memory injection (best-effort, bounded, local-only) ------
923
- // M16: prefer a synthesized playbook over raw recall when playbookOnRun is on.
924
- // Falls back to the existing raw-recall block on any playbook failure.
925
- // Skipped when: noMemory is set, cfg disables injection, or this is a resume
926
- // with existing tasks (context was already embedded in those task goals).
927
- let memoryContext = '';
928
- const injectOnRun = cfg.genome?.injectOnRun ?? true;
929
- if (!noMemory && injectOnRun && state.tasks.length === 0) {
930
- // Only attempt playbook injection when genome is explicitly configured and
931
- // playbookOnRun is not disabled. When cfg.genome is absent there is nothing
932
- // to recall, and the playbook module makes local Ollama fetch calls even on
933
- // an empty recall — which would interfere with scripted fetch mocks in tests
934
- // and add unnecessary latency in unconfigured environments.
935
- const playbookOnRun = cfg.genome != null && cfg.genome.playbookOnRun !== false;
936
- let playbookInjected = false;
937
- if (playbookOnRun) {
938
- try {
939
- // Dynamic import: tolerates the module being absent (pre-M16 build).
940
- // eslint-disable-next-line @typescript-eslint/no-explicit-any
941
- const pbMod = await import('../genome/playbook.js');
942
- if (typeof pbMod.buildPlaybook === 'function' &&
943
- typeof pbMod.playbookText === 'function') {
944
- const playbook = await pbMod.buildPlaybook(goal, cfg);
945
- const pbText = pbMod.playbookText(playbook, GENOME_INJECT_CHAR_CAP);
946
- if (pbText && pbText.length > 0) {
947
- memoryContext = pbText;
948
- playbookInjected = true;
949
- process.stderr.write(`[ashlr run] genome: injecting ${memoryContext.length} chars of playbook context\n`);
950
- }
1010
+ // M42: capture the sandbox diff into the approval inbox + tear down. Idempotent
1011
+ // and best-effort — never throws, called on every runGoal exit path.
1012
+ const finalizeEngineer = () => {
1013
+ if (!activeSandbox || !sandboxModule)
1014
+ return;
1015
+ const sb = activeSandbox;
1016
+ const wt = sandboxModule;
1017
+ activeSandbox = null;
1018
+ try {
1019
+ const diff = wt.sandboxDiff(sb);
1020
+ if (diff.files > 0 && diff.patch.trim().length > 0) {
1021
+ try {
1022
+ const proposal = selectInboxStore(cfg).create({
1023
+ repo: sb.sourceRepo,
1024
+ origin: 'agent',
1025
+ kind: 'patch',
1026
+ title: `engineer run: ${goal.slice(0, 80)}`,
1027
+ summary: `Sandboxed --engineer run produced ${diff.files} file(s) ` +
1028
+ `(+${diff.insertions}/-${diff.deletions}). Review before applying.`,
1029
+ diff: scrubSecrets(diff.patch),
1030
+ sandboxId: sb.id,
1031
+ });
1032
+ process.stderr.write(`[ashlr run] engineer diff → inbox proposal ${proposal.id} ` +
1033
+ `(${diff.files} files); review with 'ashlr inbox'\n`);
1034
+ }
1035
+ catch (err) {
1036
+ process.stderr.write(`[ashlr run] could not file engineer proposal: ${String(err)}\n`);
951
1037
  }
952
1038
  }
1039
+ }
1040
+ catch {
1041
+ // diff capture best-effort
1042
+ }
1043
+ finally {
1044
+ try {
1045
+ wt.removeSandbox(sb);
1046
+ }
953
1047
  catch {
954
- // Playbook module absent or failed — fall through to raw recall below.
1048
+ // removal is idempotent
955
1049
  }
956
1050
  }
957
- if (!playbookInjected) {
958
- // M7 fallback: raw recall injection.
959
- memoryContext = await buildMemoryBlock(goal, cfg);
960
- if (memoryContext.length > 0) {
961
- process.stderr.write(`[ashlr run] genome: injecting ${memoryContext.length} chars of memory context\n`);
1051
+ };
1052
+ // M42: guarantee sandbox teardown + diff capture on EVERY exit path, including
1053
+ // a throw from planning/synthesis. finalizeEngineer is idempotent and a no-op
1054
+ // when no sandbox is active, so this is safe for non-engineer runs too.
1055
+ try {
1056
+ // M41: resolve once — gates adaptive prompts for planning AND every task.
1057
+ const adaptivePrompts = adaptivePromptsEnabled(cfg);
1058
+ // -- M16/M7: Genome memory injection (best-effort, bounded, local-only) ------
1059
+ // M16: prefer a synthesized playbook over raw recall when playbookOnRun is on.
1060
+ // Falls back to the existing raw-recall block on any playbook failure.
1061
+ // Skipped when: noMemory is set, cfg disables injection, or this is a resume
1062
+ // with existing tasks (context was already embedded in those task goals).
1063
+ let memoryContext = '';
1064
+ const injectOnRun = cfg.genome?.injectOnRun ?? true;
1065
+ if (!noMemory && injectOnRun && state.tasks.length === 0) {
1066
+ // Only attempt playbook injection when genome is explicitly configured and
1067
+ // playbookOnRun is not disabled. When cfg.genome is absent there is nothing
1068
+ // to recall, and the playbook module makes local Ollama fetch calls even on
1069
+ // an empty recall — which would interfere with scripted fetch mocks in tests
1070
+ // and add unnecessary latency in unconfigured environments.
1071
+ const playbookOnRun = cfg.genome != null && cfg.genome.playbookOnRun !== false;
1072
+ let playbookInjected = false;
1073
+ if (playbookOnRun) {
1074
+ try {
1075
+ // Dynamic import: tolerates the module being absent (pre-M16 build).
1076
+ // eslint-disable-next-line @typescript-eslint/no-explicit-any
1077
+ const pbMod = await import('../genome/playbook.js');
1078
+ if (typeof pbMod.buildPlaybook === 'function' &&
1079
+ typeof pbMod.playbookText === 'function') {
1080
+ const playbook = await pbMod.buildPlaybook(goal, cfg);
1081
+ const pbText = pbMod.playbookText(playbook, GENOME_INJECT_CHAR_CAP);
1082
+ if (pbText && pbText.length > 0) {
1083
+ memoryContext = pbText;
1084
+ playbookInjected = true;
1085
+ process.stderr.write(`[ashlr run] genome: injecting ${memoryContext.length} chars of playbook context\n`);
1086
+ }
1087
+ }
1088
+ }
1089
+ catch {
1090
+ // Playbook module absent or failed — fall through to raw recall below.
1091
+ }
962
1092
  }
963
- }
964
- }
965
- // -- Plan (unless resuming with existing tasks) ------------------------------
966
- if (state.tasks.length === 0) {
967
- const planStep = {
968
- ts: new Date().toISOString(),
969
- taskId: '__plan__',
970
- kind: 'plan',
971
- summary: `Planning: decomposing goal into tasks`,
972
- };
973
- state.steps.push(planStep);
974
- state.updatedAt = planStep.ts;
975
- saveRun(state);
976
- let planTokensIn = 0;
977
- let planTokensOut = 0;
978
- const tasks = await planGoal(goal, client, (u) => {
979
- planTokensIn = u.tokensIn;
980
- planTokensOut = u.tokensOut;
981
- }, memoryContext || undefined);
982
- state.tasks = tasks;
983
- // Charge the planning call to the run budget so usage/cost stay accurate.
984
- // (Previously the planning tokens were silently discarded.)
985
- // Accumulate incrementally (price ONLY the planning tokens at the planner's
986
- // provider) so this is consistent with the per-step accumulation below and
987
- // never re-prices later task tokens at the planner's provider.
988
- state.usage.tokensIn += planTokensIn;
989
- state.usage.tokensOut += planTokensOut;
990
- state.usage.steps += 1;
991
- state.usage.estCostUsd += estCostUsd(client.id, planTokensIn, planTokensOut);
992
- state.updatedAt = new Date().toISOString();
993
- const planDoneStep = {
994
- ts: state.updatedAt,
995
- taskId: '__plan__',
996
- kind: 'plan',
997
- summary: `Planned ${tasks.length} task(s): ${tasks.map((t) => t.id).join(', ')}`,
998
- usage: { tokensIn: planTokensIn, tokensOut: planTokensOut, steps: 1, estCostUsd: 0 },
999
- };
1000
- state.steps.push(planDoneStep);
1001
- cliOnStep?.(planDoneStep, state.tasks);
1002
- saveRun(state);
1003
- }
1004
- // -- DAG execution loop ------------------------------------------------------
1005
- let aborted = false;
1006
- while (!allTerminal(state.tasks) && !aborted) {
1007
- // Check global budget before picking next batch
1008
- if (overBudget(state.usage, budget)) {
1009
- aborted = true;
1010
- break;
1011
- }
1012
- const ready = readyTasks(state.tasks);
1013
- if (ready.length === 0) {
1014
- // No ready tasks but not all terminal — means some tasks have deps on
1015
- // failed/skipped tasks. Mark them skipped.
1016
- const pendingBlocked = state.tasks.filter((t) => t.status === 'pending');
1017
- if (pendingBlocked.length > 0) {
1018
- for (const t of pendingBlocked) {
1019
- t.status = 'skipped';
1020
- t.error = 'Dependency failed or was skipped';
1093
+ if (!playbookInjected) {
1094
+ // M7 fallback: raw recall injection.
1095
+ memoryContext = await buildMemoryBlock(goal, cfg);
1096
+ if (memoryContext.length > 0) {
1097
+ process.stderr.write(`[ashlr run] genome: injecting ${memoryContext.length} chars of memory context\n`);
1021
1098
  }
1022
- state.updatedAt = new Date().toISOString();
1023
- saveRun(state);
1024
1099
  }
1025
- break;
1026
1100
  }
1027
- // Run up to `parallel` tasks concurrently
1028
- const batch = ready.slice(0, parallel);
1029
- // Mark them running before spawning
1030
- for (const task of batch) {
1031
- task.status = 'running';
1101
+ // -- Plan (unless resuming with existing tasks) ------------------------------
1102
+ if (state.tasks.length === 0) {
1103
+ const planStep = {
1104
+ ts: new Date().toISOString(),
1105
+ taskId: '__plan__',
1106
+ kind: 'plan',
1107
+ summary: `Planning: decomposing goal into tasks`,
1108
+ };
1109
+ state.steps.push(planStep);
1110
+ state.updatedAt = planStep.ts;
1111
+ saveRun(state);
1112
+ let planTokensIn = 0;
1113
+ let planTokensOut = 0;
1114
+ const tasks = await planGoal(goal, client, (u) => {
1115
+ planTokensIn = u.tokensIn;
1116
+ planTokensOut = u.tokensOut;
1117
+ }, memoryContext || undefined, adaptivePrompts);
1118
+ state.tasks = tasks;
1119
+ // Charge the planning call to the run budget so usage/cost stay accurate.
1120
+ // (Previously the planning tokens were silently discarded.)
1121
+ // Accumulate incrementally (price ONLY the planning tokens at the planner's
1122
+ // provider) so this is consistent with the per-step accumulation below and
1123
+ // never re-prices later task tokens at the planner's provider.
1124
+ state.usage.tokensIn += planTokensIn;
1125
+ state.usage.tokensOut += planTokensOut;
1126
+ state.usage.steps += 1;
1127
+ state.usage.estCostUsd += estCostUsd(client.id, planTokensIn, planTokensOut);
1128
+ state.updatedAt = new Date().toISOString();
1129
+ const planDoneStep = {
1130
+ ts: state.updatedAt,
1131
+ taskId: '__plan__',
1132
+ kind: 'plan',
1133
+ summary: `Planned ${tasks.length} task(s): ${tasks.map((t) => t.id).join(', ')}`,
1134
+ usage: { tokensIn: planTokensIn, tokensOut: planTokensOut, steps: 1, estCostUsd: 0 },
1135
+ };
1136
+ state.steps.push(planDoneStep);
1137
+ cliOnStep?.(planDoneStep, state.tasks);
1138
+ saveRun(state);
1032
1139
  }
1033
- state.updatedAt = new Date().toISOString();
1034
- saveRun(state);
1035
- // Run the batch in parallel; each task must not crash the whole run
1036
- await Promise.all(batch.map(async (task) => {
1037
- try {
1038
- // M11: emit task-start event.
1039
- emit(sink, { kind: 'task-start', taskId: task.id, text: task.goal });
1040
- // M15: Choose route for this task (local-first; cloud only when
1041
- // allowCloud + escalation reason + key present). Best-effort — falls
1042
- // back to the run-level client when router is unavailable.
1043
- const { client: taskClient, decision: taskDecision } = await routeTask(task.goal, cfg, { allowCloud, attempt: 1, lastReason: 'none' }, client);
1044
- emit(sink, {
1045
- kind: 'log',
1046
- taskId: task.id,
1047
- text: `route: ${taskDecision.provider}/${taskDecision.model} [${taskDecision.tier}] — ${taskDecision.reason}`,
1048
- });
1049
- // Build per-task onStep callback (single-writer invariant preserved).
1050
- // M15: cost attribution uses the provider that actually served EACH step.
1051
- // We ACCUMULATE cost incrementally (+= this step's tokens priced at this
1052
- // step's provider) rather than recomputing estCostUsd over the cumulative
1053
- // run-wide totals at the current provider. Recomputing-from-cumulative is
1054
- // wrong for mixed local+cloud runs: it would re-price an earlier local
1055
- // task's tokens at a later cloud escalation's rates (over-charging), or
1056
- // re-price an earlier cloud task's tokens at $0 when a later step is local
1057
- // (erasing real spend). Incremental accumulation keeps local steps at $0
1058
- // regardless of any later cloud escalation, and prices cloud escalations
1059
- // on only the tokens they served.
1060
- const makeTaskOnStep = (providerForCost) => (step) => {
1061
- state.steps.push(step);
1062
- // SINGLE-WRITER INVARIANT: orchestrator is the only mutator of state.usage.
1063
- if (step.usage) {
1064
- state.usage.tokensIn += step.usage.tokensIn;
1065
- state.usage.tokensOut += step.usage.tokensOut;
1066
- state.usage.steps += step.usage.steps;
1067
- state.usage.estCostUsd += estCostUsd(providerForCost, step.usage.tokensIn, step.usage.tokensOut);
1140
+ // -- DAG execution loop ------------------------------------------------------
1141
+ let aborted = false;
1142
+ while (!allTerminal(state.tasks) && !aborted) {
1143
+ // Check global budget before picking next batch
1144
+ if (overBudget(state.usage, budget)) {
1145
+ aborted = true;
1146
+ break;
1147
+ }
1148
+ const ready = readyTasks(state.tasks);
1149
+ if (ready.length === 0) {
1150
+ // No ready tasks but not all terminal — means some tasks have deps on
1151
+ // failed/skipped tasks. Mark them skipped.
1152
+ const pendingBlocked = state.tasks.filter((t) => t.status === 'pending');
1153
+ if (pendingBlocked.length > 0) {
1154
+ for (const t of pendingBlocked) {
1155
+ t.status = 'skipped';
1156
+ t.error = 'Dependency failed or was skipped';
1068
1157
  }
1069
1158
  state.updatedAt = new Date().toISOString();
1070
- cliOnStep?.(step, state.tasks);
1071
1159
  saveRun(state);
1072
- };
1073
- let taskOnStep = makeTaskOnStep(taskDecision.provider);
1074
- // M11: Retry policy — bounded, budget-aware.
1075
- // We retry on transient/tool failures only; hard budget stops are not retryable.
1076
- const RETRY_POLICY = { maxAttempts: 2, baseDelayMs: 500 };
1077
- const isRetryable = (err) => {
1078
- // Don't retry if budget is already exhausted.
1079
- if (overBudget(state.usage, budget))
1160
+ }
1161
+ break;
1162
+ }
1163
+ // Run up to `parallel` tasks concurrently
1164
+ const batch = ready.slice(0, parallel);
1165
+ // Mark them running before spawning
1166
+ for (const task of batch) {
1167
+ task.status = 'running';
1168
+ }
1169
+ state.updatedAt = new Date().toISOString();
1170
+ saveRun(state);
1171
+ // Run the batch in parallel; each task must not crash the whole run
1172
+ await Promise.all(batch.map(async (task) => {
1173
+ try {
1174
+ // M11: emit task-start event.
1175
+ emit(sink, { kind: 'task-start', taskId: task.id, text: task.goal });
1176
+ // M15: Choose route for this task (local-first; cloud only when
1177
+ // allowCloud + escalation reason + key present). Best-effort — falls
1178
+ // back to the run-level client when router is unavailable.
1179
+ const { client: taskClient, decision: taskDecision } = await routeTask(task.goal, cfg, { allowCloud, attempt: 1, lastReason: 'none' }, client);
1180
+ emit(sink, {
1181
+ kind: 'log',
1182
+ taskId: task.id,
1183
+ text: `route: ${taskDecision.provider}/${taskDecision.model} [${taskDecision.tier}] — ${taskDecision.reason}`,
1184
+ });
1185
+ // Build per-task onStep callback (single-writer invariant preserved).
1186
+ // M15: cost attribution uses the provider that actually served EACH step.
1187
+ // We ACCUMULATE cost incrementally (+= this step's tokens priced at this
1188
+ // step's provider) rather than recomputing estCostUsd over the cumulative
1189
+ // run-wide totals at the current provider. Recomputing-from-cumulative is
1190
+ // wrong for mixed local+cloud runs: it would re-price an earlier local
1191
+ // task's tokens at a later cloud escalation's rates (over-charging), or
1192
+ // re-price an earlier cloud task's tokens at $0 when a later step is local
1193
+ // (erasing real spend). Incremental accumulation keeps local steps at $0
1194
+ // regardless of any later cloud escalation, and prices cloud escalations
1195
+ // on only the tokens they served.
1196
+ const makeTaskOnStep = (providerForCost) => (step) => {
1197
+ state.steps.push(step);
1198
+ // SINGLE-WRITER INVARIANT: orchestrator is the only mutator of state.usage.
1199
+ if (step.usage) {
1200
+ state.usage.tokensIn += step.usage.tokensIn;
1201
+ state.usage.tokensOut += step.usage.tokensOut;
1202
+ state.usage.steps += step.usage.steps;
1203
+ state.usage.estCostUsd += estCostUsd(providerForCost, step.usage.tokensIn, step.usage.tokensOut);
1204
+ }
1205
+ state.updatedAt = new Date().toISOString();
1206
+ cliOnStep?.(step, state.tasks);
1207
+ saveRun(state);
1208
+ };
1209
+ let taskOnStep = makeTaskOnStep(taskDecision.provider);
1210
+ // M11: Retry policy — bounded, budget-aware.
1211
+ // We retry on transient/tool failures only; hard budget stops are not retryable.
1212
+ const RETRY_POLICY = { maxAttempts: 2, baseDelayMs: 500 };
1213
+ const isRetryable = (err) => {
1214
+ // Don't retry if budget is already exhausted.
1215
+ if (overBudget(state.usage, budget))
1216
+ return false;
1217
+ // Retry on network/transient errors (not on deterministic task failures).
1218
+ if (err instanceof Error) {
1219
+ const msg = err.message.toLowerCase();
1220
+ return (msg.includes('network') ||
1221
+ msg.includes('timeout') ||
1222
+ msg.includes('econnrefused') ||
1223
+ msg.includes('fetch') ||
1224
+ msg.includes('socket'));
1225
+ }
1080
1226
  return false;
1081
- // Retry on network/transient errors (not on deterministic task failures).
1082
- if (err instanceof Error) {
1083
- const msg = err.message.toLowerCase();
1084
- return (msg.includes('network') ||
1085
- msg.includes('timeout') ||
1086
- msg.includes('econnrefused') ||
1087
- msg.includes('fetch') ||
1088
- msg.includes('socket'));
1089
- }
1090
- return false;
1091
- };
1092
- await withRetry(async (attempt) => {
1093
- if (attempt > 1) {
1094
- emit(sink, {
1095
- kind: 'retry',
1096
- taskId: task.id,
1097
- text: `attempt ${attempt} of ${RETRY_POLICY.maxAttempts}`,
1098
- });
1099
- // Reset task state for re-run on retry.
1100
- task.status = 'running';
1101
- task.result = undefined;
1102
- task.error = undefined;
1103
- }
1104
- // M20: bounded self-heal for OOM/rate-limit on model calls.
1105
- // Opt-out: ASHLR_NO_HEAL skips the wrapper entirely.
1106
- const noHeal = process.env['ASHLR_NO_HEAL'] === '1';
1107
- const runWithHeal = async (healAttempt) => {
1108
- // On heal attempt > 1 with a 'model-downgrade' event the client
1109
- // was already logged via onHeal; chooseRoute will pick a smaller
1110
- // model on the next routeTask call if the outer attempt increments,
1111
- // so we just re-run with the current client here (the heal retry
1112
- // is bounded by policy.maxRestarts and stays fully local).
1113
- if (healAttempt > 1) {
1114
- // Re-route to a smaller local model for the downgrade attempt.
1115
- // Best-effort: fall back to existing taskClient on any error.
1116
- try {
1117
- const { client: smallerClient } = await routeTask(task.goal, cfg, { allowCloud: false, attempt: healAttempt, lastReason: 'none' }, taskClient);
1118
- task.status = 'running';
1119
- task.result = undefined;
1120
- task.error = undefined;
1121
- await runTask(task, smallerClient, {
1122
- tools,
1123
- budget,
1124
- usage: state.usage,
1125
- sink,
1126
- onStep: makeTaskOnStep(smallerClient.id),
1127
- });
1128
- return;
1227
+ };
1228
+ await withRetry(async (attempt) => {
1229
+ if (attempt > 1) {
1230
+ emit(sink, {
1231
+ kind: 'retry',
1232
+ taskId: task.id,
1233
+ text: `attempt ${attempt} of ${RETRY_POLICY.maxAttempts}`,
1234
+ });
1235
+ // Reset task state for re-run on retry.
1236
+ task.status = 'running';
1237
+ task.result = undefined;
1238
+ task.error = undefined;
1239
+ }
1240
+ // M20: bounded self-heal for OOM/rate-limit on model calls.
1241
+ // Opt-out: ASHLR_NO_HEAL skips the wrapper entirely.
1242
+ const noHeal = process.env['ASHLR_NO_HEAL'] === '1';
1243
+ const runWithHeal = async (healAttempt) => {
1244
+ // On heal attempt > 1 with a 'model-downgrade' event the client
1245
+ // was already logged via onHeal; chooseRoute will pick a smaller
1246
+ // model on the next routeTask call if the outer attempt increments,
1247
+ // so we just re-run with the current client here (the heal retry
1248
+ // is bounded by policy.maxRestarts and stays fully local).
1249
+ if (healAttempt > 1) {
1250
+ // Re-route to a smaller local model for the downgrade attempt.
1251
+ // Best-effort: fall back to existing taskClient on any error.
1252
+ try {
1253
+ const { client: smallerClient } = await routeTask(task.goal, cfg, { allowCloud: false, attempt: healAttempt, lastReason: 'none' }, taskClient);
1254
+ task.status = 'running';
1255
+ task.result = undefined;
1256
+ task.error = undefined;
1257
+ await runTask(task, smallerClient, {
1258
+ tools,
1259
+ budget,
1260
+ usage: state.usage,
1261
+ sink,
1262
+ adaptivePrompts,
1263
+ onStep: makeTaskOnStep(smallerClient.id),
1264
+ });
1265
+ return;
1266
+ }
1267
+ catch {
1268
+ // Fall through to original client below.
1269
+ }
1129
1270
  }
1130
- catch {
1131
- // Fall through to original client below.
1271
+ await runTask(task, taskClient, {
1272
+ tools,
1273
+ budget,
1274
+ usage: state.usage,
1275
+ sink,
1276
+ adaptivePrompts,
1277
+ onStep: taskOnStep,
1278
+ });
1279
+ };
1280
+ if (noHeal) {
1281
+ await runTask(task, taskClient, {
1282
+ tools,
1283
+ budget,
1284
+ usage: state.usage,
1285
+ sink,
1286
+ adaptivePrompts,
1287
+ onStep: taskOnStep,
1288
+ });
1289
+ }
1290
+ else {
1291
+ const healPolicy = defaultHealPolicy();
1292
+ await withHeal(runWithHeal, healPolicy, (event) => {
1293
+ emit(sink, {
1294
+ kind: 'log',
1295
+ taskId: task.id,
1296
+ text: `[self-heal] ${event.kind} attempt ${event.attempt}: ${event.detail}`,
1297
+ });
1298
+ process.stderr.write(`[ashlr run] self-heal(${event.kind}) task ${task.id} attempt ${event.attempt}: ${event.detail}\n`);
1299
+ }, allowCloud).catch((healErr) => {
1300
+ // withHeal exhausted — re-throw so the outer withRetry sees it.
1301
+ throw healErr;
1302
+ });
1303
+ }
1304
+ // If runTask set status to failed, surface as a throw so withRetry
1305
+ // can decide whether to retry (only on retryable errors).
1306
+ if (task.status === 'failed') {
1307
+ const errMsg = task.error ?? 'task failed';
1308
+ // Only transient errors get retried; model/parsing errors do not.
1309
+ // We check if the error looks retryable before throwing.
1310
+ if (isRetryable(new Error(errMsg))) {
1311
+ throw new Error(errMsg);
1132
1312
  }
1313
+ // Non-retryable failure: don't throw (withRetry would still catch
1314
+ // and re-throw since isRetryable returns false). Fall through.
1133
1315
  }
1134
- await runTask(task, taskClient, {
1135
- tools,
1136
- budget,
1137
- usage: state.usage,
1138
- sink,
1139
- onStep: taskOnStep,
1140
- });
1141
- };
1142
- if (noHeal) {
1143
- await runTask(task, taskClient, {
1144
- tools,
1145
- budget,
1146
- usage: state.usage,
1147
- sink,
1148
- onStep: taskOnStep,
1149
- });
1150
- }
1151
- else {
1152
- const healPolicy = defaultHealPolicy();
1153
- await withHeal(runWithHeal, healPolicy, (event) => {
1316
+ }, RETRY_POLICY, isRetryable).catch((err) => {
1317
+ // withRetry exhausted all attempts or got a non-retryable error.
1318
+ // task.status is already 'failed' (set by runTask); just ensure error is set.
1319
+ if (task.status !== 'failed') {
1320
+ task.status = 'failed';
1321
+ task.error = err instanceof Error ? err.message : String(err);
1322
+ }
1323
+ });
1324
+ // M15: On task failure, attempt ONE escalated routed retry.
1325
+ // Escalation is gated by: allowCloud AND escalate.onFailure AND !overBudget.
1326
+ // chooseRoute enforces the additional cloud-key check; if it returns a
1327
+ // local route again (key absent, allowCloud false, etc.) we just stay local.
1328
+ if (task.status === 'failed' &&
1329
+ allowCloud &&
1330
+ (cfg.models.escalate?.onFailure ?? false) &&
1331
+ !overBudget(state.usage, budget)) {
1332
+ const { client: escalatedClient, decision: escalatedDecision } = await routeTask(task.goal, cfg, { allowCloud, attempt: 2, lastReason: 'task-failed' }, client);
1333
+ // Only actually escalate if chooseRoute returned a DIFFERENT (cloud)
1334
+ // route AND buildRoutedClient was able to construct a client for that
1335
+ // cloud provider. If the cloud client could not be built (key absent,
1336
+ // cloud completions unimplemented), buildRoutedClient falls back to a
1337
+ // LOCAL client whose .id is the local provider — in that case we must
1338
+ // NOT print "escalating to cloud" or charge cloud rates. Cost is
1339
+ // attributed by the ACTUAL client.id, never the intended provider.
1340
+ const cloudEscalated = escalatedDecision.tier === 'cloud' &&
1341
+ escalatedClient.id === escalatedDecision.provider;
1342
+ if (cloudEscalated) {
1154
1343
  emit(sink, {
1155
- kind: 'log',
1344
+ kind: 'retry',
1156
1345
  taskId: task.id,
1157
- text: `[self-heal] ${event.kind} attempt ${event.attempt}: ${event.detail}`,
1346
+ text: `escalating to cloud: ${escalatedDecision.provider}/${escalatedDecision.model} — ${escalatedDecision.reason}`,
1347
+ });
1348
+ task.status = 'running';
1349
+ task.result = undefined;
1350
+ task.error = undefined;
1351
+ // Attribute cost to the ACTUAL serving client (cloud here).
1352
+ taskOnStep = makeTaskOnStep(escalatedClient.id);
1353
+ await runTask(task, escalatedClient, {
1354
+ tools,
1355
+ budget,
1356
+ usage: state.usage,
1357
+ sink,
1358
+ adaptivePrompts,
1359
+ onStep: taskOnStep,
1360
+ }).catch((err) => {
1361
+ if (task.status !== 'failed') {
1362
+ task.status = 'failed';
1363
+ task.error = err instanceof Error ? err.message : String(err);
1364
+ }
1158
1365
  });
1159
- process.stderr.write(`[ashlr run] self-heal(${event.kind}) task ${task.id} attempt ${event.attempt}: ${event.detail}\n`);
1160
- }, allowCloud).catch((healErr) => {
1161
- // withHeal exhausted — re-throw so the outer withRetry sees it.
1162
- throw healErr;
1163
- });
1164
- }
1165
- // If runTask set status to failed, surface as a throw so withRetry
1166
- // can decide whether to retry (only on retryable errors).
1167
- if (task.status === 'failed') {
1168
- const errMsg = task.error ?? 'task failed';
1169
- // Only transient errors get retried; model/parsing errors do not.
1170
- // We check if the error looks retryable before throwing.
1171
- if (isRetryable(new Error(errMsg))) {
1172
- throw new Error(errMsg);
1173
1366
  }
1174
- // Non-retryable failure: don't throw (withRetry would still catch
1175
- // and re-throw since isRetryable returns false). Fall through.
1367
+ // If escalation could not reach cloud (still local / cloud client
1368
+ // unbuildable), leave task.status as 'failed' — no further action,
1369
+ // no misleading cloud event, no cloud cost.
1176
1370
  }
1177
- }, RETRY_POLICY, isRetryable).catch((err) => {
1178
- // withRetry exhausted all attempts or got a non-retryable error.
1179
- // task.status is already 'failed' (set by runTask); just ensure error is set.
1180
- if (task.status !== 'failed') {
1181
- task.status = 'failed';
1182
- task.error = err instanceof Error ? err.message : String(err);
1183
- }
1184
- });
1185
- // M15: On task failure, attempt ONE escalated routed retry.
1186
- // Escalation is gated by: allowCloud AND escalate.onFailure AND !overBudget.
1187
- // chooseRoute enforces the additional cloud-key check; if it returns a
1188
- // local route again (key absent, allowCloud false, etc.) we just stay local.
1189
- if (task.status === 'failed' &&
1190
- allowCloud &&
1191
- (cfg.models.escalate?.onFailure ?? false) &&
1192
- !overBudget(state.usage, budget)) {
1193
- const { client: escalatedClient, decision: escalatedDecision } = await routeTask(task.goal, cfg, { allowCloud, attempt: 2, lastReason: 'task-failed' }, client);
1194
- // Only actually escalate if chooseRoute returned a DIFFERENT (cloud)
1195
- // route AND buildRoutedClient was able to construct a client for that
1196
- // cloud provider. If the cloud client could not be built (key absent,
1197
- // cloud completions unimplemented), buildRoutedClient falls back to a
1198
- // LOCAL client whose .id is the local provider — in that case we must
1199
- // NOT print "escalating to cloud" or charge cloud rates. Cost is
1200
- // attributed by the ACTUAL client.id, never the intended provider.
1201
- const cloudEscalated = escalatedDecision.tier === 'cloud' &&
1202
- escalatedClient.id === escalatedDecision.provider;
1203
- if (cloudEscalated) {
1204
- emit(sink, {
1205
- kind: 'retry',
1206
- taskId: task.id,
1207
- text: `escalating to cloud: ${escalatedDecision.provider}/${escalatedDecision.model} — ${escalatedDecision.reason}`,
1208
- });
1209
- task.status = 'running';
1210
- task.result = undefined;
1211
- task.error = undefined;
1212
- // Attribute cost to the ACTUAL serving client (cloud here).
1213
- taskOnStep = makeTaskOnStep(escalatedClient.id);
1214
- await runTask(task, escalatedClient, {
1215
- tools,
1216
- budget,
1217
- usage: state.usage,
1218
- sink,
1219
- onStep: taskOnStep,
1220
- }).catch((err) => {
1221
- if (task.status !== 'failed') {
1222
- task.status = 'failed';
1223
- task.error = err instanceof Error ? err.message : String(err);
1224
- }
1225
- });
1226
- }
1227
- // If escalation could not reach cloud (still local / cloud client
1228
- // unbuildable), leave task.status as 'failed' — no further action,
1229
- // no misleading cloud event, no cloud cost.
1230
- }
1231
- // M11: Verify completed tasks; one retry on !ok if budget allows.
1232
- // Skip verify entirely once the run is over budget: a budget abort can
1233
- // leave a task 'done' with a result annotated by an abort/needs-attention
1234
- // marker, which the heuristic's error-sentinel check would flag as a
1235
- // benign false-positive "verify fail". Skipping keeps the abort path
1236
- // clean (no confusing verify line) and avoids any model call past the
1237
- // ceiling. (Real verification still runs on every in-budget completion.)
1238
- if (task.status === 'done' && !overBudget(state.usage, budget)) {
1239
- const verdict = await verifyTask(task, taskClient, budget, state.usage, {
1240
- model: verifyModel,
1241
- });
1242
- emit(sink, {
1243
- kind: 'verify',
1244
- taskId: task.id,
1245
- text: verdict.reason,
1246
- data: verdict,
1247
- });
1248
- if (!verdict.ok) {
1249
- if (!overBudget(state.usage, budget)) {
1250
- // M15: verify-failed escalation path — attempt ONE routed retry.
1251
- // If allowCloud + escalate.onFailure + key present, chooseRoute
1252
- // may return a cloud route; otherwise stays local.
1253
- const { client: verifyRetryClient, decision: verifyRetryDecision } = await routeTask(task.goal, cfg, { allowCloud, attempt: 2, lastReason: 'verify-failed' }, taskClient);
1254
- // Only treat this as a cloud escalation if the cloud client was
1255
- // actually built (decision is cloud AND the returned client's id
1256
- // matches the routed cloud provider). Otherwise buildRoutedClient
1257
- // fell back to local — keep the event + cost attribution local.
1258
- const escalatingToCloud = verifyRetryDecision.tier === 'cloud' &&
1259
- verifyRetryClient.id === verifyRetryDecision.provider;
1260
- // One verification-driven retry: re-run the task.
1371
+ // M11/M43: structured verify + bounded verify→repair loop. Skip once
1372
+ // over budget (a budget abort can annotate a 'done' result the
1373
+ // heuristic would flag as a false-positive fail).
1374
+ if (task.status === 'done' && !overBudget(state.usage, budget)) {
1375
+ const maxRepairs = Math.max(0, opts.maxRepairs ?? (engCtx?.allowExec ? 2 : 1)); // flag-off parity: plain run = 1 retry; engineer runs get up to 2 bounded repairs
1376
+ // Single source for the verify options (avoids drift between the
1377
+ // initial verify and the in-loop re-verify).
1378
+ const verifyOpts = {
1379
+ model: verifyModel,
1380
+ workspaceRoot: engCtx?.workspaceRoot,
1381
+ allowExec: engCtx?.allowExec ?? false,
1382
+ cfg,
1383
+ };
1384
+ let verifyClient = taskClient;
1385
+ let verdict = await verifyTaskStructured(task, verifyClient, budget, state.usage, verifyOpts);
1386
+ emit(sink, { kind: 'verify', taskId: task.id, text: verdict.reason, data: verdict });
1387
+ let repair = 0;
1388
+ while (!verdict.ok && repair < maxRepairs && !overBudget(state.usage, budget)) {
1389
+ repair++;
1390
+ // M15: verify-failed escalation — may return a cloud route when
1391
+ // allowCloud + escalate.onFailure + key present; otherwise local.
1392
+ const { client: retryClient, decision: retryDecision } = await routeTask(task.goal, cfg, { allowCloud, attempt: repair + 1, lastReason: 'verify-failed' }, taskClient);
1393
+ const escalatingToCloud = retryDecision.tier === 'cloud' && retryClient.id === retryDecision.provider;
1394
+ verifyClient = retryClient;
1261
1395
  emit(sink, {
1262
1396
  kind: 'retry',
1263
1397
  taskId: task.id,
1264
1398
  text: escalatingToCloud
1265
- ? `verify failed (${verdict.reason}) — escalating to cloud retry: ${verifyRetryDecision.provider}`
1266
- : `verify failed (${verdict.reason}) — retrying once`,
1399
+ ? `verify failed (${verdict.reason}) — cloud repair ${repair}/${maxRepairs}: ${retryDecision.provider}`
1400
+ : `verify failed (${verdict.reason}) — repair ${repair}/${maxRepairs}`,
1267
1401
  });
1402
+ // Feed the concrete failure back to the same task (shared id keeps
1403
+ // persistence/onStep stable; the canonical goal stays in state).
1404
+ const repairGoal = `${task.goal}\n\n[VERIFY FAILED — repair ${repair}/${maxRepairs}]\n` +
1405
+ (verdict.command ? `Command: ${verdict.command}\n` : '') +
1406
+ (verdict.failure ? `Output:\n${verdict.failure}` : verdict.reason);
1407
+ const repairTask = { ...task, goal: repairGoal };
1268
1408
  task.status = 'running';
1269
1409
  task.result = undefined;
1270
1410
  task.error = undefined;
1271
- // Attribute cost to the ACTUAL serving client (never the intended
1272
- // provider) so a local fallback stays $0.
1273
- const verifyRetryOnStep = makeTaskOnStep(verifyRetryClient.id);
1274
- await runTask(task, verifyRetryClient, {
1411
+ await runTask(repairTask, retryClient, {
1275
1412
  tools,
1276
1413
  budget,
1277
1414
  usage: state.usage,
1278
1415
  sink,
1279
- onStep: verifyRetryOnStep,
1416
+ adaptivePrompts,
1417
+ onStep: makeTaskOnStep(retryClient.id),
1280
1418
  });
1281
- // Re-verify after the retry (best-effort; don't loop).
1282
- // Cast through string: TS narrowed to 'running' after the assignment above,
1283
- // but runTask mutates task.status in place so it may be 'done' now.
1284
- if (task.status === 'done') {
1285
- const verdict2 = await verifyTask(task, verifyRetryClient, budget, state.usage, {
1286
- model: verifyModel,
1287
- });
1288
- emit(sink, {
1289
- kind: 'verify',
1290
- taskId: task.id,
1291
- text: verdict2.reason,
1292
- data: verdict2,
1293
- });
1294
- if (!verdict2.ok) {
1295
- // Still failing: annotate result but keep status 'done'.
1296
- task.result = `[needs-attention: ${verdict2.reason}]\n${task.result ?? ''}`;
1297
- }
1298
- }
1419
+ // Copy the execution outcome back onto the canonical task.
1420
+ task.status = repairTask.status;
1421
+ task.result = repairTask.result;
1422
+ task.error = repairTask.error;
1423
+ task.usage = repairTask.usage;
1424
+ if (task.status !== 'done')
1425
+ break;
1426
+ verdict = await verifyTaskStructured(task, verifyClient, budget, state.usage, verifyOpts);
1427
+ emit(sink, { kind: 'verify', taskId: task.id, text: verdict.reason, data: verdict });
1299
1428
  }
1300
- else {
1301
- // Budget exhausted: annotate but keep status 'done'.
1429
+ if (!verdict.ok && task.status === 'done') {
1430
+ // Still failing (or budget exhausted): annotate but keep 'done'.
1302
1431
  task.result = `[needs-attention: ${verdict.reason}]\n${task.result ?? ''}`;
1303
1432
  }
1304
1433
  }
1434
+ // M15: latency-threshold escalation (cfg.models.escalate?.latencyMs).
1435
+ // Latency is tracked by checking whether the task took longer than
1436
+ // the configured threshold. We use task.usage.steps as a proxy:
1437
+ // if the task completed but the run-level elapsed since task-start
1438
+ // is not directly available here, we record the threshold check as
1439
+ // informational only — the latency escalation path is a stub that
1440
+ // emits a log event when cfg.models.escalate.latencyMs is set and
1441
+ // the task usage steps are unusually high (>= TASK_STEP_CAP / 2).
1442
+ // Full wall-clock latency tracking can be wired in a follow-up.
1443
+ if (task.status === 'done' &&
1444
+ allowCloud &&
1445
+ cfg.models.escalate?.latencyMs !== undefined &&
1446
+ (task.usage?.steps ?? 0) >= 10 // heuristic: many steps → slow task
1447
+ ) {
1448
+ emit(sink, {
1449
+ kind: 'log',
1450
+ taskId: task.id,
1451
+ text: `[M15] task completed with ${task.usage?.steps ?? 0} steps; latency threshold ${cfg.models.escalate.latencyMs}ms configured (cloud escalation on latency available when re-running with --allow-cloud)`,
1452
+ });
1453
+ }
1454
+ // M11: emit task-done (or failed) event.
1455
+ if (task.status === 'done') {
1456
+ emit(sink, { kind: 'task-done', taskId: task.id, text: task.goal });
1457
+ }
1458
+ else {
1459
+ emit(sink, {
1460
+ kind: 'log',
1461
+ taskId: task.id,
1462
+ text: `task ${task.id} ${task.status}: ${task.error ?? ''}`,
1463
+ });
1464
+ }
1305
1465
  }
1306
- // M15: latency-threshold escalation (cfg.models.escalate?.latencyMs).
1307
- // Latency is tracked by checking whether the task took longer than
1308
- // the configured threshold. We use task.usage.steps as a proxy:
1309
- // if the task completed but the run-level elapsed since task-start
1310
- // is not directly available here, we record the threshold check as
1311
- // informational only — the latency escalation path is a stub that
1312
- // emits a log event when cfg.models.escalate.latencyMs is set and
1313
- // the task usage steps are unusually high (>= TASK_STEP_CAP / 2).
1314
- // Full wall-clock latency tracking can be wired in a follow-up.
1315
- if (task.status === 'done' &&
1316
- allowCloud &&
1317
- cfg.models.escalate?.latencyMs !== undefined &&
1318
- (task.usage?.steps ?? 0) >= 10 // heuristic: many steps → slow task
1319
- ) {
1320
- emit(sink, {
1321
- kind: 'log',
1322
- taskId: task.id,
1323
- text: `[M15] task completed with ${task.usage?.steps ?? 0} steps; latency threshold ${cfg.models.escalate.latencyMs}ms configured (cloud escalation on latency available when re-running with --allow-cloud)`,
1324
- });
1325
- }
1326
- // M11: emit task-done (or failed) event.
1327
- if (task.status === 'done') {
1328
- emit(sink, { kind: 'task-done', taskId: task.id, text: task.goal });
1466
+ catch (err) {
1467
+ // Defensive: runTask should handle its own errors, but catch any leak
1468
+ const msg = err instanceof Error ? err.message : String(err);
1469
+ task.status = 'failed';
1470
+ task.error = `Unexpected orchestrator error: ${msg}`;
1471
+ process.stderr.write(`[ashlr run] task ${task.id} crashed unexpectedly: ${msg}\n`);
1329
1472
  }
1330
- else {
1331
- emit(sink, {
1332
- kind: 'log',
1333
- taskId: task.id,
1334
- text: `task ${task.id} ${task.status}: ${task.error ?? ''}`,
1335
- });
1473
+ }));
1474
+ state.updatedAt = new Date().toISOString();
1475
+ saveRun(state);
1476
+ // Check budget after batch completes
1477
+ if (overBudget(state.usage, budget)) {
1478
+ aborted = true;
1479
+ break;
1480
+ }
1481
+ }
1482
+ // -- Abort: mark remaining pending/running tasks as aborted ------------------
1483
+ if (aborted) {
1484
+ for (const task of state.tasks) {
1485
+ if (task.status === 'pending' || task.status === 'running') {
1486
+ task.status = 'failed';
1487
+ task.error = ABORT_TASK_ERROR;
1336
1488
  }
1337
1489
  }
1338
- catch (err) {
1339
- // Defensive: runTask should handle its own errors, but catch any leak
1340
- const msg = err instanceof Error ? err.message : String(err);
1341
- task.status = 'failed';
1342
- task.error = `Unexpected orchestrator error: ${msg}`;
1343
- process.stderr.write(`[ashlr run] task ${task.id} crashed unexpectedly: ${msg}\n`);
1490
+ state.status = 'aborted';
1491
+ state.updatedAt = new Date().toISOString();
1492
+ saveRun(state);
1493
+ // M19: Emit telemetry (best-effort, opt-in). Awaited so the local sink is
1494
+ // flushed before the process exits; bounded + fully caught, never throws.
1495
+ await fireEmitRun(state, cfg);
1496
+ // M16: Auto-capture on abort path (fire-and-forget).
1497
+ const noCaptureAbort = opts.noCapture === true;
1498
+ if (!noCaptureAbort) {
1499
+ void (async () => {
1500
+ try {
1501
+ // eslint-disable-next-line @typescript-eslint/no-explicit-any
1502
+ const capMod = await import('../genome/capture.js');
1503
+ if (typeof capMod.captureFromRun === 'function') {
1504
+ capMod.captureFromRun(state, cfg);
1505
+ }
1506
+ }
1507
+ catch {
1508
+ // Never surface capture errors to the caller.
1509
+ }
1510
+ })();
1344
1511
  }
1345
- }));
1346
- state.updatedAt = new Date().toISOString();
1512
+ return state;
1513
+ }
1514
+ // -- Synthesize final answer -------------------------------------------------
1515
+ const synthStep = {
1516
+ ts: new Date().toISOString(),
1517
+ taskId: '__synthesize__',
1518
+ kind: 'synthesize',
1519
+ summary: 'Synthesizing final answer from task results',
1520
+ };
1521
+ state.steps.push(synthStep);
1522
+ state.updatedAt = synthStep.ts;
1347
1523
  saveRun(state);
1348
- // Check budget after batch completes
1524
+ // Budget guard for synthesis: if the run already hit the ceiling, do NOT
1525
+ // spend another model call. Fall back to concatenating the completed task
1526
+ // results so maxTokens stays a hard ceiling at the synthesis boundary too.
1527
+ let synthResult;
1528
+ let synthUsage;
1349
1529
  if (overBudget(state.usage, budget)) {
1350
- aborted = true;
1351
- break;
1530
+ const doneTasks = state.tasks.filter((t) => t.status === 'done' && t.result);
1531
+ synthResult =
1532
+ doneTasks.length > 0
1533
+ ? doneTasks.map((t) => `[${t.id}] ${t.result ?? ''}`).join('\n')
1534
+ : 'No tasks completed successfully — no result to synthesize.';
1535
+ synthUsage = { tokensIn: 0, tokensOut: 0 };
1536
+ process.stderr.write(`[ashlr run] budget reached — skipping model synthesis, using concatenated task results\n`);
1352
1537
  }
1353
- }
1354
- // -- Abort: mark remaining pending/running tasks as aborted ------------------
1355
- if (aborted) {
1356
- for (const task of state.tasks) {
1357
- if (task.status === 'pending' || task.status === 'running') {
1358
- task.status = 'failed';
1359
- task.error = ABORT_TASK_ERROR;
1360
- }
1538
+ else {
1539
+ const synth = await synthesize(goal, state.tasks, client);
1540
+ synthResult = synth.content;
1541
+ synthUsage = synth.usage;
1361
1542
  }
1362
- state.status = 'aborted';
1543
+ state.usage.tokensIn += synthUsage.tokensIn;
1544
+ state.usage.tokensOut += synthUsage.tokensOut;
1545
+ state.usage.steps += 1;
1546
+ // Accumulate incrementally (price ONLY the synthesis tokens at the synthesis
1547
+ // provider). Recomputing from cumulative totals at client.id here would CLOBBER
1548
+ // the per-step mixed-provider cost already accumulated by the task loop —
1549
+ // re-pricing earlier cloud-escalation tokens at the local run-level provider
1550
+ // (erasing real spend) or vice-versa.
1551
+ state.usage.estCostUsd += estCostUsd(client.id, synthUsage.tokensIn, synthUsage.tokensOut);
1552
+ const synthDoneStep = {
1553
+ ts: new Date().toISOString(),
1554
+ taskId: '__synthesize__',
1555
+ kind: 'synthesize',
1556
+ summary: 'Synthesis complete',
1557
+ usage: { tokensIn: synthUsage.tokensIn, tokensOut: synthUsage.tokensOut, steps: 1, estCostUsd: 0 },
1558
+ };
1559
+ state.steps.push(synthDoneStep);
1560
+ cliOnStep?.(synthDoneStep, state.tasks);
1561
+ state.result = synthResult;
1562
+ // Determine final status
1563
+ const failedCount = state.tasks.filter((t) => t.status === 'failed').length;
1564
+ state.status = failedCount === state.tasks.length ? 'failed' : 'done';
1363
1565
  state.updatedAt = new Date().toISOString();
1364
1566
  saveRun(state);
1365
- // M19: Emit telemetry (best-effort, opt-in). Awaited so the local sink is
1366
- // flushed before the process exits; bounded + fully caught, never throws.
1567
+ // -- M19: Emit telemetry (best-effort, opt-in) ------------------------------
1568
+ // Awaited so the local sink is flushed before the process exits; bounded +
1569
+ // fully caught, never throws.
1367
1570
  await fireEmitRun(state, cfg);
1368
- // M16: Auto-capture on abort path (fire-and-forget).
1369
- const noCaptureAbort = opts.noCapture === true;
1370
- if (!noCaptureAbort) {
1571
+ // -- M16: Auto-capture (fire-and-forget, never throws, never blocks) ---------
1572
+ // Read noCapture via extended property (same pattern as noMemory above).
1573
+ const noCapture = opts.noCapture === true;
1574
+ if (!noCapture) {
1575
+ // Wrap in void + try to guarantee fire-and-forget with zero blocking.
1371
1576
  void (async () => {
1372
1577
  try {
1373
1578
  // eslint-disable-next-line @typescript-eslint/no-explicit-any
@@ -1383,82 +1588,9 @@ export async function runGoal(goal, cfg, opts) {
1383
1588
  }
1384
1589
  return state;
1385
1590
  }
1386
- // -- Synthesize final answer -------------------------------------------------
1387
- const synthStep = {
1388
- ts: new Date().toISOString(),
1389
- taskId: '__synthesize__',
1390
- kind: 'synthesize',
1391
- summary: 'Synthesizing final answer from task results',
1392
- };
1393
- state.steps.push(synthStep);
1394
- state.updatedAt = synthStep.ts;
1395
- saveRun(state);
1396
- // Budget guard for synthesis: if the run already hit the ceiling, do NOT
1397
- // spend another model call. Fall back to concatenating the completed task
1398
- // results so maxTokens stays a hard ceiling at the synthesis boundary too.
1399
- let synthResult;
1400
- let synthUsage;
1401
- if (overBudget(state.usage, budget)) {
1402
- const doneTasks = state.tasks.filter((t) => t.status === 'done' && t.result);
1403
- synthResult =
1404
- doneTasks.length > 0
1405
- ? doneTasks.map((t) => `[${t.id}] ${t.result ?? ''}`).join('\n')
1406
- : 'No tasks completed successfully — no result to synthesize.';
1407
- synthUsage = { tokensIn: 0, tokensOut: 0 };
1408
- process.stderr.write(`[ashlr run] budget reached — skipping model synthesis, using concatenated task results\n`);
1409
- }
1410
- else {
1411
- const synth = await synthesize(goal, state.tasks, client);
1412
- synthResult = synth.content;
1413
- synthUsage = synth.usage;
1414
- }
1415
- state.usage.tokensIn += synthUsage.tokensIn;
1416
- state.usage.tokensOut += synthUsage.tokensOut;
1417
- state.usage.steps += 1;
1418
- // Accumulate incrementally (price ONLY the synthesis tokens at the synthesis
1419
- // provider). Recomputing from cumulative totals at client.id here would CLOBBER
1420
- // the per-step mixed-provider cost already accumulated by the task loop —
1421
- // re-pricing earlier cloud-escalation tokens at the local run-level provider
1422
- // (erasing real spend) or vice-versa.
1423
- state.usage.estCostUsd += estCostUsd(client.id, synthUsage.tokensIn, synthUsage.tokensOut);
1424
- const synthDoneStep = {
1425
- ts: new Date().toISOString(),
1426
- taskId: '__synthesize__',
1427
- kind: 'synthesize',
1428
- summary: 'Synthesis complete',
1429
- usage: { tokensIn: synthUsage.tokensIn, tokensOut: synthUsage.tokensOut, steps: 1, estCostUsd: 0 },
1430
- };
1431
- state.steps.push(synthDoneStep);
1432
- cliOnStep?.(synthDoneStep, state.tasks);
1433
- state.result = synthResult;
1434
- // Determine final status
1435
- const failedCount = state.tasks.filter((t) => t.status === 'failed').length;
1436
- state.status = failedCount === state.tasks.length ? 'failed' : 'done';
1437
- state.updatedAt = new Date().toISOString();
1438
- saveRun(state);
1439
- // -- M19: Emit telemetry (best-effort, opt-in) ------------------------------
1440
- // Awaited so the local sink is flushed before the process exits; bounded +
1441
- // fully caught, never throws.
1442
- await fireEmitRun(state, cfg);
1443
- // -- M16: Auto-capture (fire-and-forget, never throws, never blocks) ---------
1444
- // Read noCapture via extended property (same pattern as noMemory above).
1445
- const noCapture = opts.noCapture === true;
1446
- if (!noCapture) {
1447
- // Wrap in void + try to guarantee fire-and-forget with zero blocking.
1448
- void (async () => {
1449
- try {
1450
- // eslint-disable-next-line @typescript-eslint/no-explicit-any
1451
- const capMod = await import('../genome/capture.js');
1452
- if (typeof capMod.captureFromRun === 'function') {
1453
- capMod.captureFromRun(state, cfg);
1454
- }
1455
- }
1456
- catch {
1457
- // Never surface capture errors to the caller.
1458
- }
1459
- })();
1591
+ finally {
1592
+ finalizeEngineer();
1460
1593
  }
1461
- return state;
1462
1594
  }
1463
1595
  // ---------------------------------------------------------------------------
1464
1596
  // Gateway tool loading (optional)