@drisp/cli 0.5.26 → 0.5.28

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1033,6 +1033,10 @@ function hasProjectWorkflow(projectDir) {
1033
1033
  }
1034
1034
 
1035
1035
  // src/core/workflows/types.ts
1036
+ var DEFAULT_MAX_TURN_TOKEN_COUNT = 13e4;
1037
+ var DEFAULT_NUDGE_CAP = 3;
1038
+ var DEFAULT_RETRY_CAP = 3;
1039
+ var DEFAULT_RETRY_BACKOFF_MS = 1e4;
1036
1040
  function pluginSpecRef(spec) {
1037
1041
  return typeof spec === "string" ? spec : spec.ref;
1038
1042
  }
@@ -1127,15 +1131,16 @@ function refreshPinnedWorkflowPlugins(workflow) {
1127
1131
  // src/core/workflows/registry.ts
1128
1132
  import fs9 from "fs";
1129
1133
  import os5 from "os";
1130
- import path7 from "path";
1134
+ import path8 from "path";
1131
1135
 
1132
1136
  // src/core/workflows/builtins/index.ts
1133
1137
  import fs7 from "fs";
1134
1138
  import os4 from "os";
1135
- import path5 from "path";
1139
+ import path6 from "path";
1136
1140
 
1137
1141
  // src/core/workflows/trackerReader.ts
1138
1142
  import fs6 from "fs";
1143
+ import path5 from "path";
1139
1144
 
1140
1145
  // src/core/workflows/templateVars.ts
1141
1146
  function substituteVariables(text, ctx) {
@@ -1158,6 +1163,7 @@ var DEFAULT_COMPLETION_MARKER = "<!-- WORKFLOW_COMPLETE -->";
1158
1163
  var DEFAULT_BLOCKED_MARKER = "<!-- WORKFLOW_BLOCKED";
1159
1164
  var DEFAULT_TRACKER_PATH = ".athena/{sessionId}/tracker.md";
1160
1165
  var TRACKER_SKELETON_MARKER = "<!-- TRACKER_SKELETON -->";
1166
+ var DEFAULT_TRACKER_TOKEN_BOUND = 8e3;
1161
1167
  var DEFAULT_CONTINUE_PROMPT = "Continue the task. Read the tracker at {trackerPath} for current progress. If the work is complete or blocked, the terminal marker must be the final non-empty line of the tracker; do not write any prose after it.";
1162
1168
  function readTracker(trackerPath) {
1163
1169
  try {
@@ -1209,12 +1215,114 @@ function parseTrackerState(content, markers = {}) {
1209
1215
  skeletonNotReplaced: content.includes(TRACKER_SKELETON_MARKER)
1210
1216
  };
1211
1217
  }
1218
+ function demoteTerminalMarkers(content, markers = {}) {
1219
+ const completionMarker = markers.completionMarker ?? DEFAULT_COMPLETION_MARKER;
1220
+ const blockedMarker = markers.blockedMarker ?? DEFAULT_BLOCKED_MARKER;
1221
+ return content.split("\n").map((line) => {
1222
+ const trimmed = line.trim();
1223
+ if (trimmed === completionMarker) return "> _Prior Run ended: complete._";
1224
+ if (isBlockedLine(trimmed, blockedMarker)) {
1225
+ const reason = extractBlockedReason(trimmed, blockedMarker);
1226
+ return reason ? `> _Prior Run ended: needed attention \u2014 ${reason}._` : "> _Prior Run ended: needed attention._";
1227
+ }
1228
+ return line;
1229
+ }).join("\n");
1230
+ }
1212
1231
  function buildContinuePrompt(loop) {
1213
1232
  const template = loop.continuePrompt ?? DEFAULT_CONTINUE_PROMPT;
1214
1233
  return substituteVariables(template, {
1215
1234
  trackerPath: loop.trackerPath ?? DEFAULT_TRACKER_PATH
1216
1235
  });
1217
1236
  }
1237
+ function buildNudgePrompt(loop, opts) {
1238
+ const completionMarker = loop.completionMarker ?? DEFAULT_COMPLETION_MARKER;
1239
+ const blockedMarker = loop.blockedMarker ?? DEFAULT_BLOCKED_MARKER;
1240
+ const trackerPath = loop.trackerPath ?? DEFAULT_TRACKER_PATH;
1241
+ const bootstrapPreamble = opts?.skeletonNotReplaced ? `You stopped without writing the tracker at ${trackerPath} \u2014 it still contains the runner's skeleton. If you were asking the human a question, do not ask it in chat: write it to the tracker and declare ${blockedMarker}: <your question> --> as the final non-empty line. Otherwise replace the skeleton with the real plan and continue. ` : `You stopped without declaring how this workflow ended. If work remains, continue it now. `;
1242
+ return bootstrapPreamble + `If everything is done and verified, write ${completionMarker} as the final non-empty line of the tracker at ${trackerPath}. If you cannot proceed without a human, write ${blockedMarker}: <reason> --> there instead. Do not stop again without either finishing the work or declaring one of these markers.`;
1243
+ }
1244
+ function estimateTokenCount(content) {
1245
+ return Math.ceil(content.length / 4);
1246
+ }
1247
+ function buildTrackerSizeNudgeSuffix(trackerPath) {
1248
+ const where = trackerPath ?? "the tracker";
1249
+ return ` Separately: ${where} has crossed the ~8,000-token shedding backstop (ADR 0015 \xA73). Before continuing, cut the completed phases of the still-open unit out of the tracker and paste them verbatim into that unit's record under units/<slug>.md, leaving a pointer row behind \u2014 this is a nudge, not a requirement, so continue the work either way.`;
1250
+ }
1251
+ var UNITS_HEADING_RE = /^##\s+Units\s*$/;
1252
+ function splitTableRow(line) {
1253
+ const trimmed = line.trim();
1254
+ const withoutEdgePipes = trimmed.replace(/^\|/, "").replace(/\|$/, "");
1255
+ return withoutEdgePipes.split("|").map((cell) => cell.trim());
1256
+ }
1257
+ function isSeparatorRow(cells) {
1258
+ return cells.length > 0 && cells.every((cell) => /^:?-+:?$/.test(cell));
1259
+ }
1260
+ function parseUnitTable(content) {
1261
+ const lines = content.split("\n");
1262
+ const headingIndex = lines.findIndex(
1263
+ (line) => UNITS_HEADING_RE.test(line.trim())
1264
+ );
1265
+ if (headingIndex === -1) return null;
1266
+ let i = headingIndex + 1;
1267
+ while (i < lines.length && lines[i].trim() === "") i++;
1268
+ const tableLines = [];
1269
+ while (i < lines.length) {
1270
+ const trimmed = lines[i].trim();
1271
+ if (!trimmed.startsWith("|")) break;
1272
+ tableLines.push(trimmed);
1273
+ i++;
1274
+ }
1275
+ if (tableLines.length < 2) return null;
1276
+ const rows = [];
1277
+ for (const line of tableLines.slice(1)) {
1278
+ const cells = splitTableRow(line);
1279
+ if (isSeparatorRow(cells)) continue;
1280
+ if (cells.length !== 2) continue;
1281
+ const [label, recordPath] = cells;
1282
+ if (!label || !recordPath) continue;
1283
+ rows.push({ label, recordPath });
1284
+ }
1285
+ return rows;
1286
+ }
1287
+ var FRONTMATTER_RE = /^---\r?\n([\s\S]*?)\r?\n---\s*(?:\r?\n|$)/;
1288
+ function parseUnitRecordFrontmatter(content) {
1289
+ const match = FRONTMATTER_RE.exec(content);
1290
+ if (!match) return null;
1291
+ const block = match[1];
1292
+ const statusMatch = /^status:\s*(\S+)\s*$/m.exec(block);
1293
+ const raw = statusMatch?.[1]?.trim().toLowerCase();
1294
+ if (raw !== "open" && raw !== "closed") return null;
1295
+ return { status: raw };
1296
+ }
1297
+ function projectTrackerTasks(trackerAbsPath) {
1298
+ const content = readTracker(trackerAbsPath);
1299
+ if (!content) return null;
1300
+ const rows = parseUnitTable(content);
1301
+ if (!rows) return null;
1302
+ const baseDir = path5.dirname(trackerAbsPath);
1303
+ const tasks = [];
1304
+ for (const row of rows) {
1305
+ const recordAbsPath = path5.resolve(baseDir, row.recordPath);
1306
+ const relToBase = path5.relative(baseDir, recordAbsPath);
1307
+ if (relToBase.startsWith("..") || path5.isAbsolute(relToBase)) continue;
1308
+ let recordContent;
1309
+ try {
1310
+ recordContent = fs6.readFileSync(recordAbsPath, "utf-8");
1311
+ } catch {
1312
+ continue;
1313
+ }
1314
+ const frontmatter = parseUnitRecordFrontmatter(recordContent);
1315
+ if (!frontmatter) continue;
1316
+ const taskId = path5.basename(row.recordPath, path5.extname(row.recordPath));
1317
+ if (!taskId) continue;
1318
+ tasks.push({
1319
+ taskId,
1320
+ content: row.label,
1321
+ status: frontmatter.status === "closed" ? "completed" : "pending"
1322
+ });
1323
+ }
1324
+ return tasks;
1325
+ }
1218
1326
 
1219
1327
  // src/core/workflows/builtins/index.ts
1220
1328
  var DEFAULT_BLOCKED_CLOSED_MARKER = `${DEFAULT_BLOCKED_MARKER} -->`;
@@ -1274,14 +1382,14 @@ If you are blocked and cannot make further progress:
1274
1382
  4. Do not write any tracker content after the terminal marker
1275
1383
  `;
1276
1384
  function ensureSystemPromptFile() {
1277
- const dir = path5.join(
1385
+ const dir = path6.join(
1278
1386
  os4.homedir(),
1279
1387
  ".config",
1280
1388
  "athena",
1281
1389
  "builtins",
1282
1390
  "default"
1283
1391
  );
1284
- const filePath = path5.join(dir, "system_prompt.md");
1392
+ const filePath = path6.join(dir, "system_prompt.md");
1285
1393
  if (!fs7.existsSync(filePath) || fs7.readFileSync(filePath, "utf-8") !== SYSTEM_PROMPT) {
1286
1394
  fs7.mkdirSync(dir, { recursive: true });
1287
1395
  fs7.writeFileSync(filePath, SYSTEM_PROMPT, "utf-8");
@@ -1312,7 +1420,7 @@ function listBuiltinWorkflows() {
1312
1420
 
1313
1421
  // src/core/workflows/sourceMetadata.ts
1314
1422
  import fs8 from "fs";
1315
- import path6 from "path";
1423
+ import path7 from "path";
1316
1424
  function legacyLocalToNew(legacyPath, legacyRepoDir) {
1317
1425
  const repoDir = legacyRepoDir ?? findMarketplaceRepoDir(legacyPath);
1318
1426
  if (!repoDir) {
@@ -1338,7 +1446,7 @@ function legacyLocalToNew(legacyPath, legacyRepoDir) {
1338
1446
  return { kind: "filesystem", path: legacyPath };
1339
1447
  }
1340
1448
  function readWorkflowSourceMetadata(workflowDir) {
1341
- const sourceFile = path6.join(workflowDir, "source.json");
1449
+ const sourceFile = path7.join(workflowDir, "source.json");
1342
1450
  if (!fs8.existsSync(sourceFile)) return void 0;
1343
1451
  let raw;
1344
1452
  try {
@@ -1386,7 +1494,7 @@ function readWorkflowSourceMetadata(workflowDir) {
1386
1494
  function writeWorkflowSourceMetadata(workflowDir, metadata) {
1387
1495
  fs8.mkdirSync(workflowDir, { recursive: true });
1388
1496
  fs8.writeFileSync(
1389
- path6.join(workflowDir, "source.json"),
1497
+ path7.join(workflowDir, "source.json"),
1390
1498
  JSON.stringify({ v: 2, ...metadata }),
1391
1499
  "utf-8"
1392
1500
  );
@@ -1394,18 +1502,18 @@ function writeWorkflowSourceMetadata(workflowDir, metadata) {
1394
1502
 
1395
1503
  // src/core/workflows/registry.ts
1396
1504
  function registryDir() {
1397
- return path7.join(os5.homedir(), ".config", "athena", "workflows");
1505
+ return path8.join(os5.homedir(), ".config", "athena", "workflows");
1398
1506
  }
1399
1507
  function ensurePathWithinRoot(rootDir, targetPath, label) {
1400
- const relative = path7.relative(rootDir, targetPath);
1401
- if (relative === "" || !relative.startsWith("..") && !path7.isAbsolute(relative)) {
1508
+ const relative = path8.relative(rootDir, targetPath);
1509
+ if (relative === "" || !relative.startsWith("..") && !path8.isAbsolute(relative)) {
1402
1510
  return;
1403
1511
  }
1404
1512
  throw new Error(`${label} resolves outside the workflow root: ${targetPath}`);
1405
1513
  }
1406
1514
  function resolveWorkflow(name) {
1407
- const workflowDir = path7.join(registryDir(), name);
1408
- const workflowPath = path7.join(workflowDir, "workflow.json");
1515
+ const workflowDir = path8.join(registryDir(), name);
1516
+ const workflowPath = path8.join(workflowDir, "workflow.json");
1409
1517
  if (!fs9.existsSync(workflowPath)) {
1410
1518
  const builtin = resolveBuiltinWorkflow(name);
1411
1519
  if (builtin) {
@@ -1456,8 +1564,8 @@ function resolveWorkflow(name) {
1456
1564
  );
1457
1565
  }
1458
1566
  const workflowFile = raw["workflowFile"];
1459
- if (typeof workflowFile === "string" && !path7.isAbsolute(workflowFile)) {
1460
- const workflowFilePath = path7.resolve(workflowDir, workflowFile);
1567
+ if (typeof workflowFile === "string" && !path8.isAbsolute(workflowFile)) {
1568
+ const workflowFilePath = path8.resolve(workflowDir, workflowFile);
1461
1569
  if (!fs9.existsSync(workflowFilePath)) {
1462
1570
  throw new Error(
1463
1571
  `Invalid workflow.json: workflowFile "${workflowFile}" not found at ${workflowFilePath}`
@@ -1480,23 +1588,23 @@ function readWorkflowSource(sourcePath) {
1480
1588
  }
1481
1589
  function copyWorkflowFiles(sourcePath, destDir) {
1482
1590
  const { content, workflow } = readWorkflowSource(sourcePath);
1483
- const absoluteSourcePath = path7.resolve(sourcePath);
1484
- const sourceDir = path7.dirname(absoluteSourcePath);
1485
- const absoluteDestDir = path7.resolve(destDir);
1591
+ const absoluteSourcePath = path8.resolve(sourcePath);
1592
+ const sourceDir = path8.dirname(absoluteSourcePath);
1593
+ const absoluteDestDir = path8.resolve(destDir);
1486
1594
  fs9.mkdirSync(absoluteDestDir, { recursive: true });
1487
1595
  fs9.writeFileSync(
1488
- path7.join(absoluteDestDir, "workflow.json"),
1596
+ path8.join(absoluteDestDir, "workflow.json"),
1489
1597
  content,
1490
1598
  "utf-8"
1491
1599
  );
1492
1600
  const copyRelativeAsset = (assetPath) => {
1493
- if (!assetPath || path7.isAbsolute(assetPath)) return;
1494
- const sourceAssetPath = path7.resolve(sourceDir, assetPath);
1601
+ if (!assetPath || path8.isAbsolute(assetPath)) return;
1602
+ const sourceAssetPath = path8.resolve(sourceDir, assetPath);
1495
1603
  ensurePathWithinRoot(sourceDir, sourceAssetPath, "Workflow asset");
1496
1604
  if (!fs9.existsSync(sourceAssetPath)) return;
1497
- const destAssetPath = path7.resolve(absoluteDestDir, assetPath);
1605
+ const destAssetPath = path8.resolve(absoluteDestDir, assetPath);
1498
1606
  ensurePathWithinRoot(absoluteDestDir, destAssetPath, "Workflow asset");
1499
- fs9.mkdirSync(path7.dirname(destAssetPath), { recursive: true });
1607
+ fs9.mkdirSync(path8.dirname(destAssetPath), { recursive: true });
1500
1608
  fs9.copyFileSync(sourceAssetPath, destAssetPath);
1501
1609
  };
1502
1610
  const rawWorkflow = workflow;
@@ -1534,7 +1642,7 @@ function installWorkflowFromSource(source, name) {
1534
1642
  'Workflow has no "name" field. Provide --name to specify one.'
1535
1643
  );
1536
1644
  }
1537
- const destDir = path7.join(registryDir(), workflowName);
1645
+ const destDir = path8.join(registryDir(), workflowName);
1538
1646
  copyWorkflowFiles(source.workflowPath, destDir);
1539
1647
  const metadata = toStoredMetadata(source);
1540
1648
  writeWorkflowSourceMetadata(destDir, metadata);
@@ -1579,7 +1687,7 @@ function reResolveFromMetadata(metadata, refresh = refreshMarketplaceRepo, onSel
1579
1687
  return { kind: "filesystem", workflowPath: metadata.path };
1580
1688
  }
1581
1689
  function updateWorkflow(name, refresh = refreshMarketplaceRepo, onSelfHeal) {
1582
- const workflowDir = path7.join(registryDir(), name);
1690
+ const workflowDir = path8.join(registryDir(), name);
1583
1691
  const metadata = readWorkflowSourceMetadata(workflowDir);
1584
1692
  if (!metadata) {
1585
1693
  throw new Error(
@@ -1635,7 +1743,7 @@ function updateWorkflows(names) {
1635
1743
  function listWorkflows() {
1636
1744
  const dir = registryDir();
1637
1745
  const installed = fs9.existsSync(dir) ? fs9.readdirSync(dir, { withFileTypes: true }).filter(
1638
- (entry) => entry.isDirectory() && fs9.existsSync(path7.join(dir, entry.name, "workflow.json"))
1746
+ (entry) => entry.isDirectory() && fs9.existsSync(path8.join(dir, entry.name, "workflow.json"))
1639
1747
  ).map((entry) => entry.name) : [];
1640
1748
  const builtins = listBuiltinWorkflows().filter(
1641
1749
  (name) => !installed.includes(name)
@@ -1643,7 +1751,7 @@ function listWorkflows() {
1643
1751
  return [...builtins, ...installed];
1644
1752
  }
1645
1753
  function removeWorkflow(name) {
1646
- const dir = path7.join(registryDir(), name);
1754
+ const dir = path8.join(registryDir(), name);
1647
1755
  if (!fs9.existsSync(dir)) {
1648
1756
  throw new Error(`Workflow "${name}" not found.`);
1649
1757
  }
@@ -1682,10 +1790,10 @@ function compileWorkflowPlan(input) {
1682
1790
 
1683
1791
  // src/core/workflows/sessionPlan.ts
1684
1792
  import fs10 from "fs";
1685
- import path8 from "path";
1793
+ import path9 from "path";
1686
1794
 
1687
1795
  // src/core/workflows/stateMachine.md
1688
- var stateMachine_default = "# Stateless Turn Protocol\n\nYou run in a stateless loop. Each Turn is a fresh process with no memory of prior Turns. **The tracker file is your only continuity** \u2014 read it, work, write it. Assume interruption: the runner may kill a long Turn, your context may collapse under tool output, you may hit token limits mid-task. Anything not in the tracker is gone.\n\n## First action, every Turn\n\n1. Read the tracker at the configured path (default: `.athena/<session_id>/tracker.md`). The runner provides the session ID \u2014 do not invent one.\n2. If the tracker contains `<!-- TRACKER_SKELETON -->` \u2192 this is Turn 1, run [**Orient**](#orient-turn-1).\n3. Otherwise \u2192 this is a continuation, run [**Execute**](#execute-turn-2) from where the tracker says, not from the start of the flow.\n\nReading first prevents two failure modes that waste whole Turns: redoing work already done, or contradicting decisions a prior Turn made.\n\n## Tracker contract\n\nThe tracker must always answer four questions:\n\n1. What are we trying to accomplish?\n2. What has been done?\n3. What's left?\n4. What should the next Turn do first?\n\nA future Turn has no other context. If something isn't here, it doesn't exist. Section headings may vary by workflow, but these four answers must be explicit and easy to find.\n\n### Terminal markers\n\nDefault markers (workflows may override \u2014 use the markers configured for the active workflow):\n\n- `<!-- WORKFLOW_COMPLETE -->` \u2014 all work done and verified\n- `<!-- WORKFLOW_BLOCKED -->` or `<!-- WORKFLOW_BLOCKED: reason -->` \u2014 cannot proceed without external intervention\n\nRules:\n\n- Only the last non-empty line of the tracker is authoritative. Marker-like text in notes, examples, or quoted instructions earlier in the file is ignored.\n- When you write a terminal marker, it must be the final non-empty line of the tracker. Put every summary, status note, and next-step sentence before the marker. Never append prose after it.\n- The runner trusts markers unconditionally. A premature marker ends the loop with no automatic recovery \u2014 write one only when its criteria are fully met.\n- Include a concrete reason after `WORKFLOW_BLOCKED:` whenever possible; the bare form is still valid.\n\n## Phases\n\n### Orient (Turn 1)\n\n1. **Replace the skeleton immediately**, before any domain work. Even a three-line tracker (goal + \"orienting\") protects you if the Turn dies during setup.\n2. Identify and load the applicable workflow skills before doing domain work. If a workflow, plugin, or local skill table names a relevant skill, read it fully and follow it. Do not assume you already know the workflow's conventions, tool sequence, quality gates, or implementation details.\n3. Use a dedicated git worktree for repository-changing work. If you are not already inside a task-specific worktree, create or enter one before editing files, record its branch/path in the tracker, and continue there. Skip this only when the workflow explicitly forbids it or the task is read-only.\n4. Run the workflow's orientation steps exactly as written. These vary by domain \u2014 a test-writing workflow explores the product in a browser; a migration workflow audits the schema. The workflow defines what orientation means. Do not skip, reorder, reinterpret, or replace workflow steps with a generic approach unless the workflow explicitly allows it or the tracker records a concrete blocker that makes the written step impossible.\n5. Refine the tracker into a granular plan. Each task a concrete, verifiable unit of work, including verification steps (running checks, reviewing output) \u2014 not just implementation. Vague tasks (\"write tests\") cannot be meaningfully resumed by a future Turn that has no idea what they mean here.\n6. Record concrete observations \u2014 what you actually saw, not what you assumed. Wrong assumptions burn entire future Turns on rework.\n7. **Single-Turn requests still go through this phase.** If the entire request is satisfied in one Turn, write a minimal tracker (what was asked, what was done, the outcome) and append `<!-- WORKFLOW_COMPLETE -->`. Leaving the skeleton in place causes the runner to classify the Turn as a failure.\n\n### Execute (Turn 2+)\n\n- Work from where the tracker says, in the workflow's prescribed sequence. Not every Turn covers every step.\n- Be strict with workflow steps. Before starting each unit, identify the next required workflow step from the workflow document and tracker, follow it as written, and record completion or blockers against that step. Do not substitute your own process, collapse separate gates into one, or advance past an unchecked step.\n- Be strict with skills. Before each new activity, check the workflow, plugin metadata, local skill table, and tracker for relevant skills. Load the appropriate skill first, read it completely, and follow its instructions. If no skill applies, record that explicitly in the tracker before proceeding. Skills carry the implementation detail (scaffolding steps, locator rules, anti-patterns, code templates) that this protocol intentionally doesn't repeat.\n- Keep repository work inside the recorded git worktree. If a continuation Turn starts outside the recorded worktree, enter it before editing. If no worktree is recorded and edits are still required, create or enter one before proceeding.\n- Delegate heavy exploration or generation to subagents via the Task tool. Pass file paths, conventions, and concrete output expectations; tell them which skill to load. Respect the workflow's **delegation constraints** \u2014 some operations must run in the main agent because their output is proof, or because the main agent needs to interpret results in context.\n- Run quality gates in order. Do not skip \u2014 they exist because skipping cascades into rework. On a failing verdict, address the issues and re-run before proceeding. Respect the workflow's **retry limits**: repeated failure usually signals a deeper issue another retry won't fix.\n\n### End\n\n1. Tracker reflects all progress, discoveries, and blockers.\n2. Tracker says clearly what the next Turn should do first.\n3. If all work is verified: append the completion marker as the final non-empty line.\n4. If an unrecoverable blocker prevents progress: append the blocked marker as the final non-empty line, with a reason if you have one.\n\n## When to write the tracker\n\nWrite on **concrete triggers**, not on a vague sense of \"meaningful progress.\" The right cadence sits between every-tool-call (noisy log, wastes tokens) and end-of-Turn (everything lost if you die mid-task).\n\n- **Discrete unit done** \u2014 file written, fix applied, test run, gate passed. Reflect the new reality before starting the next unit.\n- **Insight learned** \u2014 API quirk, config field that turned out to matter, dead end ruled out, decision between two approaches. Insights are tracker-worthy even when no code changed; rediscovering them costs the next Turn a full re-exploration. The tracker is a knowledge ledger, not just a task log.\n- **About to do something risky or long-running** \u2014 subagent dispatch, long build, flaky external call, large refactor. Write _first_, then act. If the operation kills your Turn, only what's on disk survives.\n- **Plan changed** \u2014 task resequenced, new task surfaced, planned task no longer needed. Stale plans poison continuation Turns.\n- **You haven't written in a while** \u2014 if you can't remember the last update, you've gone too long. A short defensive update (\"doing X, last completed Y, next is Z\") beats nothing.\n\nEach update covers: what changed (work or knowledge), what's now next, and any caveat the next Turn needs. Don't transcribe tool calls \u2014 the tracker is a contract with your future self, not a replay log.\n\nThe cost of one extra tracker update is a few tokens. The cost of dying without one is a whole wasted Turn. Bias toward writing.\n\n## Task UI projection\n\nThe tracker is the durable source of truth. Your harness's task tools are a Turn-scoped UI projection of the same plan, shown to the user in their CLI widget. They do not survive process exit.\n\n{{TASK_TOOL_INSTRUCTIONS}}\n\n- **Turn 1, after orientation:** project the tracker's task plan into the task tools.\n- **Turn 2+, after reading the tracker:** recreate the projection from the tracker; do not assume task IDs from prior Turns still exist.\n- **During work:** update both \u2014 the task tools for immediate UI feedback, the tracker for persistence \u2014 in the same working phase.\n\n## Turn bounding\n\nEach fresh Turn starts with a clean context window and a compact tracker \u2014 effectively self-compaction. As you work, context fills with tool outputs and intermediate state. The longer you run, the more attention is spread across tokens that are no longer relevant, degrading precision on the work that matters now.\n\nWork a bounded chunk per Turn. Ending early and letting the next Turn pick up from a clean tracker is almost always better than pushing through with a heavy context. Natural checkpoints:\n\n- After a quality gate\n- After crossing multiple phases (explored \u2192 planned \u2192 wrote specs) \u2014 stop before pushing into the next\n- When your context is visibly heavy with tool output from earlier work\n\n## Quick reference\n\n- [ ] Read the tracker before doing anything else\n- [ ] Replace the skeleton immediately, even for single-Turn requests\n- [ ] Update on concrete triggers \u2014 unit done, insight learned, risky op pending, plan changed\n- [ ] Project the tracker plan into task tools at Turn start; keep both in sync as work lands\n- [ ] Follow the workflow steps as written; do not skip, reorder, or substitute your own process\n- [ ] Load the appropriate skill before each activity; do not rely on assumed knowledge\n- [ ] Use and record a dedicated git worktree for repository-changing work\n- [ ] Run quality gates in order; respect delegation constraints and retry limits\n- [ ] Write the completion marker only when all work is verified, and make it the final non-empty line\n- [ ] Checkpoint and end before context goes stale\n";
1796
+ var stateMachine_default = "# Turn Protocol\n\nYou run inside a managed workflow loop. **Run until the work is done.** The Dossier \u2014 `.athena/<session_id>/` \u2014 is your durable memory; `tracker.md` is your index into it: read it, work, write it as you go. Your conversation context usually survives between stops \u2014 the runner resumes your session rather than restarting it \u2014 but never rely on that: the runner may kill a long Turn, your session may be replaced at a context bound, the process may die mid-task. Anything not in the Dossier can be lost.\n\nTwo kinds of Turn exist, and you should know which you are in:\n\n- **A fresh Turn** \u2014 the first Turn of the run, or the Turn right after a Handover (the runner's context reset). You start with no memory of prior work; the tracker (and, after a Handover, the Handoff file) is all you have.\n- **A resumed Turn** \u2014 the runner continued your existing session with a new instruction (a corrective nudge, a retry after a transient failure, or a human's reply). Your context is intact; act on the new instruction and keep going.\n\n## First action, in a fresh Turn\n\n1. Read the tracker at the configured path (default: `.athena/<session_id>/tracker.md`). The runner provides the session ID \u2014 do not invent one.\n2. If the tracker contains `<!-- TRACKER_SKELETON -->` \u2192 this is Turn 1, run [**Orient**](#orient-turn-1).\n3. Otherwise \u2192 this is a continuation, run [**Execute**](#execute-continuation) from where the tracker says, not from the start of the flow.\n4. If the runner's prompt names a Handoff file, read it too \u2014 it's mandatory alongside the tracker: it carries the in-flight context the tracker never checkpointed. Before any domain work, fold whatever in it is still durable into the tracker (or the open unit's record, if it's been shed) \u2014 the Handoff is a one-time relay, not a permanent Dossier file, so anything worth keeping has to land in `tracker.md`/`units/<slug>.md` now or it is lost once the Handoff falls off the chain. Then read whatever the tracker's fifth question names (see [Tracker contract](#tracker-contract)) \u2014 that, the tracker, and any named Handoff file are your complete required reading. Do not redo completed work or re-litigate decisions any of them record.\n\nReading first prevents two failure modes that waste whole Turns: redoing work already done, or contradicting decisions a prior Turn made.\n\nIn a resumed Turn, your context is already loaded \u2014 skim the tracker only if you have any doubt it still matches reality, then continue.\n\n## Tracker contract\n\nThe tracker must always answer five questions:\n\n1. What are we trying to accomplish?\n2. What has been done?\n3. What's left?\n4. What should the next Turn do first?\n5. What else, beyond this file, must the next Turn read to have full context?\n\nA fresh Turn has no other context. If something isn't here or in a file question 5 names, it doesn't exist. Section headings may vary by workflow, but these five answers must be explicit and easy to find. Question 5 is the tracker's pointer into the rest of the Dossier (see [The Dossier](#the-dossier)): on a single-file Run its honest answer is \"nothing else\"; once anything has been shed, it names exactly which files, so a fresh Turn's opening read stays small even when the Dossier has grown.\n\n### Terminal markers\n\nDefault markers (workflows may override \u2014 use the markers configured for the active workflow):\n\n- `<!-- WORKFLOW_COMPLETE -->` \u2014 all work done and verified\n- `<!-- WORKFLOW_BLOCKED -->` or `<!-- WORKFLOW_BLOCKED: reason -->` \u2014 you need a human: a question only they can answer, or an external blocker only they can clear\n\nRules:\n\n- Only the last non-empty line of the tracker is authoritative. Marker-like text in notes, examples, or quoted instructions earlier in the file is ignored.\n- When you write a terminal marker, it must be the final non-empty line of the tracker. Put every summary, status note, and next-step sentence before the marker. Never append prose after it.\n- The runner trusts markers unconditionally. A premature `WORKFLOW_COMPLETE` ends the run with no automatic recovery \u2014 write it only when its criteria are fully met.\n- Include a concrete reason after `WORKFLOW_BLOCKED:` whenever possible \u2014 it is what the human sees when deciding how to answer you. Put the full question there.\n\n## The Dossier\n\n`.athena/<session_id>/` is the Dossier: the tracker plus everywhere it can shed cold detail once shedding earns its keep.\n\n```\ntracker.md state: index, open loops, next action, terminal marker\nunits/<slug>.md journal: contract, problem, design + rationale, build state, gate evidence\norientation.md cross-unit knowledge, revised in place\nhandoff/NNN.md episodic distillation, written at a Handover (chain, newest is mandatory reading)\n```\n\n**On most Runs the Dossier never grows past `tracker.md` itself** \u2014 one file, no ceremony, no extra directories, nothing below this line to act on. Nothing here changes how a short, single-unit Run works.\n\nA **unit** is a bounded piece of the plan that reaches its own closed state while the Run keeps going \u2014 one issue in a multi-issue epic, one phase of a larger goal. A Run with only one unit never closes a unit while another stays open, so it can never trip the first shed trigger below.\n\n### When to shed\n\nTwo triggers, both structural. A single-unit Run can only ever reach the second, and only once that unit has outgrown the bound:\n\n- **A unit closes while another is still open.** Cut that unit's whole detail out of the tracker now, before starting the next one.\n- **The tracker crosses ~8,000 tokens** (roughly 32,000 characters \u2014 a token is about 4 characters at this codebase's measured rate). This is a backstop for one long single unit, not a target to design toward \u2014 shed the completed phases of the still-open unit.\n\n### How to shed\n\nShedding is three positive acts on one whole named `##` section \u2014 never a summary, never a partial move:\n\n1. **Cut** the section, verbatim, out of `tracker.md`.\n2. **Paste** it, verbatim, into `units/<slug>.md` (create the file on that unit's first shed).\n3. **Pointer** \u2014 leave one index row in the tracker's unit table with a relative path to where the detail went.\n\nContent is demoted, never deleted. If you find yourself rewording or condensing while moving text, stop \u2014 that is not shedding, that is exactly where fidelity leaks.\n\n### The unit table\n\nThe tracker's unit table is the index the runner reads to keep the CLI's own task list in sync with your plan \u2014 it is parsed by tooling, so its shape is fixed, not free-form prose. Keep it under a `## Units` heading, as a GFM table with exactly two columns:\n\n```\n## Units\n\n| Unit | Record |\n| ---------------------- | --------------------------- |\n| Add the size nudge | units/size-nudge.md |\n| Wire up task projection | units/task-projection.md |\n```\n\n- **Unit** \u2014 a short human-readable label for the unit (what step 3 of [How to shed](#how-to-shed) points at).\n- **Record** \u2014 the path to that unit's `units/<slug>.md`, relative to the tracker's own directory.\n\nOne row per unit that has been shed; units still fully inline in the tracker (never shed) don't need a row. A row whose Record file is missing, unreadable, or malformed is simply skipped by the runner \u2014 never guessed at, never a failure (see the next section). Malformed rows in an otherwise-fine table don't invalidate the rest of the table.\n\nEach `units/<slug>.md` record opens with a small YAML frontmatter block the runner reads to know the unit's state:\n\n```\n---\nstatus: open\ngates: []\n---\n\n<the shed section content, verbatim>\n```\n\n- `status` \u2014 `open` or `closed`, exactly (case-insensitive). Any other value, or a missing `status` key, means the runner treats the whole record as unparseable for projection purposes \u2014 it still exists as your durable journal, it just doesn't surface in the task list.\n- `gates` \u2014 reserved for future gate-evidence tracking; not read by anything yet. Leave it present but empty, or omit it \u2014 either is fine today.\n\nThis frontmatter is the only structured part of a unit record; everything after the closing `---` is free-form prose exactly like the rest of the Dossier.\n\n### Parsing is best-effort, never a gate\n\nThe unit table and unit-record frontmatter exist so the runner can mirror your plan into the harness's own task list (see [Task UI projection](#task-ui-projection)) and, separately, so it can nudge you when the tracker grows past its size backstop. Both are conveniences layered on top of the Dossier, not requirements on it:\n\n- A missing table, an extra column, a typo'd status, a unit record that fails to parse \u2014 none of it fails your Turn, and none of it is worth stopping to fix for the runner's sake. The runner silently skips whatever it can't parse and, at most, folds a size nudge into your next prompt.\n- Never restructure the tracker or a unit record just to make automated parsing happy at the expense of the content itself. If the table and the prose disagree, the prose (what you actually did) is the truth.\n\n### orientation.md\n\nKnowledge that spans more than one unit lives here, revised in place instead of appended to \u2014 the one Dossier surface you edit rather than move things into. Two sections with different rules:\n\n- **Established** \u2014 decisions. Never decay; only explicitly superseded (\"we chose X over Y because Z \u2014 superseded 2026-09-03: \u2026\").\n- **Observed** \u2014 facts carrying a `file.ts:123`-style anchor. Revalidate the anchor before relying on it; code moves.\n\nA correction is an edit to the existing entry, not a new capitalized warning stacked on top of stale text.\n\n## Run until done; declare when blocked\n\nThe loop's contract with you:\n\n- **Do not stop early.** There is no checkpoint budget and no reason to end a Turn \"to be safe\" \u2014 context refresh is the runner's job, not yours. When your context approaches its bound the runner performs a **Handover**: your conversation is distilled into a Handoff file and a fresh session picks up seamlessly from it plus the tracker. You will not see this happen; just keep the tracker current so nothing is lost.\n- **Stopping without a marker is a mistake**, not a signal. The runner reads it as a premature stop and resumes you with a corrective prompt; repeated markerless stops without tracker progress escalate to a human. Never stop as a way of asking \"should I continue?\" \u2014 the answer is always to continue or to declare.\n- **Need a human? Declare it.** Write `WORKFLOW_BLOCKED: <your question or blocker>` as the tracker's final non-empty line and end. The run suspends until a human replies; their reply resumes your session with the answer. This is the only correct way to wait for a person \u2014 an interactive question asked into an unattended run cannot be answered.\n- Transient infrastructure failures are not yours to manage: the runner retries them by resuming your session. Just make sure the tracker reflects reality before risky operations.\n\n## Phases\n\n### Orient (Turn 1)\n\n1. **Replace the skeleton immediately**, before any domain work. Even a three-line tracker (goal + \"orienting\") protects you if the Turn dies during setup.\n2. Identify and load the applicable workflow skills before doing domain work. If a workflow, plugin, or local skill table names a relevant skill, read it fully and follow it. Do not assume you already know the workflow's conventions, tool sequence, quality gates, or implementation details.\n3. Use a dedicated git worktree for repository-changing work. If you are not already inside a task-specific worktree, create or enter one before editing files, record its branch/path in the tracker, and continue there. Skip this only when the workflow explicitly forbids it or the task is read-only.\n4. Run the workflow's orientation steps exactly as written. These vary by domain \u2014 a test-writing workflow explores the product in a browser; a migration workflow audits the schema. The workflow defines what orientation means. Do not skip, reorder, reinterpret, or replace workflow steps with a generic approach unless the workflow explicitly allows it or the tracker records a concrete blocker that makes the written step impossible.\n5. Refine the tracker into a granular plan. Each task a concrete, verifiable unit of work, including verification steps (running checks, reviewing output) \u2014 not just implementation. Vague tasks (\"write tests\") cannot be meaningfully resumed by a future Turn that has no idea what they mean here.\n6. Record concrete observations \u2014 what you actually saw, not what you assumed. Wrong assumptions burn entire future Turns on rework.\n7. **Single-Turn requests still go through this phase.** If the entire request is satisfied quickly, write a minimal tracker (what was asked, what was done, the outcome) and append `<!-- WORKFLOW_COMPLETE -->`. Leaving the skeleton in place gets you nudged \u2014 the runner cannot trust work it can't read from the tracker \u2014 and repeated stops with an untouched skeleton escalate to a human.\n\n### Execute (continuation)\n\n- Work from where the tracker says, in the workflow's prescribed sequence, and keep going until the work is done or a declared blocker stops you.\n- Be strict with workflow steps. Before starting each unit, identify the next required workflow step from the workflow document and tracker, follow it as written, and record completion or blockers against that step. Do not substitute your own process, collapse separate gates into one, or advance past an unchecked step.\n- Be strict with skills. Before each new activity, check the workflow, plugin metadata, local skill table, and tracker for relevant skills. Load the appropriate skill first, read it completely, and follow its instructions. If no skill applies, record that explicitly in the tracker before proceeding. Skills carry the implementation detail (scaffolding steps, locator rules, anti-patterns, code templates) that this protocol intentionally doesn't repeat.\n- Keep repository work inside the recorded git worktree. If a continuation starts outside the recorded worktree, enter it before editing. If no worktree is recorded and edits are still required, create or enter one before proceeding.\n- Delegate heavy exploration or generation to subagents via the Task tool. Pass file paths, conventions, and concrete output expectations; tell them which skill to load. Respect the workflow's **delegation constraints** \u2014 some operations must run in the main agent because their output is proof, or because the main agent needs to interpret results in context.\n- Run quality gates in order. Do not skip \u2014 they exist because skipping cascades into rework. On a failing verdict, address the issues and re-run before proceeding. Respect the workflow's **retry limits**: repeated failure usually signals a deeper issue another retry won't fix.\n\n### End\n\nYou end the run only by declaring:\n\n1. Tracker reflects all progress, discoveries, and blockers.\n2. Tracker says clearly what a fresh Turn would need to do first (a Handover can happen at any time).\n3. If all work is verified: append the completion marker as the final non-empty line.\n4. If a human is needed to proceed: append the blocked marker as the final non-empty line, with the question or blocker spelled out as the reason.\n\n## When to write the tracker\n\nWrite on **concrete triggers**, not on a vague sense of \"meaningful progress.\" The right cadence sits between every-tool-call (noisy log, wastes tokens) and end-of-run (everything lost if you die mid-task). This matters more, not less, now that Turns run long: the tracker (plus the Handoff file at a Handover) is what carries a killed or reset session.\n\n- **Discrete unit done** \u2014 file written, fix applied, test run, gate passed. Reflect the new reality before starting the next unit. If another unit is still open, this is also a shed trigger (see [The Dossier](#the-dossier)): cut this unit's detail into `units/<slug>.md` before you start the next one.\n- **Insight learned** \u2014 API quirk, config field that turned out to matter, dead end ruled out, decision between two approaches. Insights are tracker-worthy even when no code changed; rediscovering them costs a future Turn a full re-exploration. The tracker is a knowledge ledger, not just a task log. Insight that spans more than one unit belongs in `orientation.md`, not the tracker.\n- **About to do something risky or long-running** \u2014 subagent dispatch, long build, flaky external call, large refactor. Write _first_, then act. If the operation kills your Turn, only what's on disk survives.\n- **Plan changed** \u2014 task resequenced, new task surfaced, planned task no longer needed. Stale plans poison continuation Turns.\n- **The tracker crosses the ~8,000-token backstop** \u2014 shed the completed phases of the still-open unit into its `units/<slug>.md` record now, mid-unit; don't wait for it to close.\n- **You haven't written in a while** \u2014 if you can't remember the last update, you've gone too long. A short defensive update (\"doing X, last completed Y, next is Z\") beats nothing.\n\nEach update covers: what changed (work or knowledge), what's now next, and any caveat a future Turn needs. Don't transcribe tool calls \u2014 the tracker is a contract with your future self, not a replay log.\n\nThe cost of one extra tracker update is a few tokens. The cost of dying without one is rework. Bias toward writing.\n\n## Task UI projection\n\nThe tracker is the durable source of truth. Your harness's task tools are a session-scoped UI projection of the same plan, shown to the user in their CLI widget. They do not survive process exit.\n\n{{TASK_TOOL_INSTRUCTIONS}}\n\n- **Turn 1, after orientation:** project the tracker's task plan into the task tools.\n- **In a fresh continuation (e.g. after a Handover):** recreate the projection from the tracker; do not assume task IDs from prior sessions still exist.\n- **During work:** update both \u2014 the task tools for immediate UI feedback, the tracker for persistence \u2014 in the same working phase.\n\nSeparately, the runner independently re-derives a task list from the tracker's [unit table](#the-unit-table) and each unit record's frontmatter after every Turn, and mirrors it into the CLI's own task display. This is a backstop, not a substitute for the above \u2014 it only ever reaches `open`/`closed`, so keep calling the task tools yourself for anything finer-grained. It never blocks or fails a Turn: an unparseable table or record just means that Turn's mirror is skipped, exactly as described in [Parsing is best-effort, never a gate](#parsing-is-best-effort-never-a-gate).\n\n## Quick reference\n\n- [ ] Fresh Turn: read the tracker (and any named Handoff file) before doing anything else\n- [ ] Replace the skeleton immediately, even for single-Turn requests\n- [ ] Run until the work is done \u2014 do not stop at checkpoints, and never stop as a way of asking permission to continue\n- [ ] Need a human? Declare it: `WORKFLOW_BLOCKED: <question>` as the final non-empty line, then end\n- [ ] Update the tracker on concrete triggers \u2014 unit done, insight learned, risky op pending, plan changed\n- [ ] Shed a unit's detail into `units/<slug>.md` the moment it closes while another stays open, or the tracker crosses ~8,000 tokens \u2014 cut, paste, pointer, never summarize\n- [ ] Keep the unit table and each shed record's `status: open|closed` frontmatter current \u2014 the runner mirrors them into the task list and skips silently on any parse miss\n- [ ] After a Handover: fold the Handoff's durable content into the tracker or open unit record before any domain work\n- [ ] Project the tracker plan into task tools at session start; keep both in sync as work lands\n- [ ] Follow the workflow steps as written; do not skip, reorder, or substitute your own process\n- [ ] Load the appropriate skill before each activity; do not rely on assumed knowledge\n- [ ] Use and record a dedicated git worktree for repository-changing work\n- [ ] Run quality gates in order; respect delegation constraints and retry limits\n- [ ] Write the completion marker only when all work is verified, and make it the final non-empty line\n";
1689
1797
 
1690
1798
  // src/core/workflows/stateMachine.ts
1691
1799
  function buildTaskToolInstructions(harness) {
@@ -1725,7 +1833,7 @@ function readWorkflowOverride(projectDir, workflow, sessionId, trackerPath, harn
1725
1833
  if (!workflow?.workflowFile) {
1726
1834
  return { workflowOverride: void 0, warnings: [] };
1727
1835
  }
1728
- const resolvedPath = path8.isAbsolute(workflow.workflowFile) ? workflow.workflowFile : path8.resolve(projectDir, workflow.workflowFile);
1836
+ const resolvedPath = path9.isAbsolute(workflow.workflowFile) ? workflow.workflowFile : path9.resolve(projectDir, workflow.workflowFile);
1729
1837
  let workflowContent;
1730
1838
  try {
1731
1839
  workflowContent = fs10.readFileSync(resolvedPath, "utf-8");
@@ -1742,8 +1850,8 @@ function readWorkflowOverride(projectDir, workflow, sessionId, trackerPath, harn
1742
1850
  sessionId,
1743
1851
  trackerPath: trackerPath ?? void 0
1744
1852
  });
1745
- const workflowDir = path8.dirname(resolvedPath);
1746
- const composedPath = path8.join(workflowDir, ".composed-system-prompt.md");
1853
+ const workflowDir = path9.dirname(resolvedPath);
1854
+ const composedPath = path9.join(workflowDir, ".composed-system-prompt.md");
1747
1855
  fs10.writeFileSync(composedPath, composed, "utf-8");
1748
1856
  return {
1749
1857
  workflowOverride: {
@@ -1771,7 +1879,7 @@ function resolveTrackerPath(input) {
1771
1879
  return null;
1772
1880
  }
1773
1881
  const promptPath = input.sessionId ? rawPath.replaceAll("{sessionId}", input.sessionId) : rawPath;
1774
- const absolutePath = path8.isAbsolute(promptPath) ? promptPath : path8.resolve(input.projectDir, promptPath);
1882
+ const absolutePath = path9.isAbsolute(promptPath) ? promptPath : path9.resolve(input.projectDir, promptPath);
1775
1883
  return {
1776
1884
  absolutePath,
1777
1885
  promptPath
@@ -1816,14 +1924,13 @@ function prepareWorkflowTurn(state, input) {
1816
1924
  import { useCallback, useEffect, useRef, useState } from "react";
1817
1925
 
1818
1926
  // src/core/workflows/workflowRunner.ts
1819
- import crypto from "crypto";
1927
+ import crypto2 from "crypto";
1820
1928
  import fs12 from "fs";
1821
- import path9 from "path";
1929
+ import path10 from "path";
1822
1930
 
1823
1931
  // src/core/workflows/terminalOutcome.ts
1824
1932
  import fs11 from "fs";
1825
1933
  var MISSING_TRACKER_MESSAGE = "the tracker file went missing during the run \u2014 the workflow can no longer verify progress";
1826
- var SKELETON_NOT_REPLACED_MESSAGE = "tracker skeleton was never replaced \u2014 Claude did not bootstrap the tracker";
1827
1934
  var MISPLACED_TERMINAL_MARKER_MESSAGE = "terminal workflow marker is not the final non-empty line of the tracker; move all summary text above the marker";
1828
1935
  function resolveTurnOutcome(input) {
1829
1936
  const { trackerPath, loop, iteration } = input;
@@ -1835,13 +1942,6 @@ function resolveTurnOutcome(input) {
1835
1942
  };
1836
1943
  }
1837
1944
  const tracker = parseTrackerState(readTracker(trackerPath), loop);
1838
- if (tracker.skeletonNotReplaced) {
1839
- return {
1840
- kind: "stop",
1841
- status: "failed",
1842
- stopReason: SKELETON_NOT_REPLACED_MESSAGE
1843
- };
1844
- }
1845
1945
  if (tracker.misplacedTerminalMarker) {
1846
1946
  return {
1847
1947
  kind: "stop",
@@ -1853,14 +1953,640 @@ function resolveTurnOutcome(input) {
1853
1953
  return { kind: "stop", status: "completed" };
1854
1954
  }
1855
1955
  if (tracker.blocked) {
1856
- return { kind: "stop", status: "blocked", stopReason: tracker.blockedReason };
1956
+ return {
1957
+ kind: "suspend",
1958
+ status: "awaiting_attention",
1959
+ stopReason: tracker.blockedReason ? `agent declared WORKFLOW_BLOCKED: ${tracker.blockedReason}` : "agent declared WORKFLOW_BLOCKED"
1960
+ };
1857
1961
  }
1858
1962
  if (iteration >= loop.maxIterations) {
1859
- return { kind: "stop", status: "exhausted" };
1963
+ return {
1964
+ kind: "suspend",
1965
+ status: "awaiting_attention",
1966
+ stopReason: `iteration ceiling reached: ${loop.maxIterations} iteration${loop.maxIterations === 1 ? "" : "s"} (maxIterations) used without a terminal marker`
1967
+ };
1860
1968
  }
1861
1969
  return { kind: "continue" };
1862
1970
  }
1863
1971
 
1972
+ // src/core/runtime/failureTaxonomy.ts
1973
+ var RULES = [
1974
+ // ── hard, specific ──
1975
+ {
1976
+ pattern: /\b401\b|authentication_error|unauthorized|invalid.?api.?key|api key.*(invalid|revoked|expired)|oauth token.*(expired|revoked)|not.?logged.?in|please.*log ?in/i,
1977
+ classification: { kind: "hard", code: "auth" }
1978
+ },
1979
+ {
1980
+ pattern: /\b402\b|billing|credit balance|insufficient.?(funds|credits?|quota)|payment required|plan limit/i,
1981
+ classification: { kind: "hard", code: "billing" }
1982
+ },
1983
+ {
1984
+ pattern: /model.{0,20}not.?(found|supported|available)|no such model/i,
1985
+ classification: { kind: "hard", code: "model_not_found" }
1986
+ },
1987
+ // ── transient ──
1988
+ {
1989
+ pattern: /\b429\b|rate.?limit/i,
1990
+ classification: { kind: "transient", code: "rate_limit" }
1991
+ },
1992
+ {
1993
+ pattern: /\b529\b|overloaded/i,
1994
+ classification: { kind: "transient", code: "overloaded" }
1995
+ },
1996
+ {
1997
+ pattern: /\b(500|502|503|504)\b|internal server error|bad gateway|service unavailable|gateway timeout|server_error|api_error/i,
1998
+ classification: { kind: "transient", code: "server_error" }
1999
+ },
2000
+ {
2001
+ pattern: /econnrefused|econnreset|etimedout|enotfound|eai_again|epipe|socket hang up|fetch failed|network error|connection (error|closed|refused|reset)/i,
2002
+ classification: { kind: "transient", code: "network" }
2003
+ },
2004
+ // ── hard, generic (after the specific hard + transient rules) ──
2005
+ {
2006
+ pattern: /\b400\b|invalid_request_error|invalid request|malformed request/i,
2007
+ classification: { kind: "hard", code: "invalid_request" }
2008
+ }
2009
+ ];
2010
+ function classifyTurnFailure(input) {
2011
+ const haystack = [input.errorMessage, input.lastStderr, input.lastMessage].filter((part) => typeof part === "string").join("\n");
2012
+ for (const rule of RULES) {
2013
+ if (rule.pattern.test(haystack)) {
2014
+ return rule.classification;
2015
+ }
2016
+ }
2017
+ return { kind: "hard", code: "unclassified" };
2018
+ }
2019
+
2020
+ // src/core/workflows/runMachine.ts
2021
+ import crypto from "crypto";
2022
+ function hashTrackerContent(content) {
2023
+ return crypto.createHash("sha256").update(content).digest("hex");
2024
+ }
2025
+ function buildHandoverSeedPrompt(handoffPath, trackerPath) {
2026
+ return `A Handover occurred: the previous agent session reached its context bound and was distilled into a Handoff file. Read the Handoff file at ${handoffPath}` + (trackerPath ? ` and the tracker at ${trackerPath}` : "") + `. Before any domain work: fold whatever durable content the Handoff file records into the tracker` + (trackerPath ? ` at ${trackerPath}` : "") + ` and the open unit's record (ADR 0015 \xA78) \u2014 that fold-in is itself the tracker's next edit. Only once it is written should you continue the work from exactly where it stands. Do not redo completed work, and do not re-litigate decisions the Handoff file records.`;
2027
+ }
2028
+ function buildWakePrompt(reply, trackerPath) {
2029
+ return `This workflow run was suspended awaiting a human; it is now resumed. The human replied:
2030
+
2031
+ ${reply}
2032
+
2033
+ ` + (trackerPath ? `Read the tracker at ${trackerPath} for the task and its current state, apply the reply, and continue the workflow. ` : `Apply the reply and continue the workflow. `) + `Keep the tracker current as you work \u2014 if it still contains the runner's skeleton, replace it while orienting \u2014 and end by declaring a terminal marker as usual.`;
2034
+ }
2035
+ function buildFailureDetail(event) {
2036
+ const parts = [];
2037
+ if (event.errorMessage) {
2038
+ parts.push(event.errorMessage);
2039
+ } else if (event.exitCode !== null) {
2040
+ parts.push(`Process exited with code ${event.exitCode}`);
2041
+ }
2042
+ if (event.lastStderr) {
2043
+ parts.push(event.lastStderr);
2044
+ } else if (event.streamMessage) {
2045
+ parts.push(event.streamMessage.slice(0, 300));
2046
+ }
2047
+ return parts.join(": ") || "Turn failed";
2048
+ }
2049
+ function createInitialRun(cfg, opts) {
2050
+ if (opts.waking && opts.resumedMemory) {
2051
+ return {
2052
+ phase: {
2053
+ kind: "awaiting_attention",
2054
+ stopReason: opts.awaitingAttentionStopReason ?? ""
2055
+ },
2056
+ memory: opts.resumedMemory
2057
+ };
2058
+ }
2059
+ if (opts.resumedMemory) {
2060
+ return {
2061
+ phase: {
2062
+ kind: "backing_off",
2063
+ ms: 0,
2064
+ resume: {
2065
+ kind: "turn",
2066
+ prompt: opts.resumedMemory.lastStopPrompt,
2067
+ continuation: opts.resumedMemory.lastStopContinuation
2068
+ }
2069
+ },
2070
+ memory: opts.resumedMemory
2071
+ };
2072
+ }
2073
+ const continuation = opts.initialContinuation ?? { mode: "fresh" };
2074
+ const iteration = 1;
2075
+ const prepared = prepareWorkflowTurn(cfg.workflowState, {
2076
+ prompt: cfg.initialPrompt,
2077
+ iteration,
2078
+ configOverride: void 0
2079
+ });
2080
+ const prompt = opts.waking ? buildWakePrompt(cfg.initialPrompt, cfg.trackerPromptPath) : prepared.prompt;
2081
+ return {
2082
+ phase: {
2083
+ kind: "turn_in_flight",
2084
+ prompt,
2085
+ continuation,
2086
+ configOverride: prepared.configOverride
2087
+ },
2088
+ memory: {
2089
+ iteration,
2090
+ nudgeStreak: 0,
2091
+ retryStreak: 0,
2092
+ lastTrackerHash: null,
2093
+ lastStopPrompt: prompt,
2094
+ lastStopContinuation: continuation,
2095
+ lastHandoffSizeBytes: null
2096
+ }
2097
+ };
2098
+ }
2099
+ function handleTurnInFlight(phase, memory, event, cfg) {
2100
+ if (event.cancelled) {
2101
+ return { phase: { kind: "cancelled" }, memory, actions: [{ type: "persist" }] };
2102
+ }
2103
+ if (event.handoverRequestHandle !== null) {
2104
+ return {
2105
+ phase: {
2106
+ kind: "handing_over",
2107
+ handle: event.handoverRequestHandle,
2108
+ // Reuse this Turn's own prepared configOverride (the same object
2109
+ // the primary `start_turn` action used) rather than recomputing —
2110
+ // matches workflowRunner.ts's original `prepared.configOverride`
2111
+ // reuse at the fork call site exactly. Stored on the phase too
2112
+ // (not just the action) so a transient-retry (§8) can re-issue the
2113
+ // fork with the same override.
2114
+ configOverride: phase.configOverride
2115
+ },
2116
+ memory,
2117
+ actions: [
2118
+ {
2119
+ type: "start_fork_turn",
2120
+ handle: event.handoverRequestHandle,
2121
+ configOverride: phase.configOverride
2122
+ }
2123
+ ]
2124
+ };
2125
+ }
2126
+ if (event.suspension) {
2127
+ return {
2128
+ phase: { kind: "awaiting_attention", stopReason: event.suspension.reason },
2129
+ memory,
2130
+ actions: [{ type: "persist" }]
2131
+ };
2132
+ }
2133
+ const failed = event.hasError || event.exitCode !== null && event.exitCode !== 0;
2134
+ if (failed) {
2135
+ const failureDetail = buildFailureDetail(event);
2136
+ if (cfg.loop?.enabled) {
2137
+ const classification = classifyTurnFailure({
2138
+ errorMessage: event.errorMessage,
2139
+ lastStderr: event.stderrTail ?? event.lastStderr,
2140
+ lastMessage: event.streamMessage
2141
+ });
2142
+ if (classification.kind === "transient") {
2143
+ const retryStreak = memory.retryStreak + 1;
2144
+ const retryCap = cfg.loop.retryCap ?? DEFAULT_RETRY_CAP;
2145
+ if (retryStreak > retryCap) {
2146
+ return {
2147
+ phase: {
2148
+ kind: "awaiting_attention",
2149
+ stopReason: `retry cap reached: ${retryCap} transient failure${retryCap === 1 ? "" : "s"} (retryCap); last (${classification.code}): ${failureDetail}`
2150
+ },
2151
+ memory: { ...memory, retryStreak },
2152
+ actions: [{ type: "persist" }]
2153
+ };
2154
+ }
2155
+ const backoffBase = cfg.loop.retryBackoffMs ?? DEFAULT_RETRY_BACKOFF_MS;
2156
+ const ms = backoffBase * 2 ** (retryStreak - 1);
2157
+ return {
2158
+ phase: {
2159
+ kind: "backing_off",
2160
+ ms,
2161
+ resume: {
2162
+ kind: "turn",
2163
+ prompt: phase.prompt,
2164
+ continuation: phase.continuation
2165
+ }
2166
+ },
2167
+ memory: { ...memory, retryStreak },
2168
+ actions: [{ type: "wait", ms }]
2169
+ };
2170
+ }
2171
+ if (phase.continuation.mode === "resume") {
2172
+ const continuation2 = { mode: "fresh" };
2173
+ const prepared2 = prepareWorkflowTurn(cfg.workflowState, {
2174
+ prompt: cfg.initialPrompt,
2175
+ iteration: memory.iteration,
2176
+ configOverride: void 0
2177
+ });
2178
+ return {
2179
+ phase: {
2180
+ kind: "turn_in_flight",
2181
+ prompt: prepared2.prompt,
2182
+ continuation: continuation2,
2183
+ configOverride: prepared2.configOverride
2184
+ },
2185
+ memory: {
2186
+ ...memory,
2187
+ lastStopPrompt: prepared2.prompt,
2188
+ lastStopContinuation: continuation2
2189
+ },
2190
+ actions: [
2191
+ { type: "persist" },
2192
+ {
2193
+ type: "start_turn",
2194
+ prompt: prepared2.prompt,
2195
+ continuation: continuation2,
2196
+ configOverride: prepared2.configOverride
2197
+ }
2198
+ ]
2199
+ };
2200
+ }
2201
+ return {
2202
+ phase: {
2203
+ kind: "awaiting_attention",
2204
+ stopReason: `hard failure (${classification.code}): ${failureDetail} \u2014 not retried; needs a human`
2205
+ },
2206
+ memory,
2207
+ actions: [{ type: "persist" }]
2208
+ };
2209
+ }
2210
+ return {
2211
+ phase: { kind: "failed", stopReason: failureDetail },
2212
+ memory,
2213
+ actions: [{ type: "persist" }]
2214
+ };
2215
+ }
2216
+ const memoryAfterSuccess = { ...memory, retryStreak: 0 };
2217
+ if (event.transportBroken) {
2218
+ return {
2219
+ phase: {
2220
+ kind: "failed",
2221
+ stopReason: `Hook transport broken: observed a tool use in the Claude stream but received no PreToolUse events.`
2222
+ },
2223
+ memory: memoryAfterSuccess,
2224
+ actions: [{ type: "persist" }]
2225
+ };
2226
+ }
2227
+ if (!cfg.loop?.enabled) {
2228
+ return {
2229
+ phase: { kind: "completed" },
2230
+ memory: memoryAfterSuccess,
2231
+ actions: [{ type: "persist" }]
2232
+ };
2233
+ }
2234
+ const loop = cfg.loop;
2235
+ if (event.outcome && (event.outcome.kind === "stop" || event.outcome.kind === "suspend")) {
2236
+ const outcome = event.outcome;
2237
+ const nextPhase = outcome.status === "completed" ? { kind: "completed" } : outcome.status === "failed" ? { kind: "failed", stopReason: outcome.stopReason } : {
2238
+ kind: "awaiting_attention",
2239
+ stopReason: outcome.stopReason ?? "stopped"
2240
+ };
2241
+ return {
2242
+ phase: nextPhase,
2243
+ memory: memoryAfterSuccess,
2244
+ actions: [{ type: "persist" }]
2245
+ };
2246
+ }
2247
+ const trackerHash = hashTrackerContent(event.trackerContent);
2248
+ let nudgeStreak = memoryAfterSuccess.nudgeStreak;
2249
+ if (trackerHash !== memoryAfterSuccess.lastTrackerHash) {
2250
+ nudgeStreak = 0;
2251
+ }
2252
+ const memoryWithHash = {
2253
+ ...memoryAfterSuccess,
2254
+ lastTrackerHash: trackerHash,
2255
+ nudgeStreak
2256
+ };
2257
+ const sizeNudgeSuffix = estimateTokenCount(event.trackerContent) > DEFAULT_TRACKER_TOKEN_BOUND ? buildTrackerSizeNudgeSuffix(cfg.trackerPromptPath) : "";
2258
+ if (event.adapterSessionId) {
2259
+ const nextNudgeStreak = nudgeStreak + 1;
2260
+ const nudgeCap = loop.nudgeCap ?? DEFAULT_NUDGE_CAP;
2261
+ if (nextNudgeStreak > nudgeCap) {
2262
+ return {
2263
+ phase: {
2264
+ kind: "awaiting_attention",
2265
+ stopReason: `nudge cap reached: ${nudgeCap} nudge${nudgeCap === 1 ? "" : "s"} (nudgeCap) without tracker progress or a terminal marker`
2266
+ },
2267
+ memory: { ...memoryWithHash, nudgeStreak: nextNudgeStreak },
2268
+ actions: [{ type: "persist" }]
2269
+ };
2270
+ }
2271
+ const promptOverride = buildNudgePrompt(
2272
+ { ...loop, trackerPath: cfg.trackerPromptPath ?? loop.trackerPath },
2273
+ {
2274
+ skeletonNotReplaced: event.trackerContent.includes(
2275
+ TRACKER_SKELETON_MARKER
2276
+ )
2277
+ }
2278
+ ) + sizeNudgeSuffix;
2279
+ const nextIteration2 = memoryWithHash.iteration + 1;
2280
+ const continuation2 = {
2281
+ mode: "resume",
2282
+ handle: event.adapterSessionId
2283
+ };
2284
+ const prepared2 = prepareWorkflowTurn(cfg.workflowState, {
2285
+ prompt: cfg.initialPrompt,
2286
+ iteration: nextIteration2,
2287
+ configOverride: void 0
2288
+ });
2289
+ return {
2290
+ phase: {
2291
+ kind: "turn_in_flight",
2292
+ prompt: promptOverride,
2293
+ continuation: continuation2,
2294
+ configOverride: prepared2.configOverride
2295
+ },
2296
+ memory: {
2297
+ ...memoryWithHash,
2298
+ nudgeStreak: nextNudgeStreak,
2299
+ iteration: nextIteration2,
2300
+ lastStopPrompt: promptOverride,
2301
+ lastStopContinuation: continuation2
2302
+ },
2303
+ actions: [
2304
+ { type: "persist" },
2305
+ { type: "notify_iteration_complete" },
2306
+ {
2307
+ type: "start_turn",
2308
+ prompt: promptOverride,
2309
+ continuation: continuation2,
2310
+ configOverride: prepared2.configOverride
2311
+ }
2312
+ ]
2313
+ };
2314
+ }
2315
+ const nextIteration = memoryWithHash.iteration + 1;
2316
+ const continuation = { mode: "fresh" };
2317
+ const prepared = prepareWorkflowTurn(cfg.workflowState, {
2318
+ prompt: cfg.initialPrompt,
2319
+ iteration: nextIteration,
2320
+ configOverride: void 0
2321
+ });
2322
+ const promptWithSizeNudge = prepared.prompt + sizeNudgeSuffix;
2323
+ return {
2324
+ phase: {
2325
+ kind: "turn_in_flight",
2326
+ prompt: promptWithSizeNudge,
2327
+ continuation,
2328
+ configOverride: prepared.configOverride
2329
+ },
2330
+ memory: {
2331
+ ...memoryWithHash,
2332
+ iteration: nextIteration,
2333
+ lastStopPrompt: promptWithSizeNudge,
2334
+ lastStopContinuation: continuation
2335
+ },
2336
+ actions: [
2337
+ { type: "persist" },
2338
+ { type: "notify_iteration_complete" },
2339
+ {
2340
+ type: "start_turn",
2341
+ prompt: promptWithSizeNudge,
2342
+ continuation,
2343
+ configOverride: prepared.configOverride
2344
+ }
2345
+ ]
2346
+ };
2347
+ }
2348
+ function handleBackingOff(phase, memory, event, cfg) {
2349
+ if (event.cancelled) {
2350
+ return { phase: { kind: "cancelled" }, memory, actions: [{ type: "persist" }] };
2351
+ }
2352
+ if (phase.resume.kind === "fork") {
2353
+ return {
2354
+ phase: {
2355
+ kind: "handing_over",
2356
+ handle: phase.resume.handle,
2357
+ configOverride: phase.resume.configOverride,
2358
+ retried: true
2359
+ },
2360
+ memory,
2361
+ actions: [
2362
+ {
2363
+ type: "start_fork_turn",
2364
+ handle: phase.resume.handle,
2365
+ configOverride: phase.resume.configOverride
2366
+ }
2367
+ ]
2368
+ };
2369
+ }
2370
+ const continuation = event.adapterSessionId ? { mode: "resume", handle: event.adapterSessionId } : phase.resume.continuation;
2371
+ const prompt = continuation.mode === "fresh" ? phase.resume.prompt : cfg.loop ? buildContinuePrompt(cfg.loop) : "Continue.";
2372
+ const prepared = prepareWorkflowTurn(cfg.workflowState, {
2373
+ prompt: cfg.initialPrompt,
2374
+ iteration: memory.iteration,
2375
+ configOverride: void 0
2376
+ });
2377
+ return {
2378
+ phase: {
2379
+ kind: "turn_in_flight",
2380
+ prompt,
2381
+ continuation,
2382
+ configOverride: prepared.configOverride
2383
+ },
2384
+ memory: {
2385
+ ...memory,
2386
+ lastStopPrompt: prompt,
2387
+ lastStopContinuation: continuation
2388
+ },
2389
+ actions: [
2390
+ { type: "persist" },
2391
+ {
2392
+ type: "start_turn",
2393
+ prompt,
2394
+ continuation,
2395
+ configOverride: prepared.configOverride
2396
+ }
2397
+ ]
2398
+ };
2399
+ }
2400
+ function handleHandingOver(phase, memory, event, cfg) {
2401
+ if (event.cancelled) {
2402
+ return { phase: { kind: "cancelled" }, memory, actions: [{ type: "persist" }] };
2403
+ }
2404
+ const nextIteration = memory.iteration + 1;
2405
+ if (event.ok) {
2406
+ const continuation2 = { mode: "fresh" };
2407
+ const seedPrompt = buildHandoverSeedPrompt(
2408
+ event.handoffPath,
2409
+ cfg.trackerAbsPath ?? void 0
2410
+ );
2411
+ const prepared2 = prepareWorkflowTurn(cfg.workflowState, {
2412
+ prompt: cfg.initialPrompt,
2413
+ iteration: nextIteration,
2414
+ configOverride: void 0
2415
+ });
2416
+ return {
2417
+ phase: {
2418
+ kind: "turn_in_flight",
2419
+ prompt: seedPrompt,
2420
+ continuation: continuation2,
2421
+ configOverride: prepared2.configOverride
2422
+ },
2423
+ memory: {
2424
+ ...memory,
2425
+ iteration: nextIteration,
2426
+ lastStopPrompt: seedPrompt,
2427
+ lastStopContinuation: continuation2,
2428
+ lastHandoffSizeBytes: event.handoffSizeBytes
2429
+ },
2430
+ actions: [
2431
+ { type: "purge_handoffs" },
2432
+ { type: "persist" },
2433
+ {
2434
+ type: "start_turn",
2435
+ prompt: seedPrompt,
2436
+ continuation: continuation2,
2437
+ configOverride: prepared2.configOverride
2438
+ }
2439
+ ]
2440
+ };
2441
+ }
2442
+ if (event.transient && !phase.retried) {
2443
+ const ms = cfg.loop?.retryBackoffMs ?? DEFAULT_RETRY_BACKOFF_MS;
2444
+ return {
2445
+ phase: {
2446
+ kind: "backing_off",
2447
+ ms,
2448
+ resume: {
2449
+ kind: "fork",
2450
+ handle: phase.handle,
2451
+ configOverride: phase.configOverride
2452
+ }
2453
+ },
2454
+ memory,
2455
+ actions: [{ type: "wait", ms }]
2456
+ };
2457
+ }
2458
+ const continuation = { mode: "resume", handle: phase.handle };
2459
+ const prepared = prepareWorkflowTurn(cfg.workflowState, {
2460
+ prompt: cfg.initialPrompt,
2461
+ iteration: nextIteration,
2462
+ configOverride: void 0
2463
+ });
2464
+ return {
2465
+ phase: {
2466
+ kind: "turn_in_flight",
2467
+ prompt: prepared.prompt,
2468
+ continuation,
2469
+ configOverride: prepared.configOverride
2470
+ },
2471
+ memory: {
2472
+ ...memory,
2473
+ iteration: nextIteration,
2474
+ lastStopPrompt: prepared.prompt,
2475
+ lastStopContinuation: continuation
2476
+ },
2477
+ actions: [
2478
+ { type: "degrade_handover", handle: phase.handle },
2479
+ { type: "persist" },
2480
+ {
2481
+ type: "start_turn",
2482
+ prompt: prepared.prompt,
2483
+ continuation,
2484
+ configOverride: prepared.configOverride
2485
+ }
2486
+ ]
2487
+ };
2488
+ }
2489
+ function handleAwaitingAttention(memory, event, cfg) {
2490
+ const nextIteration = memory.iteration + 1;
2491
+ const prompt = buildWakePrompt(cfg.initialPrompt, cfg.trackerPromptPath);
2492
+ const prepared = prepareWorkflowTurn(cfg.workflowState, {
2493
+ prompt: cfg.initialPrompt,
2494
+ iteration: nextIteration,
2495
+ configOverride: void 0
2496
+ });
2497
+ return {
2498
+ phase: {
2499
+ kind: "turn_in_flight",
2500
+ prompt,
2501
+ continuation: event.continuation,
2502
+ configOverride: prepared.configOverride
2503
+ },
2504
+ memory: {
2505
+ ...memory,
2506
+ iteration: nextIteration,
2507
+ lastStopPrompt: prompt,
2508
+ lastStopContinuation: event.continuation
2509
+ },
2510
+ actions: [
2511
+ { type: "persist" },
2512
+ {
2513
+ type: "start_turn",
2514
+ prompt,
2515
+ continuation: event.continuation,
2516
+ configOverride: prepared.configOverride
2517
+ }
2518
+ ]
2519
+ };
2520
+ }
2521
+ function step(phase, memory, event, cfg) {
2522
+ switch (phase.kind) {
2523
+ case "turn_in_flight": {
2524
+ if (event.type !== "turn_finished") {
2525
+ throw new Error(
2526
+ `runMachine: phase 'turn_in_flight' received unexpected event '${event.type}'`
2527
+ );
2528
+ }
2529
+ return handleTurnInFlight(phase, memory, event, cfg);
2530
+ }
2531
+ case "backing_off": {
2532
+ if (event.type !== "backoff_elapsed") {
2533
+ throw new Error(
2534
+ `runMachine: phase 'backing_off' received unexpected event '${event.type}'`
2535
+ );
2536
+ }
2537
+ return handleBackingOff(phase, memory, event, cfg);
2538
+ }
2539
+ case "handing_over": {
2540
+ if (event.type !== "fork_finished") {
2541
+ throw new Error(
2542
+ `runMachine: phase 'handing_over' received unexpected event '${event.type}'`
2543
+ );
2544
+ }
2545
+ return handleHandingOver(phase, memory, event, cfg);
2546
+ }
2547
+ case "awaiting_attention": {
2548
+ if (event.type !== "woken") {
2549
+ throw new Error(
2550
+ `runMachine: phase 'awaiting_attention' received unexpected event '${event.type}'`
2551
+ );
2552
+ }
2553
+ return handleAwaitingAttention(memory, event, cfg);
2554
+ }
2555
+ case "completed":
2556
+ case "failed":
2557
+ case "cancelled": {
2558
+ throw new Error(
2559
+ `runMachine: step() called on terminal phase '${phase.kind}'`
2560
+ );
2561
+ }
2562
+ default: {
2563
+ const _exhaustive = phase;
2564
+ throw new Error(
2565
+ `runMachine: step() called on unrecognized phase kind '${_exhaustive.kind}'`
2566
+ );
2567
+ }
2568
+ }
2569
+ }
2570
+ function serializeRunMemory(memory) {
2571
+ return JSON.stringify(memory);
2572
+ }
2573
+ function deserializeRunMemory(json) {
2574
+ if (!json) return null;
2575
+ let parsed;
2576
+ try {
2577
+ parsed = JSON.parse(json);
2578
+ } catch {
2579
+ return null;
2580
+ }
2581
+ if (!parsed || typeof parsed !== "object") return null;
2582
+ const candidate = parsed;
2583
+ const continuation = candidate.lastStopContinuation;
2584
+ if (typeof candidate.iteration !== "number" || typeof candidate.nudgeStreak !== "number" || typeof candidate.retryStreak !== "number" || candidate.lastTrackerHash !== null && typeof candidate.lastTrackerHash !== "string" || typeof candidate.lastStopPrompt !== "string" || !continuation || typeof continuation.mode !== "string") {
2585
+ return null;
2586
+ }
2587
+ return candidate;
2588
+ }
2589
+
1864
2590
  // src/core/workflows/workflowRunner.ts
1865
2591
  var NULL_TOKENS = {
1866
2592
  input: null,
@@ -1881,7 +2607,7 @@ var TRACKER_SKELETON_TEMPLATE = `${TRACKER_SKELETON_MARKER}
1881
2607
  ---
1882
2608
 
1883
2609
  > This tracker was created by the runner. Update it as you work.
1884
- > See the Stateless Turn Protocol for tracker conventions.
2610
+ > See the Turn Protocol for tracker conventions.
1885
2611
 
1886
2612
  ## Status
1887
2613
 
@@ -1895,6 +2621,33 @@ _To be created during orientation._
1895
2621
 
1896
2622
  _No progress yet._
1897
2623
  `;
2624
+ function openRunSection(trackerPath, opts) {
2625
+ let existing;
2626
+ try {
2627
+ existing = fs12.readFileSync(trackerPath, "utf-8");
2628
+ } catch {
2629
+ return;
2630
+ }
2631
+ const banner = `
2632
+
2633
+ ---
2634
+
2635
+ ## New Workflow Run
2636
+
2637
+ **Run**: ${opts.runId}
2638
+ **Goal**: ${opts.goal}
2639
+
2640
+ _Sections above belong to earlier Workflow Runs in this Athena Session._
2641
+ `;
2642
+ try {
2643
+ fs12.writeFileSync(
2644
+ trackerPath,
2645
+ demoteTerminalMarkers(existing.trimEnd(), opts.markers) + banner,
2646
+ "utf-8"
2647
+ );
2648
+ } catch {
2649
+ }
2650
+ }
1898
2651
  function mergeTokens(base, next) {
1899
2652
  const input = (base.input ?? 0) + (next.input ?? 0);
1900
2653
  const output = (base.output ?? 0) + (next.output ?? 0);
@@ -1917,21 +2670,73 @@ function mergeTokens(base, next) {
1917
2670
  contextWindowSize: next.contextWindowSize ?? base.contextWindowSize
1918
2671
  };
1919
2672
  }
2673
+ function buildHandoffInvocationPrompt(handoffPath) {
2674
+ return `Invoke the handoff skill to write a Handoff file to ${handoffPath}. Do nothing else: no code changes, no tracker updates \u2014 only the Handoff file.`;
2675
+ }
2676
+ async function delayWithCancel(ms, isCancelled) {
2677
+ const slice = 250;
2678
+ for (let waited = 0; waited < ms && !isCancelled(); waited += slice) {
2679
+ await new Promise(
2680
+ (resolve) => setTimeout(resolve, Math.min(slice, ms - waited))
2681
+ );
2682
+ }
2683
+ }
2684
+ var HANDOFF_DIR_NAME = "handoff";
2685
+ var HANDOFF_RETAIN = 2;
2686
+ function listHandoffSeqs(dir) {
2687
+ try {
2688
+ return fs12.readdirSync(dir).map((name) => /^(\d{3})\.md$/.exec(name)?.[1]).filter((seq) => seq !== void 0).map(Number).sort((a, b) => a - b);
2689
+ } catch {
2690
+ return [];
2691
+ }
2692
+ }
2693
+ function handoffPathFor(dir, seq) {
2694
+ return path10.join(dir, `${String(seq).padStart(3, "0")}.md`);
2695
+ }
2696
+ function nextHandoffPath(dir) {
2697
+ const next = (listHandoffSeqs(dir).at(-1) ?? 0) + 1;
2698
+ fs12.mkdirSync(dir, { recursive: true });
2699
+ return handoffPathFor(dir, next);
2700
+ }
2701
+ function purgeHandoffs(dir, keep) {
2702
+ const seqs = listHandoffSeqs(dir);
2703
+ for (const seq of seqs.slice(0, Math.max(0, seqs.length - keep))) {
2704
+ try {
2705
+ fs12.rmSync(handoffPathFor(dir, seq), { force: true });
2706
+ } catch {
2707
+ }
2708
+ }
2709
+ }
1920
2710
  function defaultCreateTracker(trackerPath, content) {
1921
- fs12.mkdirSync(path9.dirname(trackerPath), { recursive: true });
2711
+ fs12.mkdirSync(path10.dirname(trackerPath), { recursive: true });
1922
2712
  try {
1923
2713
  fs12.writeFileSync(trackerPath, content, { encoding: "utf-8", flag: "wx" });
1924
2714
  } catch (e) {
1925
2715
  if (e.code !== "EEXIST") throw e;
1926
2716
  }
1927
2717
  }
2718
+ function isTerminalPhase(phase) {
2719
+ return phase.kind === "awaiting_attention" || phase.kind === "completed" || phase.kind === "failed" || phase.kind === "cancelled";
2720
+ }
2721
+ function terminalPhaseToStatus(phase) {
2722
+ switch (phase.kind) {
2723
+ case "awaiting_attention":
2724
+ return { status: "awaiting_attention", stopReason: phase.stopReason };
2725
+ case "completed":
2726
+ return { status: "completed" };
2727
+ case "failed":
2728
+ return { status: "failed", stopReason: phase.stopReason };
2729
+ case "cancelled":
2730
+ return { status: "cancelled" };
2731
+ }
2732
+ }
1928
2733
  function createWorkflowRunner(input) {
1929
- const runId = crypto.randomUUID();
2734
+ const runId = input.resumeRunId ?? crypto2.randomUUID();
1930
2735
  let cancelled = false;
1931
2736
  let status = "running";
1932
- let iterations = 0;
1933
2737
  let cumulativeTokens = { ...NULL_TOKENS };
1934
2738
  let stopReason;
2739
+ let memory;
1935
2740
  const trackerResolved = resolveTrackerPath({
1936
2741
  projectDir: input.projectDir,
1937
2742
  sessionId: input.sessionId,
@@ -1940,15 +2745,18 @@ function createWorkflowRunner(input) {
1940
2745
  const trackerAbsPath = trackerResolved?.absolutePath ?? null;
1941
2746
  const trackerPromptPath = trackerResolved?.promptPath;
1942
2747
  function snapshot() {
2748
+ const adapterSessionId = input.currentAdapterSessionId?.() ?? void 0;
1943
2749
  return {
1944
2750
  runId,
1945
2751
  sessionId: input.sessionId,
1946
2752
  workflowName: input.workflow?.name,
1947
- iteration: iterations,
2753
+ iteration: memory?.iteration ?? 0,
1948
2754
  maxIterations: input.workflow?.loop?.maxIterations ?? 1,
1949
2755
  status,
1950
2756
  stopReason,
1951
- trackerPath: trackerPromptPath
2757
+ trackerPath: trackerPromptPath,
2758
+ ...adapterSessionId ? { adapterSessionId } : {},
2759
+ ...memory ? { runMemoryJson: serializeRunMemory(memory) } : {}
1952
2760
  };
1953
2761
  }
1954
2762
  function persist() {
@@ -1957,6 +2765,16 @@ function createWorkflowRunner(input) {
1957
2765
  } catch {
1958
2766
  }
1959
2767
  }
2768
+ function handoffDirFor() {
2769
+ return path10.join(
2770
+ trackerAbsPath ? path10.dirname(trackerAbsPath) : path10.resolve(
2771
+ input.projectDir,
2772
+ ".athena",
2773
+ input.sessionId || "session"
2774
+ ),
2775
+ HANDOFF_DIR_NAME
2776
+ );
2777
+ }
1960
2778
  const result = (async () => {
1961
2779
  await Promise.resolve();
1962
2780
  if (trackerAbsPath && input.workflow?.loop?.enabled) {
@@ -1966,7 +2784,18 @@ function createWorkflowRunner(input) {
1966
2784
  input: input.prompt
1967
2785
  });
1968
2786
  const write = input.createTracker ?? defaultCreateTracker;
2787
+ const trackerExisted = fs12.existsSync(trackerAbsPath);
1969
2788
  write(trackerAbsPath, content);
2789
+ if (trackerExisted && !input.resumeRunId) {
2790
+ openRunSection(trackerAbsPath, {
2791
+ runId,
2792
+ goal: input.prompt,
2793
+ markers: {
2794
+ completionMarker: input.workflow.loop.completionMarker,
2795
+ blockedMarker: input.workflow.loop.blockedMarker
2796
+ }
2797
+ });
2798
+ }
1970
2799
  }
1971
2800
  persist();
1972
2801
  const workflowState = createWorkflowRunState({
@@ -1975,80 +2804,271 @@ function createWorkflowRunner(input) {
1975
2804
  workflow: input.workflow,
1976
2805
  harness: input.harness
1977
2806
  });
1978
- let nextContinuation = input.initialContinuation ?? {
1979
- mode: "fresh"
1980
- };
1981
2807
  const loop = input.workflow?.loop;
1982
- while (!cancelled) {
1983
- iterations++;
1984
- const prepared = prepareWorkflowTurn(workflowState, {
1985
- prompt: input.prompt,
1986
- iteration: iterations,
1987
- configOverride: void 0
1988
- });
2808
+ const cfg = {
2809
+ workflowState,
2810
+ initialPrompt: input.prompt,
2811
+ loop,
2812
+ trackerAbsPath,
2813
+ trackerPromptPath
2814
+ };
2815
+ const initial = createInitialRun(cfg, {
2816
+ initialContinuation: input.initialContinuation,
2817
+ waking: !!(input.resumeRunId && loop?.enabled),
2818
+ resumedMemory: input.resumedRunMemory,
2819
+ awaitingAttentionStopReason: input.resumedStopReason
2820
+ });
2821
+ let phase = initial.phase;
2822
+ memory = initial.memory;
2823
+ async function performStartTurn(prompt, continuation, configOverride) {
1989
2824
  const turnResult = await input.startTurn({
1990
- prompt: prepared.prompt,
1991
- continuation: nextContinuation,
1992
- configOverride: prepared.configOverride
2825
+ prompt,
2826
+ continuation,
2827
+ configOverride
1993
2828
  });
1994
2829
  if (cancelled) {
1995
- status = "cancelled";
1996
- persist();
1997
- break;
2830
+ return {
2831
+ type: "turn_finished",
2832
+ cancelled: true,
2833
+ hasError: false,
2834
+ exitCode: null,
2835
+ streamMessage: null,
2836
+ transportBroken: false,
2837
+ handoverRequestHandle: null,
2838
+ suspension: null,
2839
+ adapterSessionId: null,
2840
+ outcome: null,
2841
+ trackerContent: ""
2842
+ };
1998
2843
  }
1999
2844
  cumulativeTokens = mergeTokens(cumulativeTokens, turnResult.tokens);
2000
- if (turnResult.error || turnResult.exitCode !== null && turnResult.exitCode !== 0) {
2001
- status = "failed";
2002
- const parts = [];
2003
- if (turnResult.error?.message) {
2004
- parts.push(turnResult.error.message);
2005
- } else if (turnResult.exitCode !== null) {
2006
- parts.push(`Process exited with code ${turnResult.exitCode}`);
2007
- }
2008
- if (turnResult.lastStderr) {
2009
- parts.push(turnResult.lastStderr);
2010
- }
2011
- stopReason = parts.join(": ") || "Turn failed";
2012
- persist();
2013
- break;
2845
+ const handoverRequest = input.handover?.takeRequest() ?? null;
2846
+ if (handoverRequest) {
2847
+ return {
2848
+ type: "turn_finished",
2849
+ cancelled: false,
2850
+ hasError: false,
2851
+ exitCode: null,
2852
+ streamMessage: null,
2853
+ transportBroken: false,
2854
+ handoverRequestHandle: handoverRequest.handle,
2855
+ suspension: null,
2856
+ adapterSessionId: null,
2857
+ outcome: null,
2858
+ trackerContent: ""
2859
+ };
2014
2860
  }
2015
- const transport = turnResult.diagnostics?.transport;
2016
- if (transport && transport.streamToolUses > 0 && transport.preToolUseEvents === 0) {
2017
- status = "failed";
2018
- stopReason = `Hook transport broken: observed ${transport.streamToolUses} tool use(s) in Claude stream but received no PreToolUse events.`;
2019
- persist();
2020
- break;
2861
+ const suspension = input.checkSuspension?.() ?? null;
2862
+ if (suspension) {
2863
+ return {
2864
+ type: "turn_finished",
2865
+ cancelled: false,
2866
+ hasError: false,
2867
+ exitCode: null,
2868
+ streamMessage: null,
2869
+ transportBroken: false,
2870
+ handoverRequestHandle: null,
2871
+ suspension,
2872
+ adapterSessionId: null,
2873
+ outcome: null,
2874
+ trackerContent: ""
2875
+ };
2021
2876
  }
2022
- if (!loop?.enabled) {
2023
- status = "completed";
2024
- persist();
2025
- break;
2877
+ const adapterSessionId = input.currentAdapterSessionId?.() ?? null;
2878
+ const hasError = !!turnResult.error;
2879
+ const failed = hasError || turnResult.exitCode !== null && turnResult.exitCode !== 0;
2880
+ if (failed) {
2881
+ return {
2882
+ type: "turn_finished",
2883
+ cancelled: false,
2884
+ hasError,
2885
+ errorMessage: turnResult.error?.message,
2886
+ exitCode: turnResult.exitCode,
2887
+ lastStderr: turnResult.lastStderr,
2888
+ stderrTail: turnResult.stderrTail,
2889
+ streamMessage: turnResult.streamMessage,
2890
+ transportBroken: false,
2891
+ handoverRequestHandle: null,
2892
+ suspension: null,
2893
+ adapterSessionId,
2894
+ outcome: null,
2895
+ trackerContent: ""
2896
+ };
2026
2897
  }
2027
- if (trackerAbsPath) {
2028
- const outcome = resolveTurnOutcome({
2898
+ const transport = turnResult.diagnostics?.transport;
2899
+ const transportBroken = !!(transport && transport.streamToolUses > 0 && transport.preToolUseEvents === 0);
2900
+ let trackerContent = "";
2901
+ let outcome = null;
2902
+ if (!transportBroken && loop?.enabled && trackerAbsPath) {
2903
+ trackerContent = readTracker(trackerAbsPath);
2904
+ outcome = resolveTurnOutcome({
2029
2905
  trackerPath: trackerAbsPath,
2030
2906
  loop,
2031
- iteration: iterations
2907
+ iteration: memory.iteration
2908
+ });
2909
+ }
2910
+ return {
2911
+ type: "turn_finished",
2912
+ cancelled: false,
2913
+ hasError: false,
2914
+ exitCode: turnResult.exitCode,
2915
+ streamMessage: turnResult.streamMessage,
2916
+ transportBroken,
2917
+ handoverRequestHandle: null,
2918
+ suspension: null,
2919
+ adapterSessionId,
2920
+ outcome,
2921
+ trackerContent
2922
+ };
2923
+ }
2924
+ async function performForkTurn(handle, configOverride) {
2925
+ const handoffDir = handoffDirFor();
2926
+ const handoffAbsPath = nextHandoffPath(handoffDir);
2927
+ input.handover?.onForkStateChange?.(true);
2928
+ let forkOk = false;
2929
+ let transient = false;
2930
+ try {
2931
+ const forkResult = await input.startTurn({
2932
+ prompt: buildHandoffInvocationPrompt(handoffAbsPath),
2933
+ continuation: { mode: "resume", handle },
2934
+ configOverride: { ...configOverride, forkSession: true }
2032
2935
  });
2033
- if (outcome.kind === "stop") {
2034
- status = outcome.status;
2035
- stopReason = outcome.stopReason;
2036
- persist();
2037
- break;
2936
+ cumulativeTokens = mergeTokens(cumulativeTokens, forkResult.tokens);
2937
+ forkOk = !forkResult.error && (forkResult.exitCode === null || forkResult.exitCode === 0) && fs12.existsSync(handoffAbsPath);
2938
+ if (!forkOk) {
2939
+ transient = classifyTurnFailure({
2940
+ errorMessage: forkResult.error?.message,
2941
+ lastStderr: forkResult.stderrTail ?? forkResult.lastStderr,
2942
+ lastMessage: forkResult.streamMessage
2943
+ }).kind === "transient";
2038
2944
  }
2945
+ } catch (e) {
2946
+ forkOk = false;
2947
+ transient = classifyTurnFailure({
2948
+ errorMessage: e instanceof Error ? e.message : String(e)
2949
+ }).kind === "transient";
2950
+ } finally {
2951
+ input.handover?.onForkStateChange?.(false);
2039
2952
  }
2040
- persist();
2041
- input.onIterationComplete?.(snapshot());
2042
- nextContinuation = { mode: "fresh" };
2953
+ let handoffSizeBytes = null;
2954
+ if (forkOk) {
2955
+ try {
2956
+ handoffSizeBytes = fs12.statSync(handoffAbsPath).size;
2957
+ } catch {
2958
+ handoffSizeBytes = null;
2959
+ }
2960
+ }
2961
+ return {
2962
+ type: "fork_finished",
2963
+ ok: forkOk,
2964
+ cancelled,
2965
+ handoffPath: handoffAbsPath,
2966
+ handoffSizeBytes,
2967
+ transient
2968
+ };
2043
2969
  }
2044
- if (cancelled && status === "running") {
2045
- status = "cancelled";
2046
- persist();
2970
+ async function performWait(ms) {
2971
+ await delayWithCancel(ms, () => cancelled);
2972
+ if (cancelled) {
2973
+ return {
2974
+ type: "backoff_elapsed",
2975
+ cancelled: true,
2976
+ adapterSessionId: null
2977
+ };
2978
+ }
2979
+ const adapterSessionId = input.currentAdapterSessionId?.() ?? null;
2980
+ return { type: "backoff_elapsed", cancelled: false, adapterSessionId };
2981
+ }
2982
+ function isKickoffAction(action) {
2983
+ return action.type === "start_turn" || action.type === "start_fork_turn" || action.type === "wait";
2984
+ }
2985
+ async function runActions(actions) {
2986
+ let kickoff = null;
2987
+ for (const action of actions) {
2988
+ if (isKickoffAction(action)) {
2989
+ kickoff = action;
2990
+ continue;
2991
+ }
2992
+ switch (action.type) {
2993
+ case "persist":
2994
+ persist();
2995
+ if (trackerAbsPath && input.projectTasks) {
2996
+ try {
2997
+ const tasks = projectTrackerTasks(trackerAbsPath);
2998
+ if (tasks) input.projectTasks(tasks);
2999
+ } catch {
3000
+ }
3001
+ }
3002
+ break;
3003
+ case "notify_iteration_complete":
3004
+ input.onIterationComplete?.(snapshot());
3005
+ break;
3006
+ case "purge_handoffs":
3007
+ purgeHandoffs(handoffDirFor(), HANDOFF_RETAIN);
3008
+ break;
3009
+ case "degrade_handover":
3010
+ input.handover?.onDegraded?.(action.handle);
3011
+ break;
3012
+ }
3013
+ }
3014
+ if (!kickoff) return null;
3015
+ switch (kickoff.type) {
3016
+ case "start_turn":
3017
+ return performStartTurn(
3018
+ kickoff.prompt,
3019
+ kickoff.continuation,
3020
+ kickoff.configOverride
3021
+ );
3022
+ case "start_fork_turn":
3023
+ return performForkTurn(kickoff.handle, kickoff.configOverride);
3024
+ case "wait":
3025
+ return performWait(kickoff.ms);
3026
+ default:
3027
+ return null;
3028
+ }
3029
+ }
3030
+ let bootstrapActions;
3031
+ if (phase.kind === "awaiting_attention") {
3032
+ const wokenEvent = {
3033
+ type: "woken",
3034
+ continuation: input.initialContinuation ?? { mode: "fresh" }
3035
+ };
3036
+ const stepResult = step(phase, memory, wokenEvent, cfg);
3037
+ phase = stepResult.phase;
3038
+ memory = stepResult.memory;
3039
+ bootstrapActions = stepResult.actions;
3040
+ } else if (phase.kind === "turn_in_flight") {
3041
+ bootstrapActions = [
3042
+ {
3043
+ type: "start_turn",
3044
+ prompt: phase.prompt,
3045
+ continuation: phase.continuation,
3046
+ configOverride: phase.configOverride
3047
+ }
3048
+ ];
3049
+ } else if (phase.kind === "backing_off") {
3050
+ bootstrapActions = [{ type: "wait", ms: phase.ms }];
3051
+ } else {
3052
+ bootstrapActions = [];
3053
+ }
3054
+ let pendingEvent = await runActions(bootstrapActions);
3055
+ while (pendingEvent) {
3056
+ const stepResult = step(phase, memory, pendingEvent, cfg);
3057
+ phase = stepResult.phase;
3058
+ memory = stepResult.memory;
3059
+ if (isTerminalPhase(phase)) {
3060
+ const terminal = terminalPhaseToStatus(phase);
3061
+ status = terminal.status;
3062
+ stopReason = terminal.stopReason;
3063
+ await runActions(stepResult.actions);
3064
+ break;
3065
+ }
3066
+ pendingEvent = await runActions(stepResult.actions);
2047
3067
  }
2048
3068
  return {
2049
3069
  runId,
2050
3070
  status,
2051
- iterations,
3071
+ iterations: memory.iteration,
2052
3072
  stopReason,
2053
3073
  tokens: cumulativeTokens
2054
3074
  };
@@ -2171,12 +3191,12 @@ function useWorkflowSessionController(base, input) {
2171
3191
 
2172
3192
  // src/infra/plugins/mcpOptions.ts
2173
3193
  import fs13 from "fs";
2174
- import path10 from "path";
3194
+ import path11 from "path";
2175
3195
  function collectMcpServersWithOptions(pluginDirs) {
2176
3196
  const result = [];
2177
3197
  const seen = /* @__PURE__ */ new Set();
2178
3198
  for (const dir of pluginDirs) {
2179
- const mcpPath = path10.join(dir, ".mcp.json");
3199
+ const mcpPath = path11.join(dir, ".mcp.json");
2180
3200
  if (!fs13.existsSync(mcpPath)) {
2181
3201
  continue;
2182
3202
  }
@@ -2195,6 +3215,8 @@ function collectMcpServersWithOptions(pluginDirs) {
2195
3215
  }
2196
3216
 
2197
3217
  export {
3218
+ DEFAULT_MAX_TURN_TOKEN_COUNT,
3219
+ deserializeRunMemory,
2198
3220
  createWorkflowRunner,
2199
3221
  useWorkflowSessionController,
2200
3222
  isMarketplaceRef,
@@ -2226,4 +3248,4 @@ export {
2226
3248
  compileWorkflowPlan,
2227
3249
  collectMcpServersWithOptions
2228
3250
  };
2229
- //# sourceMappingURL=chunk-73V7GXV6.js.map
3251
+ //# sourceMappingURL=chunk-2QV7YSX6.js.map