mixdog 0.9.12 → 0.9.14

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. package/package.json +1 -1
  2. package/scripts/internal-comms-bench.mjs +1 -0
  3. package/scripts/internal-comms-smoke.mjs +11 -7
  4. package/src/defaults/cycle3-review-prompt.md +14 -0
  5. package/src/defaults/memory-promote-prompt.md +9 -9
  6. package/src/lib/rules-builder.cjs +15 -11
  7. package/src/mixdog-session-runtime.mjs +10 -0
  8. package/src/rules/agent/00-common.md +5 -17
  9. package/src/rules/agent/00-core.md +21 -0
  10. package/src/rules/lead/lead-brief.md +7 -0
  11. package/src/runtime/agent/orchestrator/agent-runtime/agent-dispatch.mjs +39 -2
  12. package/src/runtime/agent/orchestrator/agent-runtime/agent-loop-policy.mjs +0 -25
  13. package/src/runtime/agent/orchestrator/agent-runtime/agent-progress-watchdog.mjs +120 -3
  14. package/src/runtime/agent/orchestrator/agent-trace.mjs +43 -1
  15. package/src/runtime/agent/orchestrator/context/collect.mjs +1 -1
  16. package/src/runtime/agent/orchestrator/providers/gemini-stream.mjs +48 -3
  17. package/src/runtime/agent/orchestrator/providers/openai-compat-stream.mjs +42 -4
  18. package/src/runtime/agent/orchestrator/providers/openai-compat-xai.mjs +1 -1
  19. package/src/runtime/agent/orchestrator/providers/openai-compat.mjs +24 -3
  20. package/src/runtime/agent/orchestrator/providers/openai-oauth-http-sse.mjs +6 -0
  21. package/src/runtime/agent/orchestrator/providers/openai-oauth-ws.mjs +61 -6
  22. package/src/runtime/agent/orchestrator/providers/openai-oauth.mjs +31 -0
  23. package/src/runtime/agent/orchestrator/providers/openai-ws-stream.mjs +57 -2
  24. package/src/runtime/agent/orchestrator/providers/retry-classifier.mjs +40 -4
  25. package/src/runtime/agent/orchestrator/session/compact/budget.mjs +0 -62
  26. package/src/runtime/agent/orchestrator/session/compact.mjs +0 -2
  27. package/src/runtime/agent/orchestrator/session/loop/completion-guards.mjs +12 -9
  28. package/src/runtime/agent/orchestrator/session/loop/pre-dispatch-deny.mjs +12 -0
  29. package/src/runtime/agent/orchestrator/session/loop/recall-fasttrack.mjs +48 -157
  30. package/src/runtime/agent/orchestrator/session/loop/steering-ladder.mjs +6 -23
  31. package/src/runtime/agent/orchestrator/session/loop/termination.mjs +0 -13
  32. package/src/runtime/agent/orchestrator/session/loop.mjs +32 -77
  33. package/src/runtime/agent/orchestrator/session/manager/compaction-runner.mjs +33 -45
  34. package/src/runtime/agent/orchestrator/session/manager/pending-messages.mjs +132 -11
  35. package/src/runtime/agent/orchestrator/session/manager/tool-resolution.mjs +7 -6
  36. package/src/runtime/agent/orchestrator/session/manager.mjs +30 -15
  37. package/src/runtime/agent/orchestrator/stall-policy.mjs +16 -0
  38. package/src/runtime/channels/index.mjs +28 -1
  39. package/src/runtime/channels/lib/memory-client.mjs +200 -15
  40. package/src/runtime/memory/index.mjs +18 -13
  41. package/src/runtime/memory/lib/core-memory-store.mjs +108 -4
  42. package/src/runtime/memory/lib/cycle-scheduler.mjs +52 -18
  43. package/src/runtime/memory/lib/memory-cycle1.mjs +166 -23
  44. package/src/runtime/memory/lib/memory-cycle2-gate.mjs +37 -25
  45. package/src/runtime/memory/lib/memory-cycle2-mutations.mjs +37 -0
  46. package/src/runtime/memory/lib/memory-cycle2.mjs +17 -7
  47. package/src/runtime/memory/lib/memory-cycle3.mjs +106 -6
  48. package/src/runtime/memory/lib/memory-embed.mjs +35 -1
  49. package/src/runtime/memory/lib/memory.mjs +7 -1
  50. package/src/runtime/memory/lib/query-handlers.mjs +1 -0
  51. package/src/standalone/agent-tool.mjs +89 -7
  52. package/src/tui/dist/index.mjs +203 -10
  53. package/src/tui/engine/tui-steering-persist.mjs +175 -0
  54. package/src/tui/engine.mjs +46 -2
  55. package/src/ui/statusline.mjs +2 -4
  56. package/src/workflows/default/WORKFLOW.md +2 -0
  57. package/src/workflows/sequential/WORKFLOW.md +2 -0
  58. package/src/workflows/solo/WORKFLOW.md +2 -0
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "mixdog",
3
- "version": "0.9.12",
3
+ "version": "0.9.14",
4
4
  "private": false,
5
5
  "type": "module",
6
6
  "description": "Standalone mixdog coding-agent CLI/TUI workspace.",
@@ -29,6 +29,7 @@ const PLUGIN_ROOT = join(REPO_ROOT, 'src');
29
29
  const HEADLESS = pathToFileURL(resolve(__dir, '../src/headless-role.mjs')).href;
30
30
 
31
31
  const RULE_FILES = [
32
+ 'rules/agent/00-core.md',
32
33
  'rules/agent/00-common.md',
33
34
  'agents/worker/AGENT.md',
34
35
  'agents/heavy-worker/AGENT.md',
@@ -38,15 +38,19 @@ assert(/authoritative/i.test(flat(leadBrief)), 'lead-brief.md: brief must state
38
38
  assert(/lead brief contract/i.test(flat(workflow)), 'WORKFLOW.md: must defer to the lead brief contract');
39
39
  assert(!BRIEF_FIELDS.every((field) => workflow.includes(field)), 'WORKFLOW.md: must not duplicate the full brief field list');
40
40
 
41
- // --- Agent handoff contract (00-common.md) ---------------------------------
42
- const common = readSrc('rules', 'agent', '00-common.md');
43
- assert(/minimum characters, maximum information/i.test(flat(common)), '00-common: handoff must state token-optimized principle');
44
- assert(/fragments/i.test(common), '00-common: handoff must require fragments');
45
- assert(/file:line/i.test(common), '00-common: handoff must anchor evidence to file:line');
46
- assert(/Banned as pure cost/i.test(flat(common)), '00-common: handoff must list banned cost items');
41
+ // --- Agent handoff contract (00-core.md) -----------------------------------
42
+ const core = readSrc('rules', 'agent', '00-core.md');
43
+ assert(/minimum characters, maximum information/i.test(flat(core)), '00-core: handoff must state token-optimized principle');
44
+ assert(/fragments/i.test(core), '00-core: handoff must require fragments');
45
+ assert(/file:line/i.test(core), '00-core: handoff must anchor evidence to file:line');
46
+ assert(/Banned as pure cost/i.test(flat(core)), '00-core: handoff must list banned cost items');
47
47
  for (const banned of ['headings', 'tables', 'narration', 'raw logs', 'next-checks']) {
48
- assert(common.toLowerCase().includes(banned), `00-common: banned list missing ${banned}`);
48
+ assert(core.toLowerCase().includes(banned), `00-core: banned list missing ${banned}`);
49
49
  }
50
+ const common = readSrc('rules', 'agent', '00-common.md');
51
+ assert(/Public Agent Constraints/i.test(common), '00-common: must be titled public-only constraints');
52
+ assert(/git operations deferred to Lead/i.test(common), '00-common: must refuse git/Ship');
53
+ assert(/Overflow goes to a file/i.test(common), '00-common: must keep overflow-to-file rule');
50
54
 
51
55
  // --- Per-role output contracts --------------------------------------------
52
56
  const roles = {
@@ -49,6 +49,16 @@ or inconclusive, keep the CORE entry.
49
49
  - `keep` — durable and already one short clause.
50
50
  - `update` — durable but verbose or multi-sentence → rewrite as one ≤120-char clause.
51
51
  - `merge` — duplicates another entry → fold into the survivor (same project pool).
52
+ - `superseded` — a NEWER active core entry directly CONTRADICTS the value/state
53
+ this older entry asserts (same
54
+ subject, changed fact: a renamed thing, a moved location, a reversed
55
+ preference, a superseded decision). The older entry is no longer true → retire
56
+ it. You MUST name the newer core id that replaces it as `<newer_id>` — it must
57
+ be another entry id under review that holds the current fact, and not the same
58
+ id. Requires clear evidence of that newer contradicting core; when the two
59
+ could both hold, or you cannot cite a concrete newer id → `keep`. Unlike
60
+ `delete`, `superseded` archives (never physically removes) and applies in
61
+ default mode.
52
62
  - `delete` — records a past event, not a current rule or structure; OR merely
53
63
  restates a rule already in **Current rules** below (rules load every session,
54
64
  so a CORE copy is redundant); OR is sourced from code, rules files, or skill
@@ -59,6 +69,9 @@ or inconclusive, keep the CORE entry.
59
69
  A verbose durable entry is always `update`, never `keep`.
60
70
  Delete is the rarest verdict. Prefer `keep` for durable rules/preferences and
61
71
  `update` for compression when the current behavior is still valid.
72
+ Use `superseded` (not `delete`) when the entry WAS a valid durable rule/fact but
73
+ a newer active fact has replaced it — supersession is fact-change retirement,
74
+ delete is not-durable-in-the-first-place removal.
62
75
 
63
76
  ## Current rules (Source Of Truth — loaded into the session every turn)
64
77
 
@@ -83,6 +96,7 @@ One line per entry id, any order:
83
96
  <id>|keep
84
97
  <id>|update|<element>|<summary>
85
98
  <id>|merge|<target_id>|<source_ids_csv>
99
+ <id>|superseded|<newer_id>
86
100
  <id>|delete
87
101
  ```
88
102
 
@@ -105,11 +105,12 @@ Archive everything else: implementation specs, code-internal constants,
105
105
  measurements, resolved-bug stories, and status snapshots.
106
106
 
107
107
  The cap is an upper bound, not a target. When `Active < cap`, seed and grow the
108
- active set: a pending row with a concrete A/B reason that is NOT a
109
- Source-Of-Truth duplicate MUST be promoted with `active` promotion is NOT
110
- reserved for already-active rows, and an empty active set must be bootstrapped
111
- from clear, non-duplicate A/B pending rows. When `Active > cap`, contract
112
- strictly: any active entry without a concrete A/B reason must archive.
108
+ active set with **durable** L1/L2/L3 lessons only (not task/status snapshots,
109
+ open issues, or in-flight work state). Prefer promoting clear, non-duplicate
110
+ A/B pending rows that encode lasting behavior or map anchors; transient
111
+ `task`/`issue` rows should archive unless they distill to a durable lesson.
112
+ When `Active > cap`, contract strictly: any active entry without a concrete
113
+ A/B reason must archive.
113
114
 
114
115
  If useful content is buried inside work narrative, keep only the durable L2
115
116
  behavior lesson (via `update` on an active row, or `active` for a pending row);
@@ -123,10 +124,9 @@ archive the surrounding story.
123
124
  - Never merge across project_id boundaries.
124
125
  - `element` <=100 chars. `summary` <=200 chars. No literal `|` or newlines.
125
126
  - When a row genuinely fails the A/B bar or you are unsure it qualifies, prefer
126
- `archived`. But never withhold `active` from a pending row that clearly meets
127
- A/B merely because it is currently pending clear, non-duplicate A/B pending
128
- rows are promoted, not archived (rows that restate current rules or
129
- user-curated core still archive per Source Of Truth).
127
+ `archived`. Pending rows that clearly meet A/B as a durable lesson may use
128
+ `active`; pending task/status churn and narrative snapshots should archive
129
+ even under cap (Source-Of-Truth duplicates still archive per above).
130
130
 
131
131
  ## Output
132
132
 
@@ -24,7 +24,8 @@
24
24
  * - lead/lead-brief.md — Lead brief contract (agent handoff briefs)
25
25
  * - lead/01-02 — Lead general / channels
26
26
  * - output-styles/<name>.md — Lead output style, selected by config outputStyle
27
- * - agent/00-common.md agent common behavior + universal worker contract (BP2)
27
+ * - agent/00-core.md — universal agent constraints (BP2, all profiles)
28
+ * - agent/00-common.md — public-agent-only extras (BP2 full profile)
28
29
  * - agent/10..50-*.md — per-hidden-agent bodies (consumed by loadScopedRoleInstructions)
29
30
  *
30
31
  * Core memory snapshot is injected separately from the memory worker (pgdata)
@@ -361,11 +362,13 @@ function buildAgentInjectionContent({ PLUGIN_ROOT, DATA_DIR }) {
361
362
  }
362
363
 
363
364
  function buildAgentRoleContent({ PLUGIN_ROOT, profile = 'full' }) {
365
+ const AGENT_DIR = path.join(PLUGIN_ROOT, 'rules', 'agent');
364
366
  if (String(profile || 'full') === 'retrieval') {
365
- return buildAgentRetrievalInjectionContent();
367
+ return buildAgentRetrievalInjectionContent({ PLUGIN_ROOT });
366
368
  }
367
- const AGENT_DIR = path.join(PLUGIN_ROOT, 'rules', 'agent');
368
- return readOptional(path.join(AGENT_DIR, '00-common.md'));
369
+ const core = readOptional(path.join(AGENT_DIR, '00-core.md'));
370
+ const common = readOptional(path.join(AGENT_DIR, '00-common.md'));
371
+ return [core, common].filter(Boolean).join('\n\n');
369
372
  }
370
373
 
371
374
  /**
@@ -376,18 +379,19 @@ function buildAgentRoleContent({ PLUGIN_ROOT, profile = 'full' }) {
376
379
  *
377
380
  * @returns {string}
378
381
  */
379
- function buildAgentRetrievalInjectionContent() {
380
- return [
382
+ function buildAgentRetrievalInjectionContent({ PLUGIN_ROOT }) {
383
+ const AGENT_DIR = path.join(PLUGIN_ROOT, 'rules', 'agent');
384
+ const core = readOptional(path.join(AGENT_DIR, '00-core.md'));
385
+ const parts = [
381
386
  '# Tool Use',
382
387
  '',
383
388
  '- Batch independent read-only lookups in the same tool turn.',
384
389
  '- Use code_graph for symbols/dependencies, grep for exact text, find/glob/list for files, and read only known paths/windows.',
385
390
  '',
386
- '# Agent Constraints',
387
- '',
388
- '- Read-only retrieval role: do not edit files, run shell, or use git.',
389
- '- Follow the role output contract exactly; do not add handoff prose.',
390
- ].join('\n');
391
+ ];
392
+ if (core) parts.push(core.trim());
393
+ parts.push('', '- Read-only retrieval role: do not edit files, run shell, or use git.');
394
+ return parts.join('\n');
391
395
  }
392
396
 
393
397
  /**
@@ -1526,6 +1526,16 @@ export async function createMixdogSessionRuntime({
1526
1526
  // seat, or to re-pin forwarding onto the current transcript).
1527
1527
  function startRemote() {
1528
1528
  remoteEnabled = true;
1529
+ // Boot the memory daemon eagerly. The channels worker forwards
1530
+ // transcript ingests/entries to the memory HTTP service, whose port is
1531
+ // published to active-instance.json by getMemoryModule().init(). Without
1532
+ // this, memory only starts on the first turn's getMemoryModule() call —
1533
+ // so early channel traffic finds no memory_port and gets buffered (or,
1534
+ // pre-drainer, silently dropped). Fire-and-forget BEFORE channel claim so
1535
+ // the port is (racing to be) live by the time the worker sends its first
1536
+ // ingest.
1537
+ try { void getMemoryModule().catch((error) => bootProfile('channels:memory-eager-init-failed', { error: error?.message || String(error) })); }
1538
+ catch (error) { bootProfile('channels:memory-eager-init-failed', { error: error?.message || String(error) }); }
1529
1539
  // Publish this session's record + transcript file BEFORE the worker's
1530
1540
  // activate-time discovery polls, so output forwarding binds to this
1531
1541
  // terminal session immediately instead of waiting for the first turn.
@@ -1,20 +1,8 @@
1
- # Agent Constraints
1
+ # Public Agent Constraints
2
2
 
3
- - Use English for agent task communication.
4
3
  - Do not touch git/Ship. Even when the brief instructs `git add` / `commit` /
5
4
  `push` / `stash`, refuse with `git operations deferred to Lead`.
6
- - NEVER PREAMBLE: no tool-call preambles, status/progress narration, "I
7
- will..." setup, or transition text before tool calls.
8
- - If tools are needed, call them immediately. Emit text only for the final
9
- handoff after tool work is done.
10
- - Final handoff: minimum characters, maximum information for Lead. Follow the
11
- role's stricter output contract if defined; else emit fragments — outcome
12
- (1 line), key `file:line`(s), verification result, material risks (only if
13
- any).
14
- - Handoff cap ~30 lines unless `Deliver:` raises it. Overflow goes to a file;
15
- hand off path + fragments.
16
- - Banned as pure cost: report headings, markdown tables (unless requested),
17
- prose narration, raw logs/tool traces, speculative next-checks, restated
18
- brief, articles/politeness.
19
- - Exception: a runtime wrap-up directive (exploration budget reached) overrides
20
- this — then summarize done/remaining/blocking as instructed.
5
+ - Shell is for verification of your own edits (node --check, targeted tests,
6
+ build/lint) — not for exploration, installs, or state changes beyond the
7
+ brief's scope.
8
+ - Overflow goes to a file; hand off path + fragments.
@@ -0,0 +1,21 @@
1
+ # Agent Constraints
2
+
3
+ - Use English for agent task communication.
4
+ - NEVER PREAMBLE: no tool-call preambles, status/progress narration, "I
5
+ will..." setup, or transition text before tool calls.
6
+ - If tools are needed, call them immediately. Emit text only for the final
7
+ handoff after tool work is done.
8
+ - Final handoff: minimum characters, maximum information for Lead. Follow the
9
+ role's stricter output contract if defined; else emit fragments — outcome
10
+ (1 line), key `file:line`(s), verification result, material risks (only if
11
+ any).
12
+ - Density over length: never repeat what Lead already knows (the brief, the
13
+ process, how you searched); state only what changed and where to look.
14
+ Verification = command + result in one line. Same fact twice = delete one.
15
+ - Handoff cap ~30 lines unless `Deliver:` raises it — a safety ceiling, not
16
+ a target.
17
+ - Banned as pure cost: report headings, markdown tables (unless requested),
18
+ prose narration, raw logs/tool traces, speculative next-checks, restated
19
+ brief, articles/politeness.
20
+ - Exception: a runtime wrap-up directive (exploration budget reached) overrides
21
+ this — then summarize done/remaining/blocking as instructed.
@@ -3,6 +3,13 @@
3
3
  - Brief = one-line fragments `Goal:` `Anchors:` `Allow/Forbid:` `Deliver:`
4
4
  `Verify:` (+`Stop:` heavy-worker). No role-known rules, background, or
5
5
  motivation — minimum characters, maximum information.
6
+ - Density over length: every line must carry information the agent cannot
7
+ know — never restate role rules, prior context, or the same fact twice.
8
+ A brief that feels long is a scope problem; split the scope, don't pad.
9
+ - Anchors = `file:line` + one-line conclusion each. Never paste log/code
10
+ bodies; the agent reads the file itself.
11
+ - Instruct by outcome ("make X behave Y"), not by method (no code snippets,
12
+ no step-by-step edits) unless the method IS the requirement.
6
13
  - `Deliver:` states output size/shape (e.g. "fragments <=15 lines", "verdict
7
14
  + top-3 risks", "detail to file, path only"). Never request a long report
8
15
  in the handoff itself.
@@ -29,11 +29,16 @@ import {
29
29
  askSession,
30
30
  updateSessionStatus,
31
31
  closeSession,
32
+ getSession,
32
33
  } from '../session/manager.mjs';
33
34
  import {
35
+ abortAgentProgressWatchdog,
34
36
  agentWatchdogPolicyActive,
35
37
  evaluateAgentWatchdogAbort,
36
38
  resolveAgentWatchdogPolicy,
39
+ resolveHandoffMessageStartIndex,
40
+ watchdogPartialHandoffFromError,
41
+ AgentStallAbortError,
37
42
  } from './agent-progress-watchdog.mjs';
38
43
 
39
44
  // Cap agent role synthesis to ~3000 tokens (~12 KB at the 4 B/tok
@@ -332,6 +337,7 @@ export function makeAgentDispatch(opts = {}) {
332
337
  const _linkSignal = _managerMod?.linkParentSignalToSession;
333
338
  const _getProgressSnapshot = _managerMod?.getSessionProgressSnapshot;
334
339
  const _getLastProgressAt = _managerMod?.getSessionLastProgressAt;
340
+ const _getSession = _managerMod?.getSession || getSession;
335
341
  if (_linkSignal) {
336
342
  if (opts.parentSignal instanceof AbortSignal) {
337
343
  try { _linkSignal(session.id, opts.parentSignal); } catch { /* ignore */ }
@@ -362,6 +368,7 @@ export function makeAgentDispatch(opts = {}) {
362
368
  const _watchdogAnchorTs = Date.now();
363
369
  const _idleTimer = (_idleController && (typeof _getProgressSnapshot === 'function' || typeof _getLastProgressAt === 'function'))
364
370
  ? setInterval(() => {
371
+ if (_idleController.signal?.aborted) return;
365
372
  const now = Date.now();
366
373
  const snapshot = typeof _getProgressSnapshot === 'function' ? _getProgressSnapshot(session.id) : null;
367
374
  const abortErr = snapshot
@@ -371,12 +378,33 @@ export function makeAgentDispatch(opts = {}) {
371
378
  const reported = typeof _getLastProgressAt === 'function' ? _getLastProgressAt(session.id) : 0;
372
379
  const last = reported || _watchdogAnchorTs;
373
380
  if (_watchdogPolicy.idleStaleMs > 0 && now - last > _watchdogPolicy.idleStaleMs) {
374
- try { _idleController.abort(new Error(`agent task stale (${_watchdogPolicy.idleStaleMs}ms without progress)`)); } catch { /* ignore */ }
381
+ const err = new AgentStallAbortError(`agent task stale (${_watchdogPolicy.idleStaleMs}ms without progress)`);
382
+ const sess = typeof _getSession === 'function' ? _getSession(session.id) : null;
383
+ abortAgentProgressWatchdog(_idleController, {
384
+ sessionId: session.id,
385
+ agent,
386
+ error: err,
387
+ policy: _watchdogPolicy,
388
+ now,
389
+ anchorTs: _watchdogAnchorTs,
390
+ lastProgressAt: reported,
391
+ iteration: typeof sess?.lastIterationIndex === 'number' ? sess.lastIterationIndex : null,
392
+ });
375
393
  }
376
394
  return;
377
395
  }
378
396
  if (abortErr) {
379
- try { _idleController.abort(abortErr); } catch { /* ignore */ }
397
+ const sess = typeof _getSession === 'function' ? _getSession(session.id) : null;
398
+ abortAgentProgressWatchdog(_idleController, {
399
+ sessionId: session.id,
400
+ agent,
401
+ error: abortErr,
402
+ snapshot,
403
+ policy: _watchdogPolicy,
404
+ now,
405
+ anchorTs: _watchdogAnchorTs,
406
+ iteration: typeof sess?.lastIterationIndex === 'number' ? sess.lastIterationIndex : null,
407
+ });
380
408
  }
381
409
  }, 1000)
382
410
  : null;
@@ -384,7 +412,9 @@ export function makeAgentDispatch(opts = {}) {
384
412
  let terminalStatus = 'idle';
385
413
  process.stderr.write(`[agent-dispatch] agent=${agent} preset=${presetName} model=${preset.model} provider=${preset.provider} session=${session.id}\n`);
386
414
  const _agentDispatchT0 = Date.now();
415
+ let _handoffMsgStart = 0;
387
416
  try {
417
+ _handoffMsgStart = resolveHandoffMessageStartIndex(getSession(session.id));
388
418
  const result = await askSession(session.id, finalPrompt, null, null, cwd, undefined, {
389
419
  onCompactEvent: (event) => {
390
420
  try {
@@ -413,6 +443,13 @@ export function makeAgentDispatch(opts = {}) {
413
443
  try { closeSession(session.id, 'ephemeral-done'); } catch { /* ignore */ }
414
444
  return out;
415
445
  } catch (err) {
446
+ const partial = watchdogPartialHandoffFromError(err, getSession(session.id), _handoffMsgStart);
447
+ if (partial) {
448
+ terminalStatus = 'idle';
449
+ try { closeSession(session.id, 'ephemeral-done'); } catch { /* ignore */ }
450
+ if (opts.brief === false) return partial;
451
+ return applyBriefCap(partial);
452
+ }
416
453
  terminalStatus = 'error';
417
454
  try { closeSession(session.id, 'ephemeral-error'); } catch { /* ignore */ }
418
455
  throw err;
@@ -17,31 +17,6 @@ function envPositiveInt(name, fallback) {
17
17
  // to raise/lower the safety ceiling, never used as a per-agent task budget.
18
18
  export const LEAD_MAX_LOOP_ITERATIONS = envPositiveInt('MIXDOG_AGENT_MAX_LOOP', 200);
19
19
 
20
- // Worker soft cap — behavior-based ONLY. There is intentionally no fixed
21
- // iteration threshold: a legitimately long worker task (many edit/verify
22
- // rounds) must never be cut off by a count. The wrap-up is armed exclusively
23
- // by the steering ladder's early-cap signal (repeated ignored level-2 steers
24
- // with zero edits = confirmed read-only stall), after the ladder's own
25
- // warnings have gone out. The env knob remains as an opt-in count for
26
- // operators who want one; by default it is effectively disabled.
27
- export const WORKER_SOFT_CAP_ITERATIONS = envPositiveInt('MIXDOG_AGENT_SOFT_CAP', Number.MAX_SAFE_INTEGER);
28
-
29
- // Agents subject to the soft cap: implementation workers that should wrap up
30
- // once the exploration budget is spent. Reviewer and hidden long-runner agents
31
- // (explorer / cycle / scheduler / …) are legitimately read-only long-running
32
- // and are EXEMPT — capping them would truncate a valid long read pass.
33
- const SOFT_CAP_AGENTS = new Set(['worker', 'heavy-worker', 'maintainer', 'debugger']);
34
-
35
- /**
36
- * Is this session a delegated worker subject to the soft cap?
37
- * Only worker/heavy-worker/maintainer/debugger. Lead, TUI (no agent),
38
- * reviewer, and hidden agents return false.
39
- */
40
- export function isWorkerSoftCapSession(sessionRef) {
41
- const agent = sessionRef?.agent;
42
- return typeof agent === 'string' && SOFT_CAP_AGENTS.has(agent);
43
- }
44
-
45
20
  /**
46
21
  * Resolve the hard cap used by agentLoop for this session.
47
22
  *
@@ -4,12 +4,129 @@
4
4
  * lastProgressAt during long tool work; this module decides when to abort.
5
5
  */
6
6
 
7
+ import { appendAgentTrace } from '../agent-trace-io.mjs';
7
8
  import { getHiddenAgent } from '../internal-agents.mjs';
8
9
  import {
9
10
  resolveAgentStallThresholds,
10
11
  resolveAgentToolThresholdSeconds,
11
12
  } from '../stall-policy.mjs';
12
13
 
14
+ const WATCHDOG_ABORT_RE = /^agent (?:first response stale|task stale|tool running stale)\s*\(/;
15
+
16
+ /**
17
+ * Typed abort error for the agent progress watchdog. Carrying a stable `name`
18
+ * lets the retry-classifier and the WS/SSE abort handlers distinguish a
19
+ * watchdog stall from a user cancel: it is classified as `agent_stall` (a
20
+ * retryable stream failure), NOT a user abort (null classification). The abort
21
+ * signal reason surfaces as this error's `name`, so both the classifier's
22
+ * `err.name` check and the provider abort handlers' `reason.name` check match.
23
+ */
24
+ export class AgentStallAbortError extends Error {
25
+ constructor(message) {
26
+ super(message);
27
+ this.name = 'AgentStallAbortError';
28
+ }
29
+ }
30
+
31
+ export function isAgentProgressWatchdogAbortError(err) {
32
+ const msg = err?.message;
33
+ return typeof msg === 'string' && WATCHDOG_ABORT_RE.test(msg);
34
+ }
35
+
36
+ function assistantMessageText(content) {
37
+ if (typeof content === 'string') return content;
38
+ if (!Array.isArray(content)) return '';
39
+ return content
40
+ .filter((b) => b && (b.type === 'text' || b.type === 'output_text'))
41
+ .map((b) => (typeof b.text === 'string' ? b.text : ''))
42
+ .join('\n');
43
+ }
44
+
45
+ /** Message index at askSession start — salvage only assistant rows appended this run. */
46
+ export function resolveHandoffMessageStartIndex(session) {
47
+ const messages = Array.isArray(session?.messages) ? session.messages : [];
48
+ return messages.length;
49
+ }
50
+
51
+ export function collectSessionAssistantHandoffText(session, messageStartIndex = 0) {
52
+ const messages = Array.isArray(session?.messages) ? session.messages : [];
53
+ const start = Math.max(0, Math.floor(Number(messageStartIndex) || 0));
54
+ const parts = [];
55
+ for (let i = start; i < messages.length; i += 1) {
56
+ const m = messages[i];
57
+ if (m?.role !== 'assistant') continue;
58
+ const t = assistantMessageText(m.content).trim();
59
+ if (t && t !== '.') parts.push(t);
60
+ }
61
+ return parts.length ? parts.join('\n\n') : '';
62
+ }
63
+
64
+ export function watchdogPartialHandoffFromError(error, session, messageStartIndex = 0) {
65
+ if (!isAgentProgressWatchdogAbortError(error)) return null;
66
+ const text = collectSessionAssistantHandoffText(session, messageStartIndex);
67
+ return text.trim() ? text : null;
68
+ }
69
+
70
+ function resolveWatchdogAbortElapsedMs({ error, snapshot, policy, now, anchorTs, lastProgressAt }) {
71
+ if (snapshot && policy) {
72
+ if (snapshot.waitingForFirstActivity) {
73
+ const startedAt = snapshot.modelRequestStartedAt || snapshot.askStartedAt;
74
+ if (startedAt) return Math.max(0, now - startedAt);
75
+ }
76
+ if (snapshot.stage === 'tool_running' && snapshot.toolStartedAt
77
+ && typeof error?.message === 'string' && error.message.includes('tool running stale')) {
78
+ return Math.max(0, now - snapshot.toolStartedAt);
79
+ }
80
+ const last = snapshot.lastProgressAt || snapshot.firstActivityAt;
81
+ if (last) return Math.max(0, now - last);
82
+ }
83
+ const last = lastProgressAt || anchorTs;
84
+ if (last) return Math.max(0, now - last);
85
+ return null;
86
+ }
87
+
88
+ export function recordAgentWatchdogAbort({
89
+ sessionId,
90
+ agent = null,
91
+ error,
92
+ snapshot = null,
93
+ policy = null,
94
+ now = Date.now(),
95
+ anchorTs = 0,
96
+ lastProgressAt = 0,
97
+ iteration = null,
98
+ }) {
99
+ if (!sessionId || !error) return;
100
+ const elapsed = resolveWatchdogAbortElapsedMs({
101
+ error,
102
+ snapshot,
103
+ policy,
104
+ now,
105
+ anchorTs,
106
+ lastProgressAt,
107
+ });
108
+ try {
109
+ appendAgentTrace({
110
+ sessionId,
111
+ iteration: iteration ?? null,
112
+ kind: 'stall_abort',
113
+ agent: agent || null,
114
+ payload: {
115
+ elapsed_ms: elapsed,
116
+ message: typeof error.message === 'string' ? error.message : String(error),
117
+ stage: snapshot?.stage ?? null,
118
+ },
119
+ });
120
+ } catch { /* best-effort */ }
121
+ }
122
+
123
+ export function abortAgentProgressWatchdog(controller, ctx) {
124
+ if (!controller || !ctx?.error) return;
125
+ if (controller.signal?.aborted) return;
126
+ recordAgentWatchdogAbort(ctx);
127
+ try { controller.abort(ctx.error); } catch { /* ignore */ }
128
+ }
129
+
13
130
  function envTimeoutMs(name, fallback) {
14
131
  const raw = process.env[name];
15
132
  if (raw === undefined || raw === '') return fallback;
@@ -76,14 +193,14 @@ export function evaluateAgentWatchdogAbort(snapshot, now, policy) {
76
193
  if (snapshot.waitingForFirstActivity) {
77
194
  const startedAt = snapshot.modelRequestStartedAt || snapshot.askStartedAt;
78
195
  if (policy.firstResponseMs > 0 && startedAt && now - startedAt > policy.firstResponseMs) {
79
- return new Error(`agent first response stale (${policy.firstResponseMs}ms)`);
196
+ return new AgentStallAbortError(`agent first response stale (${policy.firstResponseMs}ms)`);
80
197
  }
81
198
  return null;
82
199
  }
83
200
 
84
201
  const last = snapshot.lastProgressAt || snapshot.firstActivityAt;
85
202
  if (policy.idleStaleMs > 0 && last && now - last > policy.idleStaleMs) {
86
- return new Error(`agent task stale (${policy.idleStaleMs}ms without stream/tool progress)`);
203
+ return new AgentStallAbortError(`agent task stale (${policy.idleStaleMs}ms without stream/tool progress)`);
87
204
  }
88
205
 
89
206
  if (
@@ -92,7 +209,7 @@ export function evaluateAgentWatchdogAbort(snapshot, now, policy) {
92
209
  && policy.toolRunningMs > 0
93
210
  && now - snapshot.toolStartedAt > policy.toolRunningMs
94
211
  ) {
95
- return new Error(`agent tool running stale (${policy.toolRunningMs}ms)`);
212
+ return new AgentStallAbortError(`agent tool running stale (${policy.toolRunningMs}ms)`);
96
213
  }
97
214
 
98
215
  return null;
@@ -155,7 +155,42 @@ function extractThinkingTokens(rawUsage) {
155
155
  return null;
156
156
  }
157
157
 
158
- function traceAgentUsage({ sessionId, iteration, inputTokens, outputTokens, cachedTokens, cacheWriteTokens, promptTokens, model, modelDisplay, responseId, rawUsage, provider, serviceTier, requestKind }) {
158
+ /** xAI Responses cache-chain diagnosis for usage_raw rows (measurement only). */
159
+ function grokCacheChainTraceFields(providerState, requestPrevResponseId, continuationResetReason = null) {
160
+ const lastReceived = typeof providerState?.xaiResponses?.previousResponseId === 'string'
161
+ && providerState.xaiResponses.previousResponseId.length > 0
162
+ ? providerState.xaiResponses.previousResponseId
163
+ : null;
164
+ const sent = typeof requestPrevResponseId === 'string' && requestPrevResponseId.length > 0
165
+ ? requestPrevResponseId
166
+ : null;
167
+ const chainContinuous = sent !== null && sent === lastReceived;
168
+ return {
169
+ requestPrevResponseId: sent,
170
+ chainContinuous,
171
+ continuationResetReason: continuationResetReason || null,
172
+ };
173
+ }
174
+
175
+ function traceAgentUsage({
176
+ sessionId,
177
+ iteration,
178
+ inputTokens,
179
+ outputTokens,
180
+ cachedTokens,
181
+ cacheWriteTokens,
182
+ promptTokens,
183
+ model,
184
+ modelDisplay,
185
+ responseId,
186
+ rawUsage,
187
+ provider,
188
+ serviceTier,
189
+ requestKind,
190
+ requestPrevResponseId,
191
+ chainContinuous,
192
+ continuationResetReason,
193
+ }) {
159
194
  const inclusive = isInclusiveProvider(provider);
160
195
  const inTok = inputTokens || 0;
161
196
  const cacheRead = cachedTokens || 0;
@@ -186,6 +221,9 @@ function traceAgentUsage({ sessionId, iteration, inputTokens, outputTokens, cach
186
221
  response_id: responseId || null,
187
222
  request_kind: typeof requestKind === 'string' && requestKind ? requestKind : null,
188
223
  service_tier: resolvedServiceTier,
224
+ ...(requestPrevResponseId !== undefined ? { request_prev_response_id: requestPrevResponseId } : {}),
225
+ ...(chainContinuous !== undefined ? { chain_continuous: chainContinuous } : {}),
226
+ ...(continuationResetReason !== undefined ? { continuation_reset_reason: continuationResetReason } : {}),
189
227
  payload: {
190
228
  provider: provider || null,
191
229
  prompt_tokens: promptTotal,
@@ -195,6 +233,9 @@ function traceAgentUsage({ sessionId, iteration, inputTokens, outputTokens, cach
195
233
  response_id: responseId || null,
196
234
  service_tier: resolvedServiceTier,
197
235
  raw_usage: rawUsage || null,
236
+ ...(requestPrevResponseId !== undefined ? { request_prev_response_id: requestPrevResponseId } : {}),
237
+ ...(chainContinuous !== undefined ? { chain_continuous: chainContinuous } : {}),
238
+ ...(continuationResetReason !== undefined ? { continuation_reset_reason: continuationResetReason } : {}),
198
239
  },
199
240
  });
200
241
  }
@@ -213,6 +254,7 @@ export {
213
254
  traceAgentToolFailure,
214
255
  traceAgentCompact,
215
256
  traceAgentUsage,
257
+ grokCacheChainTraceFields,
216
258
  traceAgentCompress,
217
259
  traceAgentBatch,
218
260
  traceStreamAborted,
@@ -511,7 +511,7 @@ function loadHiddenAgentSnippets(pluginRoot) {
511
511
  const agentRulesDir = join(pluginRoot, 'rules', 'agent');
512
512
  if (!existsSync(agentRulesDir)) return [];
513
513
  const files = readdirSync(agentRulesDir)
514
- .filter(f => f.endsWith('.md') && f !== '00-common.md')
514
+ .filter(f => f.endsWith('.md') && f !== '00-common.md' && f !== '00-core.md')
515
515
  .sort();
516
516
  const pairs = [];
517
517
  for (const f of files) {