mixdog 0.9.24 → 0.9.26

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (84) hide show
  1. package/package.json +2 -1
  2. package/scripts/build-runtime-windows.ps1 +242 -242
  3. package/scripts/explore-bench-tmp.mjs +1 -1
  4. package/scripts/recall-usecase-cases.json +1 -1
  5. package/scripts/smoke-runtime-negative.ps1 +106 -106
  6. package/scripts/steering-fold-provenance-test.mjs +71 -0
  7. package/scripts/tool-efficiency-diag.mjs +1 -1
  8. package/scripts/webhook-smoke.mjs +46 -53
  9. package/src/defaults/cycle3-review-prompt.md +11 -4
  10. package/src/defaults/memory-promote-prompt.md +9 -0
  11. package/src/defaults/skills/setup/SKILL.md +6 -1
  12. package/src/rules/lead/02-channels.md +3 -3
  13. package/src/rules/shared/01-tool.md +8 -4
  14. package/src/runtime/agent/orchestrator/agent-trace-io.mjs +17 -0
  15. package/src/runtime/agent/orchestrator/mcp/child-tree.mjs +2 -2
  16. package/src/runtime/agent/orchestrator/mcp/client.mjs +28 -10
  17. package/src/runtime/agent/orchestrator/providers/anthropic-oauth.mjs +8 -1
  18. package/src/runtime/agent/orchestrator/providers/anthropic.mjs +14 -1
  19. package/src/runtime/agent/orchestrator/providers/oauth-usage.mjs +51 -15
  20. package/src/runtime/agent/orchestrator/session/agent-loop.mjs +5 -1
  21. package/src/runtime/agent/orchestrator/session/loop/steering-ladder.mjs +164 -9
  22. package/src/runtime/agent/orchestrator/session/loop/stored-tool-args.mjs +45 -0
  23. package/src/runtime/agent/orchestrator/session/manager/compaction-runner.mjs +4 -0
  24. package/src/runtime/agent/orchestrator/session/tool-batch.mjs +12 -1
  25. package/src/runtime/agent/orchestrator/tools/builtin/shell-jobs.mjs +2 -2
  26. package/src/runtime/agent/orchestrator/tools/shell-state.mjs +7 -3
  27. package/src/runtime/channels/backends/discord.mjs +10 -3
  28. package/src/runtime/channels/lib/config.mjs +13 -11
  29. package/src/runtime/channels/lib/inbound-handler.mjs +45 -1
  30. package/src/runtime/channels/lib/memory-client.mjs +22 -2
  31. package/src/runtime/channels/lib/owned-runtime.mjs +1 -1
  32. package/src/runtime/channels/lib/scheduler.mjs +226 -208
  33. package/src/runtime/channels/lib/status-snapshot.mjs +16 -0
  34. package/src/runtime/channels/lib/webhook/deliveries.mjs +7 -318
  35. package/src/runtime/channels/lib/webhook.mjs +98 -150
  36. package/src/runtime/channels/lib/worker-main.mjs +1 -1
  37. package/src/runtime/memory/index.mjs +50 -25
  38. package/src/runtime/memory/lib/cycle-scheduler.mjs +20 -2
  39. package/src/runtime/memory/lib/memory-config-flags.mjs +13 -1
  40. package/src/runtime/memory/lib/memory-cycle2-mutations.mjs +59 -0
  41. package/src/runtime/memory/lib/memory-cycle2.mjs +10 -3
  42. package/src/runtime/memory/lib/memory-cycle3.mjs +79 -9
  43. package/src/runtime/memory/lib/pg/adapter.mjs +6 -1
  44. package/src/runtime/memory/lib/pg/process.mjs +19 -1
  45. package/src/runtime/memory/lib/pg/supervisor.mjs +125 -14
  46. package/src/runtime/memory/lib/query-handlers.mjs +102 -0
  47. package/src/runtime/memory/lib/recall-format.mjs +60 -22
  48. package/src/runtime/memory/lib/session-ingest-runtime.mjs +20 -6
  49. package/src/runtime/memory/tool-defs.mjs +1 -1
  50. package/src/runtime/shared/child-guardian.mjs +2 -2
  51. package/src/runtime/shared/open-url.mjs +2 -1
  52. package/src/runtime/shared/schedules-db.mjs +350 -0
  53. package/src/runtime/shared/service-discovery.mjs +169 -0
  54. package/src/runtime/shared/spawn-flags.mjs +27 -0
  55. package/src/runtime/shared/tool-primitives.mjs +3 -1
  56. package/src/runtime/shared/tool-surface.mjs +19 -3
  57. package/src/runtime/shared/update-checker.mjs +54 -10
  58. package/src/runtime/shared/webhooks-db.mjs +405 -0
  59. package/src/session-runtime/channel-config-api.mjs +13 -13
  60. package/src/session-runtime/lifecycle-api.mjs +14 -0
  61. package/src/session-runtime/runtime-core.mjs +39 -20
  62. package/src/standalone/agent-tool.mjs +42 -11
  63. package/src/standalone/channel-admin.mjs +173 -121
  64. package/src/standalone/channel-worker.mjs +2 -2
  65. package/src/standalone/memory-runtime-proxy.mjs +10 -5
  66. package/src/tui/App.jsx +32 -10
  67. package/src/tui/app/channel-pickers.mjs +9 -9
  68. package/src/tui/app/clipboard.mjs +14 -0
  69. package/src/tui/app/doctor.mjs +1 -1
  70. package/src/tui/app/maintenance-pickers.mjs +17 -7
  71. package/src/tui/app/settings-picker.mjs +2 -2
  72. package/src/tui/app/transcript-window.mjs +37 -1
  73. package/src/tui/app/use-mouse-input.mjs +19 -6
  74. package/src/tui/components/ToolExecution.jsx +12 -4
  75. package/src/tui/display-width.mjs +10 -4
  76. package/src/tui/dist/index.mjs +166 -58
  77. package/src/tui/engine/agent-job-feed.mjs +42 -3
  78. package/src/tui/engine/session-api-ext.mjs +12 -12
  79. package/src/tui/index.jsx +12 -2
  80. package/src/ui/statusline-segments.mjs +12 -1
  81. package/src/workflows/default/WORKFLOW.md +29 -38
  82. package/src/workflows/solo/WORKFLOW.md +15 -20
  83. package/vendor/ink/build/display-width.js +5 -4
  84. package/src/runtime/shared/schedules-store.mjs +0 -82
@@ -5,6 +5,7 @@ import { tmpdir } from 'os';
5
5
  import { randomUUID } from 'crypto';
6
6
  import { smartReadTruncate } from '../tools/builtin/read-formatting.mjs';
7
7
  import { shutdownStdioChild, killStdioChildTreeFast } from './child-tree.mjs';
8
+ import { readServicePort, markServiceUnreachable, isConnRefuseError } from '../../../shared/service-discovery.mjs';
8
9
  // --- Types ---
9
10
  /** Known auto-detect targets: port file path relative to tmpdir.
10
11
  * Note: `mixdog` used to self-loopback via active-instance.json's
@@ -13,7 +14,7 @@ import { shutdownStdioChild, killStdioChildTreeFast } from './child-tree.mjs';
13
14
  * in-process through agent's toolExecutor (see orchestrator/internal-tools),
14
15
  * so this registry is for genuinely external port-based MCP targets only. */
15
16
  const AUTO_DETECT_PORTS = {
16
- 'mixdog-memory': { dir: 'mixdog', file: 'active-instance.json', portField: 'memory_port', endpoint: '/mcp' },
17
+ 'mixdog-memory': { discovery: 'memory', dir: 'mixdog', file: 'active-instance.json', portField: 'memory_port', endpoint: '/mcp' },
17
18
  };
18
19
  const DEFAULT_MCP_CALL_TIMEOUT_MS = 0;
19
20
  // Per-server STARTUP handshake budget (connect + listTools). Codex parity: 10s.
@@ -452,20 +453,31 @@ async function connectServer(name, cfg, genAtStart = _connectAbortGeneration) {
452
453
  const client = new Client({ name: `mixdog-agent/${name}`, version: '1.0.0' });
453
454
  let transport;
454
455
  const kind = resolveMcpTransportKind(cfg);
456
+ // When the autoDetect port comes from a discovery advert (not the legacy
457
+ // port file), remember { service, port } so a connect/handshake failure can
458
+ // distrust it — a pid-live advert can point at a recycled-pid corpse port,
459
+ // and this transport has no other health probe of its own.
460
+ let _autoDetectAdvert = null;
455
461
  // Auto-detect: read port from a running service's port file
456
462
  if (kind === 'autoDetect') {
457
463
  const spec = AUTO_DETECT_PORTS[cfg.autoDetect];
458
464
  if (!spec)
459
465
  throw new Error(`Unknown autoDetect target: "${cfg.autoDetect}"`);
460
- const portFile = spec.dir === 'mixdog' && process.env.MIXDOG_RUNTIME_ROOT
466
+ // Prefer the single-writer discovery advert (discovery/<service>.json),
467
+ // pid-validated. Fall back to the legacy active-instance.json portField
468
+ // when no live advert is present (cross-version compat).
469
+ let port = spec.discovery ? readServicePort(spec.discovery, { requirePid: false }) : null;
470
+ if (port && spec.discovery) _autoDetectAdvert = { service: spec.discovery, port };
471
+ let portFile = null;
472
+ if (!port) {
473
+ portFile = spec.dir === 'mixdog' && process.env.MIXDOG_RUNTIME_ROOT
461
474
  ? join(process.env.MIXDOG_RUNTIME_ROOT, spec.file)
462
475
  : join(tmpdir(), spec.dir, spec.file);
463
- if (!existsSync(portFile)) {
476
+ if (!existsSync(portFile)) {
464
477
  throw new Error(`autoDetect server "${name}": port file missing (${portFile})`);
465
- }
466
- let port;
467
- const raw = readFileSync(portFile, 'utf-8').trim();
468
- if (spec.portField) {
478
+ }
479
+ const raw = readFileSync(portFile, 'utf-8').trim();
480
+ if (spec.portField) {
469
481
  try {
470
482
  const json = JSON.parse(raw);
471
483
  const v = json[spec.portField];
@@ -477,12 +489,13 @@ async function connectServer(name, cfg, genAtStart = _connectAbortGeneration) {
477
489
  if (jsonErr instanceof Error && jsonErr.message.startsWith('autoDetect server')) throw jsonErr;
478
490
  throw new Error(`autoDetect server "${name}": invalid JSON in port file ${portFile}`);
479
491
  }
480
- }
481
- else {
492
+ }
493
+ else {
482
494
  port = parseInt(raw, 10);
495
+ }
483
496
  }
484
497
  if (!Number.isFinite(port) || port < 1 || port > 65535) {
485
- throw new Error(`autoDetect server "${name}": invalid port value in ${portFile}`);
498
+ throw new Error(`autoDetect server "${name}": invalid port value${portFile ? ` in ${portFile}` : ''}`);
486
499
  }
487
500
  const url = `http://127.0.0.1:${port}${spec.endpoint}`;
488
501
  transport = new StreamableHTTPClientTransport(new URL(url));
@@ -568,6 +581,11 @@ async function connectServer(name, cfg, genAtStart = _connectAbortGeneration) {
568
581
  ({ instructions, toolsResult } = await handshake);
569
582
  }
570
583
  } catch (err) {
584
+ // A discovery-advert port that fails to connect is a corpse (recycled
585
+ // pid): distrust it so the next connect falls back to the legacy port
586
+ // file instead of re-trusting the same advert. Connection-level errors
587
+ // ONLY — a startup/handshake timeout is a slow-but-alive server.
588
+ if (_autoDetectAdvert && isConnRefuseError(err)) markServiceUnreachable(_autoDetectAdvert.service, _autoDetectAdvert.port);
571
589
  if (startupTimedOut) {
572
590
  // Tear down the pending transport/child so a hung handshake never
573
591
  // leaks a stdio process or socket — fire-and-forget (like the
@@ -396,7 +396,14 @@ function toAnthropicMessages(messages) {
396
396
  // turn after tool_result renders as `</function_results>\n\nHuman:`
397
397
  // on the wire and trains the model toward 3-token empty end_turn
398
398
  // completions (see foldUserTextIntoToolResultTail).
399
- if (m.role === 'user' && foldUserTextIntoToolResultTail(result, normalizeContentForAnthropic(m.content))) {
399
+ // EXCEPTION: steering-origin user messages (human/TUI interjections)
400
+ // must keep their own user turn so their provenance survives — folding
401
+ // them into the preceding tool_result would disguise user input as
402
+ // tool output. Anthropic accepts a user text message after a
403
+ // tool_result message, so the distinct turn stays request-valid.
404
+ const isSteering = m.role === 'user' && m.meta?.source === 'steering';
405
+ if (m.role === 'user' && !isSteering
406
+ && foldUserTextIntoToolResultTail(result, normalizeContentForAnthropic(m.content))) {
400
407
  continue;
401
408
  }
402
409
  result.push({ role: m.role, content: normalizeContentForAnthropic(m.content) });
@@ -397,7 +397,13 @@ function toAnthropicMessages(messages) {
397
397
  // First-party client parity: fold a user text turn that directly follows a
398
398
  // tool_result turn into that tool_result's content (empty end_turn
399
399
  // livelock prevention; see foldUserTextIntoToolResultTail).
400
- if (m.role === 'user' && foldUserTextIntoToolResultTail(result, normalizeContentForAnthropic(m.content))) {
400
+ // EXCEPTION: steering-origin user messages (human/TUI interjections)
401
+ // keep their own user turn so provenance survives — folding them would
402
+ // disguise user input as tool output. Anthropic accepts a user text
403
+ // message after a tool_result message, so the turn stays request-valid.
404
+ const isSteering = m.role === 'user' && m.meta?.source === 'steering';
405
+ if (m.role === 'user' && !isSteering
406
+ && foldUserTextIntoToolResultTail(result, normalizeContentForAnthropic(m.content))) {
401
407
  continue;
402
408
  }
403
409
  result.push({ role: m.role, content: normalizeContentForAnthropic(m.content) });
@@ -405,6 +411,13 @@ function toAnthropicMessages(messages) {
405
411
  return sanitizeAnthropicContentPairs(result);
406
412
  }
407
413
 
414
+ // Test-only: expose the lowering so the steering-provenance test can assert
415
+ // the API-key provider keeps steering-tagged user turns distinct (mirrors
416
+ // anthropic-oauth._buildRequestBodyForCacheSmoke coverage).
417
+ export function _toAnthropicMessagesForTest(messages) {
418
+ return toAnthropicMessages(messages);
419
+ }
420
+
408
421
  // Applies cache_control markers to the FINAL, already-sanitized Anthropic
409
422
  // message array — by INVARIANT, never by pre-sanitize index. Because
410
423
  // sanitizeAnthropicContentPairs has already run (and must NOT run again
@@ -242,10 +242,15 @@ function windowFromPercent(label, value, source) {
242
242
  }
243
243
 
244
244
  // Grok Build billing periods are migrating from monthly to a shared weekly
245
- // pool (xAI rollout is per-account). Derive the window cadence from the
246
- // actual billing period length instead of hardcoding monthly: <=10 days
247
- // means a weekly cycle, anything longer (or unknown) stays monthly.
245
+ // pool (xAI rollout is per-account). Unified-billing accounts state the
246
+ // cadence explicitly via currentPeriod.type (USAGE_PERIOD_TYPE_WEEKLY);
247
+ // otherwise derive it from the actual billing period length instead of
248
+ // hardcoding monthly: <=10 days means a weekly cycle, anything longer (or
249
+ // unknown) stays monthly.
248
250
  function grokBillingCadence(config) {
251
+ const periodType = cleanString(config?.currentPeriod?.type ?? config?.current_period?.type ?? '').toUpperCase();
252
+ if (periodType.includes('WEEKLY')) return { label: 'W', period: 'weekly' };
253
+ if (periodType.includes('MONTHLY')) return { label: 'M', period: 'monthly' };
249
254
  if (config?.weeklyLimit !== undefined || config?.weekly_limit !== undefined) {
250
255
  return { label: 'W', period: 'weekly' };
251
256
  }
@@ -259,6 +264,11 @@ function grokBillingCadence(config) {
259
264
 
260
265
  function creditWindowFromBilling(config) {
261
266
  if (!config || typeof config !== 'object') return null;
267
+ const cadence = grokBillingCadence(config);
268
+ const resetAt = resetAtMs(
269
+ config.currentPeriod?.end ?? config.current_period?.end
270
+ ?? config.billingPeriodEnd ?? config.billing_period_end,
271
+ );
262
272
  const limit = num(
263
273
  config.weeklyLimit?.val ?? config.weeklyLimit
264
274
  ?? config.monthlyLimit?.val ?? config.monthlyLimit
@@ -266,10 +276,27 @@ function creditWindowFromBilling(config) {
266
276
  null,
267
277
  );
268
278
  const used = num(config.used?.val ?? config.includedUsed?.val ?? config.used, null);
269
- if (limit === null || used === null || !(limit > 0)) return null;
270
- const resetAt = resetAtMs(config.billingPeriodEnd ?? config.billing_period_end);
279
+ if (limit === null || used === null || !(limit > 0)) {
280
+ // Unified-billing accounts (shared weekly pool) report utilization as a
281
+ // percentage instead of credit totals; absent means 0% used.
282
+ const unified = config.isUnifiedBillingUser === true
283
+ || config.is_unified_billing_user === true
284
+ || config.currentPeriod || config.current_period;
285
+ if (!unified) return null;
286
+ const usedPct = num(
287
+ config.creditUsagePercent?.val ?? config.creditUsagePercent
288
+ ?? config.credit_usage_percent,
289
+ 0,
290
+ );
291
+ return {
292
+ label: cadence.label,
293
+ source: 'grok-build-billing',
294
+ usedPct: round(Math.min(100, Math.max(0, usedPct)), 2),
295
+ ...(resetAt ? { resetAt } : {}),
296
+ };
297
+ }
271
298
  return {
272
- label: grokBillingCadence(config).label,
299
+ label: cadence.label,
273
300
  source: 'grok-build-billing',
274
301
  usedPct: round(Math.min(100, used * 100 / limit), 2),
275
302
  usedCredits: round(used, 2),
@@ -495,29 +522,38 @@ async function fetchGrokUsage(providerObj, routeInfo) {
495
522
  Authorization: `Bearer ${token}`,
496
523
  'X-XAI-Token-Auth': 'xai-grok-cli',
497
524
  'x-userid': userId,
525
+ 'x-grok-client-version': '0.2.87',
498
526
  Accept: 'application/json',
499
- 'User-Agent': 'xai-grok-build/0.2.16',
527
+ 'User-Agent': 'xai-grok-build/0.2.87',
500
528
  };
501
529
 
502
- try {
503
- const res = await fetch('https://cli-chat-proxy.grok.com/v1/billing', fetchOptions(cliHeaders));
504
- if (res.ok) {
530
+ // format=credits is what the official grok CLI requests; unified-billing
531
+ // (shared weekly pool) accounts only report their real cadence and reset
532
+ // there. The legacy /v1/billing shape keeps serving a stale monthly cycle
533
+ // for migrated accounts, so it is only a fallback.
534
+ for (const url of [
535
+ 'https://cli-chat-proxy.grok.com/v1/billing?format=credits',
536
+ 'https://cli-chat-proxy.grok.com/v1/billing',
537
+ ]) {
538
+ try {
539
+ const res = await fetch(url, fetchOptions(cliHeaders));
540
+ if (!res.ok) continue;
505
541
  const data = await res.json();
506
542
  const config = data?.config && typeof data.config === 'object' ? data.config : data;
507
- const monthly = creditWindowFromBilling(config);
508
- if (monthly) {
543
+ const window = creditWindowFromBilling(config);
544
+ if (window) {
509
545
  return {
510
546
  provider: routeInfo?.provider || 'grok-oauth',
511
547
  model: routeInfo?.model || null,
512
548
  source: 'grok-build-billing',
513
- quotaWindows: [monthly],
549
+ quotaWindows: [window],
514
550
  balance: balanceFromGrokBilling(config),
515
551
  rawKeys: Object.keys(data || {}).sort(),
516
552
  };
517
553
  }
554
+ } catch {
555
+ // Fall through to the next candidate / generic probes below.
518
556
  }
519
- } catch {
520
- // Fall through to generic probes below.
521
557
  }
522
558
 
523
559
  // xAI documents per-request cost tracking and console rate-limit pages, but
@@ -202,7 +202,11 @@ export async function agentLoop(provider, messages, model, tools, onToolCall, cw
202
202
  if (typeof options.beforeAppend === 'function') {
203
203
  try { options.beforeAppend(); } catch { /* best-effort hook */ }
204
204
  }
205
- messages.push({ role: 'user', content: merged.content });
205
+ // Tag steering-origin user messages so provider lowering keeps them a
206
+ // distinct user turn (see anthropic-oauth foldUserTextIntoToolResultTail):
207
+ // human/TUI steering must stay attributed as user input, never folded
208
+ // into a preceding tool_result where it reads as tool output.
209
+ messages.push({ role: 'user', content: merged.content, meta: { source: 'steering' } });
206
210
  try { opts.onSteerMessage?.(merged.text || steeringContentText(merged.content)); } catch {}
207
211
  if (sessionId) {
208
212
  try { process.stderr.write(`[steer] sess=${sessionId} injected ${stage} user message (merged=${merged.count} len=${String(merged.text || '').length})\n`); } catch {}
@@ -40,6 +40,38 @@ function _pathsOf(arg) {
40
40
  function _normPath(p) { return String(p).replace(/\\/g, '/').toLowerCase(); }
41
41
  function _baseName(p) { const s = _normPath(p); const i = s.lastIndexOf('/'); return i >= 0 ? s.slice(i + 1) : s; }
42
42
  function _dirName(p) { const s = _normPath(p); const i = s.lastIndexOf('/'); return i >= 0 ? s.slice(0, i) : ''; }
43
+ // A path targets a specific file (not a directory scope) when its basename
44
+ // carries a file extension — the signal that a grep was already file-scoped.
45
+ function _isFileScopedPath(p) { return /\.[A-Za-z0-9]{1,12}$/.test(_baseName(p)); }
46
+ // Grep context flag: -C/-A/-B count as context ONLY when > 0. An explicit
47
+ // -C:0 / -A:0 / -B:0 is NO context (bare match lines).
48
+ function _hasGrepContext(a) { return Number(a?.['-A']) > 0 || Number(a?.['-B']) > 0 || Number(a?.['-C']) > 0; }
49
+ // Canonical path segments: normalize slashes/case, drop '.' and resolve '..'
50
+ // so ./ and ../ variants collapse. Used by the abs/rel-tolerant path match.
51
+ function _canonSegs(p) {
52
+ const out = [];
53
+ for (const seg of _normPath(p).split('/')) {
54
+ if (seg === '' || seg === '.') continue;
55
+ if (seg === '..') { if (out.length && out[out.length - 1] !== '..') out.pop(); else out.push('..'); continue; }
56
+ out.push(seg);
57
+ }
58
+ return out;
59
+ }
60
+ function _canonPath(p) { return _canonSegs(p).join('/'); }
61
+ // abs/rel-tolerant same-file test: equal when one canonical path is a
62
+ // segment-aligned suffix of the other (so C:/proj/src/x.mjs, src/x.mjs and
63
+ // ./src/x.mjs all match). Basename equality is implied by the suffix compare.
64
+ function _pathsMatch(a, b) {
65
+ if (!a || !b) return false;
66
+ const A = _canonSegs(a); const B = _canonSegs(b);
67
+ if (!A.length || !B.length) return false;
68
+ const long = A.length >= B.length ? A : B;
69
+ const short = A.length >= B.length ? B : A;
70
+ for (let i = 0; i < short.length; i += 1) {
71
+ if (long[long.length - short.length + i] !== short[i]) return false;
72
+ }
73
+ return true;
74
+ }
43
75
  // Exact (normalized) same-file test — used where "related" is too loose.
44
76
  function _samePath(a, b) { return !!a && !!b && _normPath(a) === _normPath(b); }
45
77
  // Loose sibling/variant relation — cluster detector pairs this WITH token overlap.
@@ -148,6 +180,15 @@ export function createSteeringLadder(ctx) {
148
180
  // same-path regions in ONE call are the recommended batched form and are
149
181
  // never counted. Fires once per path per session.
150
182
  const _readWindowsByPath = new Map();
183
+ // Detector A state: prior turn's file-scoped grep paths that ran WITHOUT
184
+ // any -C/-B/-A context (a bare match-line grep on a specific file). Seeds
185
+ // the grep-no-context-then-read nudge for the next turn.
186
+ let _lastFileScopedGrepNoCtxPaths = null;
187
+ // Detector B state: per-path consecutive-read streak (canonical path ->
188
+ // count). Tracked as a Map so multi-file turns keep independent streaks
189
+ // ([A,B]->B->B fires on B). Files not read this turn are pruned; a path is
190
+ // dropped after it fires.
191
+ const _readStreaks = new Map();
151
192
 
152
193
  // Level-2 steering emitter shared by both ladder paths (single-call
153
194
  // level-1 streak and the independent all-read-only streak). Sets the latch
@@ -178,20 +219,37 @@ export function createSteeringLadder(ctx) {
178
219
  // Steering hint gate: at most ONE hint per turn (priority:
179
220
  // level-2 > grep/shell detectors > level-1).
180
221
  let _hintFiredThisTurn = hintAlreadyFired;
222
+ // Specific-detector predicates (A: grep-no-context then re-read; B:
223
+ // 3rd consecutive same-file read). Evaluated up front so the GENERIC
224
+ // level-1 batching nudge can defer to them (more specific wins),
225
+ // while the level-2 runaway escalation below still outranks them.
226
+ const _readCallsThisTurn = calls.filter((c) => c?.name === 'read');
227
+ const _aWouldFire = !!_lastFileScopedGrepNoCtxPaths
228
+ && _readCallsThisTurn.some((c) => _pathsOf(c?.arguments?.path)
229
+ .some((p) => [..._lastFileScopedGrepNoCtxPaths].some((g) => _pathsMatch(p, g))));
230
+ let _bWouldFire = false;
231
+ for (const c of _readCallsThisTurn) {
232
+ for (const w of _readWindows(c)) {
233
+ for (const [k, cnt] of _readStreaks) { if (cnt >= 2 && _pathsMatch(k, w.path)) { _bWouldFire = true; break; } }
234
+ if (_bWouldFire) break;
235
+ }
236
+ if (_bWouldFire) break;
237
+ }
181
238
  // Missed-parallelism steering: 2+ consecutive turns of a single
182
239
  // read-only tool call suggest the model isn't batching independent
183
240
  // lookups. Nudge once, then reset (fires again after 2 more).
184
241
  if (calls.length === 1 && isEagerDispatchable(calls[0].name, tools)) {
185
242
  _serialReadOnlyStreak += 1;
186
- if (_serialReadOnlyStreak >= 2 && !_hintFiredThisTurn) {
243
+ // Escalation ladder (Step 1). Cumulative level-1 fires are tracked
244
+ // and NEVER reset. Once level-1 has fired >=3 times with ZERO edits,
245
+ // escalate to level-2 steering (latched once per 5 turns). The
246
+ // level-2 escalation always wins; the plain batching nudge instead
247
+ // DEFERS to the specific A/B detectors when either would fire.
248
+ const _canEscalate = (_level1FireCount + 1) >= 3 && editCount === 0 && (iterations - _level2LatchAtIteration) >= 5;
249
+ if (_serialReadOnlyStreak >= 2 && !_hintFiredThisTurn && (_canEscalate || (!_aWouldFire && !_bWouldFire))) {
187
250
  _serialReadOnlyStreak = 0;
188
- // Escalation ladder (Step 1). Cumulative level-1 fires are
189
- // tracked and NEVER reset. Once level-1 has fired >=3 times with
190
- // ZERO edits, escalate to level-2 steering (blocked-report is a
191
- // valid completion) instead of the batching nudge — latched to at
192
- // most once per 5 turns.
193
251
  _level1FireCount += 1;
194
- if (_level1FireCount >= 3 && editCount === 0 && (iterations - _level2LatchAtIteration) >= 5) {
252
+ if (_canEscalate) {
195
253
  _emitLevel2Steer();
196
254
  } else {
197
255
  pushSystemReminder('Last 2 turns each ran a single read-only tool. Batch independent lookups (read/grep/glob/code_graph) into ONE turn, or start editing if you have enough context.');
@@ -303,8 +361,10 @@ export function createSteeringLadder(ctx) {
303
361
  const _next = new Set();
304
362
  for (const c of _grepCalls) {
305
363
  const a = c?.arguments || {};
306
- const _hasCtx = a['-A'] != null || a['-B'] != null || a['-C'] != null;
307
- const _isContent = a.output_mode === 'content' || a.output_mode === 'content_with_context' || _hasCtx;
364
+ // Bare output_mode:'content' WITHOUT real context is routed to
365
+ // Detector A (which gives the add-`-C` advice); this detector
366
+ // only owns greps that actually returned context.
367
+ const _isContent = _hasGrepContext(a) || a.output_mode === 'content_with_context';
308
368
  const _pathsOnly = a.output_mode === 'files_with_matches' || a.output_mode === 'count';
309
369
  const _capped = a.head_limit != null;
310
370
  if (_isContent && !_pathsOnly && !_capped) {
@@ -313,6 +373,55 @@ export function createSteeringLadder(ctx) {
313
373
  }
314
374
  _lastGrepContentPaths = _next.size ? _next : null;
315
375
  }
376
+ // Detector A (fire): a file-scoped grep WITHOUT context last turn,
377
+ // then a read of that same file this turn. Fires ahead of the read-
378
+ // fragmentation / B detectors below; the generic level-1 nudge
379
+ // already deferred to it via _aWouldFire, while the level-2
380
+ // escalation above still outranks it. On pre-emption by an
381
+ // earlier hint, the pending set is kept (not cleared) so A can fire
382
+ // next turn.
383
+ let _aPending = false;
384
+ if (_lastFileScopedGrepNoCtxPaths) {
385
+ const _rp = [];
386
+ for (const c of _readCallsThisTurn) { for (const p of _pathsOf(c?.arguments?.path)) _rp.push(p); }
387
+ const _hit = _rp.find((p) => [..._lastFileScopedGrepNoCtxPaths].some((g) => _pathsMatch(p, g)));
388
+ if (_hit) {
389
+ if (!_hintFiredThisTurn) {
390
+ pushSystemReminder(`You grepped ${_baseName(_hit)} last turn then re-opened it — request context with -C on file-scoped grep so the match window is already on screen.`);
391
+ try {
392
+ appendAgentTrace({
393
+ sessionId,
394
+ iteration: iterations,
395
+ kind: 'steer',
396
+ payload: { tag: 'grep_nocontext_reread', file: _baseName(_hit) },
397
+ agent: sessionAgent || null,
398
+ });
399
+ } catch { /* best-effort */ }
400
+ _hintFiredThisTurn = true;
401
+ _lastFileScopedGrepNoCtxPaths = null;
402
+ } else {
403
+ _aPending = true; // pre-empted this turn — keep pending for next turn
404
+ }
405
+ }
406
+ }
407
+ // Detector A (seed): record this turn's file-scoped greps that ran
408
+ // WITHOUT context so a same-file read next turn triggers the fire
409
+ // block above. When A was pre-empted this turn (_aPending) the
410
+ // existing pending set is kept rather than cleared.
411
+ if (!_aPending) {
412
+ const _next = new Set();
413
+ for (const c of _grepCalls) {
414
+ const a = c?.arguments || {};
415
+ // Exact complement of the content detector's seed: skip greps
416
+ // that carried real context (positive -A/-B/-C) OR ran in
417
+ // content_with_context mode — those belong to that detector.
418
+ if (_hasGrepContext(a) || a.output_mode === 'content_with_context') continue;
419
+ for (const p of _pathsOf(a.path)) {
420
+ if (_isFileScopedPath(p)) _next.add(p);
421
+ }
422
+ }
423
+ _lastFileScopedGrepNoCtxPaths = _next.size ? _next : null;
424
+ }
316
425
  // Detector: read fragmentation — 2nd+ single-window read into the
317
426
  // SAME file with offsets inside an 800-line span means the model is
318
427
  // paging small windows across turns instead of reading one wider
@@ -345,6 +454,50 @@ export function createSteeringLadder(ctx) {
345
454
  }
346
455
  }
347
456
  }
457
+ // Detector B: 3rd consecutive turn reading the SAME file (any
458
+ // window). Per-path streaks in a Map so multi-file turns keep
459
+ // independent counts ([A,B]->B->B fires on B). rel/abs spellings of
460
+ // one file collapse onto a single streak key; files not read this
461
+ // turn are pruned; a path is dropped after it fires.
462
+ {
463
+ // Resolve each read path to one streak key, merging rel/abs
464
+ // spellings against existing keys and keys already seen this turn.
465
+ const _touched = new Set();
466
+ for (const c of calls.filter((x) => x?.name === 'read')) {
467
+ for (const w of _readWindows(c)) {
468
+ let _key = null;
469
+ for (const k of _readStreaks.keys()) { if (_pathsMatch(k, w.path)) { _key = k; break; } }
470
+ if (!_key) { for (const k of _touched) { if (_pathsMatch(k, w.path)) { _key = k; break; } } }
471
+ _touched.add(_key || _canonPath(w.path));
472
+ }
473
+ }
474
+ // Prune streaks for files not read this turn (breaks consecutiveness).
475
+ for (const k of [..._readStreaks.keys()]) { if (!_touched.has(k)) _readStreaks.delete(k); }
476
+ // Increment each file read this turn (once per turn).
477
+ for (const k of _touched) {
478
+ _readStreaks.set(k, (_readStreaks.get(k) || 0) + 1);
479
+ if (_readStreaks.size > 32) _readStreaks.delete(_readStreaks.keys().next().value);
480
+ }
481
+ if (!_hintFiredThisTurn) {
482
+ for (const [k, cnt] of _readStreaks) {
483
+ if (cnt >= 3) {
484
+ pushSystemReminder(`3rd window into ${_baseName(k)} — take the remaining span in ONE wide read (offset+limit) instead of paging.`);
485
+ try {
486
+ appendAgentTrace({
487
+ sessionId,
488
+ iteration: iterations,
489
+ kind: 'steer',
490
+ payload: { tag: 'consecutive_same_file_read', file: _baseName(k) },
491
+ agent: sessionAgent || null,
492
+ });
493
+ } catch { /* best-effort */ }
494
+ _readStreaks.delete(k);
495
+ _hintFiredThisTurn = true;
496
+ break;
497
+ }
498
+ }
499
+ }
500
+ }
348
501
  // Detector: read-only shell (git status/log/diff --stat, ls/dir/cat/
349
502
  // type/Get-ChildItem) inspects state the dedicated tools cover.
350
503
  // Nudge toward them; never blocks execution.
@@ -366,6 +519,8 @@ export function createSteeringLadder(ctx) {
366
519
  _lastGrepContentPaths = null;
367
520
  _identGrepStreak = 0;
368
521
  _identGrepToken = null;
522
+ _lastFileScopedGrepNoCtxPaths = null;
523
+ _readStreaks.clear();
369
524
  },
370
525
  };
371
526
  }
@@ -106,3 +106,48 @@ function _restoreCompactedBodies(tcVal, origVal, key) {
106
106
  }
107
107
  return tcVal;
108
108
  }
109
+
110
+ // Marker prefix emitted by compactStoredToolArgString for every compacted
111
+ // body/long arg (`[mixdog compacted <key>: <N> chars, sha256:…]`). Both the
112
+ // body-only form (marker alone) and the long form (marker + head/tail preview)
113
+ // begin with this exact span.
114
+ const COMPACTED_MARKER_PREFIX = '[mixdog compacted ';
115
+
116
+ function _isCompactedPlaceholderString(v) {
117
+ return typeof v === 'string' && v.startsWith(COMPACTED_MARKER_PREFIX);
118
+ }
119
+
120
+ // Recursively drop every key whose stored value is a compacted-placeholder
121
+ // string, at any depth (batch shapes like edits[].old_string carry nested
122
+ // compacted bodies too). Non-placeholder values are kept verbatim.
123
+ function _dropCompactedPlaceholders(val) {
124
+ if (Array.isArray(val)) return val.map((item) => _dropCompactedPlaceholders(item));
125
+ if (val && typeof val === 'object') {
126
+ const out = {};
127
+ for (const [k, v] of Object.entries(val)) {
128
+ if (_isCompactedPlaceholderString(v)) continue; // drop the key entirely
129
+ out[k] = _dropCompactedPlaceholders(v);
130
+ }
131
+ return out;
132
+ }
133
+ return val;
134
+ }
135
+
136
+ // A SUCCESSFUL tool call's compacted body/long arg (patch / old_string /
137
+ // new_string / content / rewrite / command / script) is never needed again —
138
+ // the edit already applied. compactToolCallsForHistory (run at push time,
139
+ // before the outcome is known) leaves a `[mixdog compacted …]` placeholder in
140
+ // its place. For a success that placeholder persists in the stored assistant
141
+ // tool_use and is transmitted back as a prior apply_patch INPUT, which the
142
+ // model copies verbatim as new patch args — caught by the patch guard, but only
143
+ // after a wasted turn. Drop those placeholder keys so no resubmittable
144
+ // placeholder body survives in history. Failed calls take the opposite path
145
+ // (restoreToolCallBodyForId expands to the full original); success and failure
146
+ // are mutually exclusive per call id, so this never races the restore path.
147
+ // Cache-safe: the caller runs this before the assistant message is transmitted.
148
+ export function dropCompactedBodyArgsForId(assistantMsg, callId) {
149
+ if (!assistantMsg || !Array.isArray(assistantMsg.toolCalls) || !callId) return;
150
+ const tc = assistantMsg.toolCalls.find((t) => t && t.id === callId);
151
+ if (!tc || !tc.arguments || typeof tc.arguments !== 'object') return;
152
+ tc.arguments = _dropCompactedPlaceholders(tc.arguments);
153
+ }
@@ -157,6 +157,10 @@ async function runRecallFastTrackForSession(session, messages, opts = {}) {
157
157
  messages,
158
158
  cwd: session?.cwd,
159
159
  limit: hydrateLimit,
160
+ // Clear/manual-compact path: these rows are about to be summarized
161
+ // away, so skip the bounded 15s synchronous embedding-flush wait —
162
+ // kick the flush fire-and-forget (dense-search immediacy is moot here).
163
+ embedWait: false,
160
164
  }, callerCtx, memoryTimeoutMs);
161
165
  } catch (err) {
162
166
  // Ingest failed (dead/timed-out memory runtime). The transcript is NOT
@@ -30,7 +30,7 @@ import { preDispatchDenyForSession } from './loop/pre-dispatch-deny.mjs';
30
30
  import { executeTool } from './loop/tool-exec.mjs';
31
31
  import { crossTurnSignature, crossTurnDedupStub, isEditProgressTool } from './loop/completion-guards.mjs';
32
32
  import { getToolKind, isEagerDispatchable, parseNativeToolSearchPayload } from './loop/tool-helpers.mjs';
33
- import { restoreToolCallBodyForId } from './loop/stored-tool-args.mjs';
33
+ import { restoreToolCallBodyForId, dropCompactedBodyArgsForId } from './loop/stored-tool-args.mjs';
34
34
 
35
35
  export async function processToolBatch(ctx) {
36
36
  const {
@@ -586,6 +586,17 @@ export async function processToolBatch(ctx) {
586
586
  toolKind: 'error',
587
587
  });
588
588
  }
589
+ // Successful call: its compacted body/long args are never needed
590
+ // again. Drop any `[mixdog compacted …]` placeholder from the stored
591
+ // assistant tool_use so a prior apply_patch INPUT never surfaces a
592
+ // resubmittable placeholder patch body the model copies back verbatim
593
+ // (the patch guard catches it, but only after a wasted turn). Gated on
594
+ // full success + clean post-processing; the failed/post-fail paths run
595
+ // restoreToolCallBodyForId instead (mutually exclusive per call id).
596
+ // Cache-safe: assistantTurnMsg is not transmitted until the next send.
597
+ if (_postProcessOk && _executeOk && _resultKind === 'normal' && call?.id) {
598
+ dropCompactedBodyArgsForId(assistantTurnMsg, call.id);
599
+ }
589
600
  // Soft-cancel after each tool: if close landed during execution,
590
601
  // discard the rest of the batch and skip the next provider.send.
591
602
  throwIfAborted();
@@ -19,6 +19,7 @@ import {
19
19
  getBackgroundTask,
20
20
  } from '../../../../shared/background-tasks.mjs';
21
21
  import { startChildGuardian } from '../../../../shared/child-guardian.mjs';
22
+ import { detachedSpawnOpts } from '../../../../shared/spawn-flags.mjs';
22
23
  import {
23
24
  getShellJobsDir,
24
25
  shellJobStdoutPath,
@@ -828,9 +829,8 @@ function _startBackgroundShellJobImpl({ command, timeoutMs, workDir, mergeStderr
828
829
  const child = spawn(shell, [wrappedTempPath], {
829
830
  cwd: workDir,
830
831
  env: scrubLoaderVars(scrubProviderSecrets({ ...spawnEnv })),
831
- detached: true,
832
832
  stdio: 'ignore',
833
- windowsHide: true,
833
+ ...detachedSpawnOpts,
834
834
  });
835
835
  startChildGuardian({
836
836
  childPid: child.pid,
@@ -129,15 +129,19 @@ export function wrapPowerShellWithCwdProbe(command, stateFile) {
129
129
  // a failing native call $LASTEXITCODE holds its real code (case3 → 7).
130
130
  //
131
131
  // The `if (-not $?)` check MUST be the first statement after the command;
132
- // any intervening statement resets $?. Terminating errors (`throw`) skip to
133
- // catch → $__ec = 1. The cwd probe runs in finally regardless, and we exit
132
+ // any intervening statement resets $?. Terminating errors (`throw`,
133
+ // `-ErrorAction Stop`, .NET exceptions) skip to catch → $__ec = 1. A
134
+ // *caught* terminating error is NOT auto-written to the error stream, so
135
+ // without re-emitting it the caller only sees `[exit code: 1] (no output)`.
136
+ // Re-surface the ErrorRecord on stderr so the failure cause reaches the
137
+ // result envelope. The cwd probe runs in finally regardless, and we exit
134
138
  // with $__ec so the probe never masks the observed code.
135
139
  return '$global:LASTEXITCODE = 0\n'
136
140
  + '$__ec = 0\n'
137
141
  + 'try {\n'
138
142
  + `${command}\n`
139
143
  + ' if (-not $?) { $__ec = if ($LASTEXITCODE) { $LASTEXITCODE } else { 1 } }\n'
140
- + '} catch { $__ec = 1 }\n'
144
+ + '} catch { [Console]::Error.WriteLine(($_ | Out-String).Trim()); $__ec = 1 }\n'
141
145
  + `finally { (Get-Location).Path | Out-File -Encoding utf8 ${qFile} }\n`
142
146
  + 'exit $__ec';
143
147
  }
@@ -435,14 +435,21 @@ class DiscordBackend {
435
435
  const msg = await ch.messages.fetch(messageId);
436
436
  if (msg.attachments.size === 0) return [];
437
437
  const results = [];
438
+ // Optional filter bounds what gets fetched (e.g. image/* only) so a
439
+ // caller after images never pulls unrelated large attachments.
440
+ const filter = typeof opts.filter === "function" ? opts.filter : null;
438
441
  for (const att of msg.attachments.values()) {
439
- const path = await this.downloadSingleAttachment(att, opts);
440
- results.push({
442
+ const meta = {
441
443
  id: att.id,
442
- path,
443
444
  name: safeAttName(att),
444
445
  contentType: att.contentType ?? "unknown",
445
446
  size: att.size
447
+ };
448
+ if (filter && !filter(meta)) continue;
449
+ const path = await this.downloadSingleAttachment(att, opts);
450
+ results.push({
451
+ ...meta,
452
+ path
446
453
  });
447
454
  }
448
455
  return results;