pi-crew 0.10.2 → 0.10.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. package/CHANGELOG.md +249 -0
  2. package/dist/index.mjs +98 -307
  3. package/package.json +2 -1
  4. package/schema.json +11 -0
  5. package/skills/real-test-pi-crew/REPORT-TEMPLATE.md +6 -2
  6. package/skills/real-test-pi-crew/SKILL.md +278 -79
  7. package/src/config/config-merge.ts +11 -1
  8. package/src/config/config-validation.ts +40 -1
  9. package/src/config/config.ts +28 -6
  10. package/src/config/defaults.ts +35 -10
  11. package/src/config/env-vars.ts +27 -2
  12. package/src/config/types.ts +36 -0
  13. package/src/extension/registration/lifecycle-handlers.ts +40 -9
  14. package/src/extension/registration/team-tool.ts +53 -5
  15. package/src/extension/team-tool/doctor.ts +364 -7
  16. package/src/extension/team-tool/handle-settings.ts +19 -0
  17. package/src/extension/team-tool/inspect.ts +10 -2
  18. package/src/extension/team-tool/status.ts +7 -0
  19. package/src/extension/team-tool.ts +35 -2
  20. package/src/hooks/registry.ts +59 -56
  21. package/src/prompt/inbox-poll.ts +90 -0
  22. package/src/prompt/message-tool.ts +166 -0
  23. package/src/prompt/prompt-runtime.ts +201 -18
  24. package/src/prompt/surface-worker.ts +720 -0
  25. package/src/prompt/worker-events-channel.ts +49 -3
  26. package/src/runtime/async-runner.ts +29 -1
  27. package/src/runtime/background-runner.ts +13 -7
  28. package/src/runtime/broker/broker-issuer.ts +27 -2
  29. package/src/runtime/broker/crew-broker-tokens.ts +56 -4
  30. package/src/runtime/broker/crew-broker.ts +261 -41
  31. package/src/runtime/child-pi/child-pi-spawn.ts +23 -9
  32. package/src/runtime/child-pi/child-pi-streams.ts +9 -1
  33. package/src/runtime/child-pi/child-pi.ts +353 -5
  34. package/src/runtime/crew-agent-records.ts +13 -1
  35. package/src/runtime/dispatch-batch.ts +12 -1
  36. package/src/runtime/event-log-tail-source.ts +374 -0
  37. package/src/runtime/finalize-run.ts +4 -0
  38. package/src/runtime/live-session/live-agent-manager.ts +34 -1
  39. package/src/runtime/live-session/live-control-realtime.ts +10 -0
  40. package/src/runtime/live-session/live-session-runtime.ts +47 -27
  41. package/src/runtime/manifest-cache.ts +128 -17
  42. package/src/runtime/model/pi-args.ts +54 -65
  43. package/src/runtime/output/sidechain-output.ts +61 -6
  44. package/src/runtime/process/proc-stat.ts +46 -0
  45. package/src/runtime/process/zombie-scanner.ts +32 -19
  46. package/src/runtime/spawn-policy.ts +27 -41
  47. package/src/runtime/surface/degrade.ts +776 -0
  48. package/src/runtime/surface/herdr-provider.ts +546 -0
  49. package/src/runtime/surface/launch-script.ts +172 -0
  50. package/src/runtime/surface/resolve-surface.ts +274 -0
  51. package/src/runtime/surface/surface-provider.ts +129 -0
  52. package/src/runtime/surface/surface-spawn.ts +475 -0
  53. package/src/runtime/surface/tmux-provider.ts +400 -0
  54. package/src/runtime/task-runner/child-executor.ts +47 -0
  55. package/src/runtime/task-runner/post-execution.ts +57 -2
  56. package/src/runtime/task-runner/prompt-builder.ts +1 -0
  57. package/src/runtime/task-runner/retrieval-orchestrator.ts +191 -56
  58. package/src/runtime/task-runner/state-helpers.ts +54 -30
  59. package/src/runtime/task-runner.ts +4 -2
  60. package/src/runtime/team-runner.ts +101 -0
  61. package/src/schema/config-schema.ts +24 -0
  62. package/src/state/atomic-write.ts +219 -40
  63. package/src/state/coordination/locks.ts +7 -5
  64. package/src/state/coordination/mailbox.ts +56 -10
  65. package/src/state/event-log/cursor.ts +413 -23
  66. package/src/state/event-log/event-log.ts +120 -113
  67. package/src/state/event-log/sequence-cache.ts +21 -3
  68. package/src/state/stores/state-store.ts +98 -6
  69. package/src/state/types.ts +51 -0
  70. package/src/ui/inline-panel/agent-pane.ts +3 -0
  71. package/src/ui/render-diff.ts +16 -8
  72. package/src/ui/run-dashboard.ts +87 -42
  73. package/src/ui/run-event-bus.ts +10 -1
  74. package/src/ui/run-snapshot-cache.ts +83 -35
  75. package/src/ui/transcript-cache.ts +101 -13
  76. package/src/ui/transcript-viewer.ts +92 -24
  77. package/src/ui/widget/index.ts +32 -8
  78. package/src/utils/visual.ts +43 -0
  79. package/src/worktree/worktree-manager.ts +65 -4
@@ -1,18 +1,17 @@
1
1
  /**
2
- * M3: Iterative file-retrieval orchestrator.
2
+ * M3: File-retrieval orchestrator.
3
3
  *
4
- * Pattern: workers progressively discover relevant context files
5
- * (e.g. "which source file handles X?") using ripgrep-driven keyword
6
- * search + the existing context-retrieval.ts scoring/convergence
7
- * helpers. Max 3 cycles, fall back to in-memory heuristic when
8
- * ripgrep is not available (e.g. minimal Windows CI runners).
4
+ * Pattern: workers discover relevant context files (e.g. "which
5
+ * source file handles X?") using ripgrep-driven keyword search + the
6
+ * existing context-retrieval.ts scoring/convergence helpers. Single
7
+ * discovery pass (perf round 3), fall back to in-memory heuristic
8
+ * when ripgrep is not available (e.g. minimal Windows CI runners).
9
9
  *
10
10
  * Signal flow:
11
11
  * renderTaskPrompt (in prompt-builder.ts)
12
12
  * → runRetrievalCycle(task, goal, cwd)
13
- * → cycle 1: rg --files, then score each file
14
- * → cycle 2: refine query, rg --json for keyword filter
15
- * → cycle 3: same; if !shouldContinue, stop early
13
+ * → single pass: rg --files, then score each file (deduped
14
+ * by absolute path)
16
15
  * → returns top-N files (5..10)
17
16
  * renderTaskPrompt injects "Suggested files to read (top-N by
18
17
  * retrieval score):" section before final prompt assembly.
@@ -24,10 +23,7 @@ import { spawn } from "node:child_process";
24
23
  import * as fs from "node:fs";
25
24
  import * as path from "node:path";
26
25
  import type { RelevanceEvaluation } from "./context-retrieval.ts";
27
- import { hasConverged, scoreRelevance, shouldContinue } from "./context-retrieval.ts";
28
-
29
- /** Max retrieval cycles per prompt render. Matches context-retrieval.MAX_CYCLES. */
30
- export const MAX_CYCLES = 3;
26
+ import { hasConverged, scoreRelevance } from "./context-retrieval.ts";
31
27
 
32
28
  /** Hard cap on suggested files injected into the worker prompt. */
33
29
  export const MAX_SUGGESTED_FILES = 10;
@@ -36,7 +32,104 @@ export const MAX_SUGGESTED_FILES = 10;
36
32
  export const MIN_SUGGESTED_FILES = 5;
37
33
 
38
34
  /** Stopwords dropped during keyword tokenization (lowercase comparison). */
39
- const STOPWORDS: ReadonlySet<string> = new Set(["the", "a", "an", "and", "or", "to", "of", "in", "for", "on", "is", "are", "be", "with"]);
35
+ // PERF round 3: expanded from 14 function words to the common verb/pronoun/
36
+ // filler set. These multiply the scoring cost (keywords × files × passes)
37
+ // and essentially never appear in code file paths. Deliberately KEPT OUT:
38
+ // domain words that DO match paths — test, cache, prompt, tool, spec,
39
+ // artifact names like "smoke" — check the keep-assertions in R3-3 before adding.
40
+ const STOPWORDS: ReadonlySet<string> = new Set([
41
+ "the",
42
+ "a",
43
+ "an",
44
+ "and",
45
+ "or",
46
+ "to",
47
+ "of",
48
+ "in",
49
+ "for",
50
+ "on",
51
+ "is",
52
+ "are",
53
+ "be",
54
+ "with",
55
+ "this",
56
+ "that",
57
+ "these",
58
+ "those",
59
+ "then",
60
+ "than",
61
+ "so",
62
+ "if",
63
+ "but",
64
+ "not",
65
+ "no",
66
+ "yes",
67
+ "it",
68
+ "its",
69
+ "they",
70
+ "them",
71
+ "their",
72
+ "we",
73
+ "you",
74
+ "your",
75
+ "us",
76
+ "our",
77
+ "i",
78
+ "was",
79
+ "were",
80
+ "been",
81
+ "has",
82
+ "have",
83
+ "had",
84
+ "will",
85
+ "would",
86
+ "can",
87
+ "could",
88
+ "should",
89
+ "may",
90
+ "might",
91
+ "must",
92
+ "shall",
93
+ "do",
94
+ "does",
95
+ "did",
96
+ "done",
97
+ "find",
98
+ "found",
99
+ "look",
100
+ "run", // 'run' — generic verb, false-positives every *runner* path (see R3-3)
101
+ "likely",
102
+ "please",
103
+ "just",
104
+ "only",
105
+ "also",
106
+ "into",
107
+ "from",
108
+ "when",
109
+ "what",
110
+ "which",
111
+ "where",
112
+ "how",
113
+ "all",
114
+ "any",
115
+ "some",
116
+ "there",
117
+ "here",
118
+ "report",
119
+ "reports",
120
+ "exact",
121
+ "once",
122
+ "twice",
123
+ "things",
124
+ "thing",
125
+ "stuff",
126
+ "make",
127
+ "makes",
128
+ "made",
129
+ "use",
130
+ "using",
131
+ "used",
132
+ ]);
40
133
 
41
134
  /** File extensions considered relevant for retrieval. */
42
135
  const RELEVANT_EXTS: ReadonlySet<string> = new Set([
@@ -68,6 +161,37 @@ interface RipgrepAvailable {
68
161
 
69
162
  let cachedRgCheck: RipgrepAvailable | undefined;
70
163
 
164
+ /**
165
+ * PERF round 3: per-cwd cache of the rg discovery result (relative paths,
166
+ * post RELEVANT_EXTS filter). Tasks in one run share the cwd but differ in
167
+ * step.task keywords, so the stableIOCache in prompt-builder.ts misses per
168
+ * task — this cache keeps the expensive part (rg spawn + 77k-line parse)
169
+ * at once per cwd per minute instead of once per task. Fallback walk is
170
+ * NOT cached (its result depends on keywords). Size-capped, insertion-
171
+ * order eviction, same TTL family as stableIOCache (60s).
172
+ */
173
+ const DISCOVERED_TTL_MS = 60_000;
174
+ const DISCOVERED_CACHE_MAX = 32;
175
+ const discoveredCache = new Map<string, { files: string[]; at: number }>();
176
+
177
+ function getCachedDiscovered(cwd: string): string[] | undefined {
178
+ const hit = discoveredCache.get(cwd);
179
+ if (hit && Date.now() - hit.at < DISCOVERED_TTL_MS) {
180
+ // shared cached array — treat as read-only
181
+ return hit.files;
182
+ }
183
+ return undefined;
184
+ }
185
+
186
+ function storeDiscovered(cwd: string, files: string[]): void {
187
+ discoveredCache.set(cwd, { files, at: Date.now() });
188
+ while (discoveredCache.size > DISCOVERED_CACHE_MAX) {
189
+ const oldest = discoveredCache.keys().next().value;
190
+ if (oldest === undefined) break;
191
+ discoveredCache.delete(oldest);
192
+ }
193
+ }
194
+
71
195
  /**
72
196
  * Detect ripgrep availability once per process. Uses `rg --version` and
73
197
  * catches ENOENT or non-zero exit. Cached so the cost (one spawn) is
@@ -113,6 +237,11 @@ export function __test_resetRipgrepCache(): void {
113
237
  cachedRgCheck = undefined;
114
238
  }
115
239
 
240
+ /** @internal Test-only: reset the discovery cache. */
241
+ export function __test_resetDiscoveredCache(): void {
242
+ discoveredCache.clear();
243
+ }
244
+
116
245
  /**
117
246
  * Tokenize a task + goal string into lowercase keywords, dropping
118
247
  * stopwords. Single-letter tokens and pure-punctuation tokens are
@@ -272,9 +401,9 @@ async function walkFilesFallback(cwd: string, keywords: string[]): Promise<Array
272
401
  }
273
402
 
274
403
  /**
275
- * Iterate up to MAX_CYCLES times. Each cycle: discover files, score
276
- * them, check convergence. Stop early on convergence or when
277
- * shouldContinue returns false.
404
+ * Single discovery pass (perf round 3): discover files once, score
405
+ * them path-only, dedupe by absolute path, check convergence on the
406
+ * deduped evaluation set.
278
407
  */
279
408
  export async function runRetrievalCycle(task: string, goal: string, cwd: string): Promise<RetrievalResult> {
280
409
  const keywords = tokenizeQuery(task, goal);
@@ -284,55 +413,61 @@ export async function runRetrievalCycle(task: string, goal: string, cwd: string)
284
413
  const rg = await detectRipgrep();
285
414
  const useRg = rg.available;
286
415
  let usedFallback = !useRg;
287
- const evaluations: RelevanceEvaluation[] = [];
288
- let cycle = 0;
289
- let converged = false;
290
- for (; cycle < MAX_CYCLES; cycle++) {
291
- let discovered: string[] = [];
292
- try {
293
- if (useRg) {
294
- // Cycle 0: enumerate all relevant files via `rg --files`.
295
- // Later cycles: filter by keywords via `rg --files | rg pattern`.
296
- // We use the simpler `rg --files` + filter strategy because
297
- // `rg --json` parsing adds complexity for marginal gain.
298
- // `rg --files` respects .gitignore by default; we add an
299
- // explicit -g '!node_modules' and -g '!.git' to be safe on
300
- // repos that don't ignore them.
416
+ // PERF round 3 (2026-08-26): single discovery pass. The previous loop ran
417
+ // up to MAX_CYCLES=3 iterations, but each iteration re-ran `rg --files`
418
+ // (identical output ~0.36s/spawn on my_pi) and re-scored the identical
419
+ // ~57k-file set: path-only scoring (content always "") cannot reach
420
+ // HIGH_RELEVANCE_THRESHOLD=0.7 (observed max 0.64), so hasConverged was
421
+ // always false and the loop ran unconditionally — 3× CPU for a zero
422
+ // result delta (measured 7055ms → 1980ms cold on the my_pi monorepo,
423
+ // full-length real-run goal — see docs/real-test/reports/perf-round3-probe.md).
424
+ let discovered: string[] = [];
425
+ try {
426
+ if (useRg) {
427
+ const cached = getCachedDiscovered(cwd);
428
+ if (cached) {
429
+ discovered = cached;
430
+ } else {
431
+ // `rg --files` respects .gitignore by default; explicit -g guards
432
+ // repos that don't ignore them (comment moved from the loop body).
301
433
  const stdout = await runRipgrep(["--files", "-g", "!node_modules", "-g", "!.git", cwd], cwd);
302
434
  discovered = stdout
303
435
  .split("\n")
304
436
  .map((p) => p.trim())
305
437
  .filter((p) => p && RELEVANT_EXTS.has(path.extname(p).toLowerCase()))
306
438
  .map((p) => path.relative(cwd, p));
307
- } else {
308
- discovered = (await walkFilesFallback(cwd, keywords)).map((f) => f.path);
439
+ storeDiscovered(cwd, discovered);
309
440
  }
310
- } catch {
311
- // rg errored mid-run — switch to fallback for this cycle.
312
- usedFallback = true;
441
+ } else {
313
442
  discovered = (await walkFilesFallback(cwd, keywords)).map((f) => f.path);
314
443
  }
315
- // Score each discovered file. Path-only scoring (no file read) so
316
- // we don't slow down prompt building for hundreds of files.
317
- const seenInThisCycle = new Set<string>();
318
- for (const relPath of discovered) {
319
- if (seenInThisCycle.has(relPath)) continue;
320
- seenInThisCycle.add(relPath);
321
- const absPath = path.isAbsolute(relPath) ? relPath : path.join(cwd, relPath);
322
- const score = scoreRelevance(absPath, "", keywords);
323
- if (score > 0) {
324
- evaluations.push({
325
- path: absPath,
326
- relevance: score,
327
- reason: reasonFor(absPath, keywords),
328
- missingContext: [],
329
- });
330
- }
444
+ } catch {
445
+ // rg errored mid-run — switch to fallback for this pass.
446
+ usedFallback = true;
447
+ discovered = (await walkFilesFallback(cwd, keywords)).map((f) => f.path);
448
+ }
449
+ // Score each discovered file. Path-only scoring (no file read) so
450
+ // we don't slow down prompt building for hundreds of files.
451
+ // PERF round 3: dedupe by ABSOLUTE path — the multi-cycle accumulation
452
+ // previously pushed the same evaluation once per cycle, so the top-10
453
+ // could contain the same file up to 3 times (observed on
454
+ // team_20260826002634: task-output-context-dep-cache.test.ts ×3).
455
+ const byPath = new Map<string, RelevanceEvaluation>();
456
+ for (const relPath of discovered) {
457
+ const absPath = path.isAbsolute(relPath) ? relPath : path.join(cwd, relPath);
458
+ if (byPath.has(absPath)) continue;
459
+ const score = scoreRelevance(absPath, "", keywords);
460
+ if (score > 0) {
461
+ byPath.set(absPath, {
462
+ path: absPath,
463
+ relevance: score,
464
+ reason: reasonFor(absPath, keywords),
465
+ missingContext: [],
466
+ });
331
467
  }
332
- converged = hasConverged(evaluations);
333
- if (converged) break;
334
- if (!shouldContinue(evaluations, cycle)) break;
335
468
  }
469
+ const evaluations = [...byPath.values()];
470
+ const converged = hasConverged(evaluations);
336
471
  // Sort by score desc, take top N (5..10).
337
472
  evaluations.sort((a, b) => b.relevance - a.relevance);
338
473
  const cap = Math.min(MAX_SUGGESTED_FILES, Math.max(MIN_SUGGESTED_FILES, evaluations.length));
@@ -341,7 +476,7 @@ export async function runRetrievalCycle(task: string, goal: string, cwd: string)
341
476
  score: e.relevance,
342
477
  reason: e.reason,
343
478
  }));
344
- return { files: top, cycles: cycle, converged, usedFallback };
479
+ return { files: top, cycles: 1, converged, usedFallback };
345
480
  }
346
481
 
347
482
  /**
@@ -1,4 +1,5 @@
1
1
  import * as fs from "node:fs";
2
+ import { loadConfig } from "../../config/config.ts";
2
3
  import { flushPendingAtomicWrites } from "../../state/atomic-write.ts";
3
4
  import { withRunLockSync } from "../../state/coordination/locks.ts";
4
5
  import { loadRunManifestById, saveRunTasksCoalesced } from "../../state/stores/state-store.ts";
@@ -30,7 +31,11 @@ export function updateTask(tasks: TeamTaskState[], updated: TeamTaskState): Team
30
31
  * SIGKILL in that window loses the terminal update and crash recovery
31
32
  * would see "running" in tasks.json while events.jsonl already shows
32
33
  * "completed". For non-terminal transitions (heartbeat, progress) the
33
- * default buffered write is fine and matches prior behavior.
34
+ * default buffered write is fine and matches prior behavior. With the
35
+ * opt-in `persistence.skipTasksFsync` flag (default off), non-terminal
36
+ * checkpoints stay in the 50ms coalesce window but drop ONLY the fsync
37
+ * (durability "best-effort") — tasks.json is reconstructible from the
38
+ * fsync'd event log, so a crash loses at most the un-flushed tail.
34
39
  *
35
40
  * @param checkpointPhase - Optional checkpoint phase to include in the task state alongside the update.
36
41
  */
@@ -41,22 +46,16 @@ export function persistSingleTaskUpdate(
41
46
  checkpointPhase?: TaskCheckpointState["phase"],
42
47
  skipCoalesce: boolean = false,
43
48
  ): TeamTaskState[] {
44
- // H5 (2026-08-10): lowered from 100 → 10. Each retry does
45
- // flushPendingAtomicWrites (global) + loadRunManifestById (stat + parse)
46
- // + statSync, ~5ms each. Every attempt now loads from disk (BUG-028);
47
- // retries only fire under real contention from best-effort writers that
48
- // don't hold the run lock (async-notifier, crash-recovery). If 10 retries
49
- // cannot converge, the system is in a pathological state where 100 would
50
- // not help either — the explicit error below surfaces it instead of
51
- // blocking the event loop for 500ms.
49
+ // H5 (2026-08-10): lowered from 100 → 10. Each retry does a scoped
50
+ // flushPendingAtomicWrites(tasksPath) + loadRunManifestById (stat + parse;
51
+ // the manifest half is typically served from the manifest cache after the
52
+ // Task 12 reuse) + statSync, ~5ms each. Every attempt now loads from disk
53
+ // (BUG-028); retries only fire under real contention from best-effort
54
+ // writers that don't hold the run lock (async-notifier, crash-recovery).
55
+ // If 10 retries cannot converge, the system is in a pathological state
56
+ // where 100 would not help either — the explicit error below surfaces it
57
+ // instead of blocking the event loop for 500ms.
52
58
  const MAX_CAS_ATTEMPTS = 10;
53
- let baseMtime = 0;
54
- try {
55
- baseMtime = fs.statSync(manifest.tasksPath).mtimeMs;
56
- } catch {
57
- // File doesn't exist yet — baseMtime=0 means "anything is fine"
58
- baseMtime = 0;
59
- }
60
59
 
61
60
  let merged: TeamTaskState[] | undefined;
62
61
 
@@ -74,15 +73,25 @@ export function persistSingleTaskUpdate(
74
73
  try {
75
74
  return withRunLockSync(manifest, () => {
76
75
  for (let attempt = 0; attempt < MAX_CAS_ATTEMPTS; attempt++) {
77
- // F4: persistSingleTaskUpdate now uses saveRunTasksCoalesced below
78
- // (50ms debounce window). Read-modify-write loops are unsafe under
79
- // coalescing — a parallel writer's buffered write is invisible to
80
- // loadRunManifestById until it actually lands. Force any pending
81
- // coalesced writes to flush first so this read sees the latest
82
- // durable state. Without this guard, a parallel writer could
83
- // overwrite our buffered write between our load and our (async)
84
- // fsync, silently losing the intermediate update.
85
- flushPendingAtomicWrites();
76
+ // PERF (2026-08-24): scoped flush — only force OUR tasks.json pending
77
+ // write to land. The old argument-less call drained every pending
78
+ // coalesced write process-wide (other runs, agents.json 250ms window)
79
+ // on every persist (~30x/s), defeating coalescing globally. The F4
80
+ // invariant only needs tasks.json durable before this read.
81
+ flushPendingAtomicWrites(manifest.tasksPath);
82
+ // PERF (2026-08-24): capture the CAS baseline INSIDE the lock, after
83
+ // the flush, immediately before the load. The old pre-lock stat almost
84
+ // always disagreed with the in-lock stat under 10 parallel writers
85
+ // (mtime moved between function entry and lock acquisition), forcing
86
+ // 2+ full flush+load cycles per call. In-lock capture makes retry mean
87
+ // exactly: "a cross-process writer committed between our load and our
88
+ // pre-write stat" — the only race the CAS can actually catch.
89
+ let baseMtime: number;
90
+ try {
91
+ baseMtime = fs.statSync(manifest.tasksPath).mtimeMs;
92
+ } catch {
93
+ baseMtime = 0;
94
+ }
86
95
  // BUG-028 (2026-08-16): ALWAYS load the committed tasks from disk
87
96
  // inside the lock — never trust fallbackTasks on attempt 0. The
88
97
  // old F4 perf shortcut assumed "the caller already obtained the
@@ -95,8 +104,9 @@ export function persistSingleTaskUpdate(
95
104
  // (siblings still "running") over disk where siblings were already
96
105
  // terminal — resurrecting them and blocking finalize ("task is
97
106
  // still running"). The mtime CAS below cannot catch this: it only
98
- // detects writers between the entry stat and the in-lock stat,
99
- // i.e. staleness acquired AFTER function entry — not fallback
107
+ // detects writers between the in-lock baseline stat (captured
108
+ // after the flush, immediately before the load) and the pre-write
109
+ // stat, i.e. staleness acquired AFTER that baseline — not fallback
100
110
  // staleness that predates it. Loading disk here makes sibling
101
111
  // state authoritative (matching mergeUnitResult / bug-027 policy)
102
112
  // while `updated` still wins for THIS task via updateTask.
@@ -121,8 +131,8 @@ export function persistSingleTaskUpdate(
121
131
  }
122
132
 
123
133
  if (currentMtime !== baseMtime) {
124
- // Another writer committed — their update is in latest, re-merge on top
125
- baseMtime = currentMtime;
134
+ // Another writer committed between our in-lock baseline and this
135
+ // stat — retry; the next iteration recaptures the baseline fresh.
126
136
  continue;
127
137
  }
128
138
 
@@ -148,7 +158,21 @@ export function persistSingleTaskUpdate(
148
158
  // ST-7: terminal transitions (skipCoalesce=true) bypass the 50ms
149
159
  // coalesce window so a SIGKILL after the persist completes cannot
150
160
  // leave tasks.json stale with a non-terminal status.
151
- saveRunTasksCoalesced(manifest, merged, skipCoalesce);
161
+ // PERF round 2, Task 3 (opt-in, default off): for NON-terminal
162
+ // checkpoints — and ONLY when persistence.skipTasksFsync is true —
163
+ // KEEP the 50ms coalesce window (the RMW grouping benefit) and drop
164
+ // ONLY durability: the coalesced entry stores "best-effort" and the
165
+ // flush forwards it to atomicWriteFile (no data/parent-dir fsync).
166
+ // tasks.json is reconstructible from the fsync'd event log, so the
167
+ // crash tail is at most the in-flight checkpoint. Terminal
168
+ // transitions (skipCoalesce=true) always stay full-durability — the
169
+ // flag never touches that path.
170
+ const skipTasksFsync = !skipCoalesce && loadConfig().config.persistence?.skipTasksFsync === true;
171
+ if (skipTasksFsync) {
172
+ saveRunTasksCoalesced(manifest, merged, false, "best-effort");
173
+ } else {
174
+ saveRunTasksCoalesced(manifest, merged, skipCoalesce);
175
+ }
152
176
  } catch (err) {
153
177
  logInternalError("persistSingleTaskUpdate", err, undefined, "error");
154
178
  throw err;
@@ -118,7 +118,9 @@ export async function runTeamTask(input: TaskRunnerInput): Promise<{ manifest: T
118
118
  const skillArtifact = ctx.skillArtifact;
119
119
  const coordinationArtifact = ctx.coordinationArtifact;
120
120
 
121
- let resultArtifact: ArtifactDescriptor;
121
+ // MuxSurface degrade path returns NO result artifact — undefined until a
122
+ // branch produces one (finalizeTaskResult's surfaceLost branch ignores it).
123
+ let resultArtifact: ArtifactDescriptor | undefined;
122
124
  let logArtifact: ArtifactDescriptor | undefined;
123
125
  let transcriptArtifact: ArtifactDescriptor | undefined;
124
126
  let exitCode: number | null = 0;
@@ -138,7 +140,7 @@ export async function runTeamTask(input: TaskRunnerInput): Promise<{ manifest: T
138
140
  const child = await runChildProcessTask(ctx);
139
141
  task = ctx.task;
140
142
  tasks = ctx.tasks;
141
- resultArtifact = child.resultArtifact;
143
+ resultArtifact = child.resultArtifact ?? resultArtifact;
142
144
  logArtifact = child.logArtifact;
143
145
  transcriptArtifact = child.transcriptArtifact;
144
146
  exitCode = child.exitCode;
@@ -258,7 +258,22 @@ export {
258
258
  // Re-export the test-only helpers so existing test imports still resolve.
259
259
  export { __test__mergeTaskUpdates, __test__shouldMergeTaskUpdate } from "./merge-gate.ts";
260
260
 
261
+ import { getActiveBrokerRevoker } from "./broker/broker-issuer.ts";
261
262
  import { injectAdaptivePlanIfReady, isAdaptiveWorkflow } from "./goal-workflow/adaptive-plan.ts";
263
+ // MuxSurface A1 (spec §7 D3 + §8.3): run-scoped surface-degrade controller —
264
+ // child-pi layer notifies spawn/exit/degrade through the registry keyed by
265
+ // runId; this runner owns policy (lockout), persistence (manifest.surface) and
266
+ // the headless re-dispatch of degraded units.
267
+ import {
268
+ clearSurfaceRuntimeController,
269
+ createSurfaceRuntimeController,
270
+ normalizeSurfaceState,
271
+ planHeadlessRedeplays,
272
+ registerSurfaceRuntimeController,
273
+ } from "./surface/degrade.ts";
274
+ // Task 5 (tab-layout §5): run end đóng toàn tab của run qua provider singleton
275
+ // cùng instance mà spawn đã dùng (map tab nội bộ sống trong provider).
276
+ import { surfaceProviderForCleanup } from "./surface/resolve-surface.ts";
262
277
 
263
278
  // formatTaskProgress / runEffectivenessLines / scratchpadSummaryLines /
264
279
  // lastProgressContentHash / writeProgress moved to ./finalize-run.ts (2026-08
@@ -862,6 +877,32 @@ async function executeTeamRunCore(
862
877
  // task-output-context.ts.
863
878
  const resultReadCache = createResultArtifactReadCache();
864
879
 
880
+ // ── MuxSurface A1 (spec §7 D3): per-run degrade controller ─────────────
881
+ // Owned here because only this runner may mutate manifest/tasks/events; the
882
+ // child-pi surface branch reaches it through the registry keyed by runId.
883
+ // Broker token revocation resolves lazily AT degrade time (the broker can
884
+ // start mid-run on the first credential request).
885
+ const surfaceController = createSurfaceRuntimeController({
886
+ runId: manifest.runId,
887
+ eventsPath: manifest.eventsPath,
888
+ // F2 (fix round 1): resume sau host restart phải kế thừa lockout +
889
+ // workerPids/sessionPaths/panes đã ghi trên manifest, nếu không evidence
890
+ // của nửa đầu run bị mất khi controller mới khởi động trống.
891
+ initialState: manifest.surface,
892
+ revoke: (taskId) => getActiveBrokerRevoker()?.(taskId),
893
+ });
894
+ registerSurfaceRuntimeController(surfaceController);
895
+ // Re-dispatch-once guard: a degraded task is replayed headless exactly once
896
+ // per run — repeated loss lands in needs_attention for a human instead of a
897
+ // mux-flap respawn loop (spec §7 anti-flap).
898
+ const surfaceLossHandled = new Set<string>();
899
+ // Pure merge of the controller snapshot onto ANY manifest view (callers pass
900
+ // either the closure local or ctx.manifest so the freshest state wins).
901
+ const attachSurfaceSnapshot = (target: TeamRunManifest): TeamRunManifest => ({
902
+ ...target,
903
+ surface: normalizeSurfaceState(surfaceController.snapshot()),
904
+ });
905
+
865
906
  // CORE-4: scheduler context — mutable state bag for extracted scheduler
866
907
  // functions. Fields are synced from closure locals at the top of each
867
908
  // loop iteration; extracted functions mutate ctx in-place.
@@ -984,6 +1025,33 @@ async function executeTeamRunCore(
984
1025
  // (cancel-during-exec check + batch summary artifact).
985
1026
  const { taskIds: settledTaskIds, result: resultToMerge } = ctx.settledMerge!;
986
1027
 
1028
+ // ── MuxSurface A1 (spec §7 steps 4–5): drain degrades → re-dispatch
1029
+ // headless. Runs BEFORE phase advance so a degraded task is `queued`
1030
+ // again while the scheduler still sees this tick's state, and before
1031
+ // handleFailedTask can ever observe needs_attention leftovers.
1032
+ const surfaceDegraded = surfaceController.takeDegraded();
1033
+ if (surfaceDegraded.length > 0) {
1034
+ const replay = planHeadlessRedeplays({
1035
+ tasks: ctx.tasks,
1036
+ degraded: surfaceDegraded,
1037
+ handledTaskIds: surfaceLossHandled,
1038
+ });
1039
+ ctx.tasks = replay.tasks;
1040
+ tasks = ctx.tasks;
1041
+ if (replay.requeuedTaskIds.length > 0) {
1042
+ await appendEventAsync(manifest.eventsPath, {
1043
+ type: "surface.requeued",
1044
+ runId: manifest.runId,
1045
+ message: `Re-dispatched ${replay.requeuedTaskIds.length} surface-degraded task(s) headless: ${replay.requeuedTaskIds.join(", ")}`,
1046
+ data: {
1047
+ taskIds: replay.requeuedTaskIds,
1048
+ skipped: replay.skipped,
1049
+ resumeComponents: ["rendered-prompt", "scratchpad-restore", "pendingSteers-replay", "resume-note"],
1050
+ },
1051
+ });
1052
+ }
1053
+ }
1054
+
987
1055
  // CORE-4 extraction 6: workflow phase advance. ctx.wfMachine is
988
1056
  // already synced from the top-of-loop sync (RT-15);
989
1057
  // advanceWorkflowPhases advances phases whose tasks are all terminal,
@@ -1089,6 +1157,11 @@ async function executeTeamRunCore(
1089
1157
  ...(groupDelivery?.artifact ? [groupDelivery.artifact] : []),
1090
1158
  ]),
1091
1159
  };
1160
+ // MuxSurface A1 (§8.3): persist the run-scoped pane/pid/lockout snapshot
1161
+ // with the batch manifest write — panes recorded this tick become
1162
+ // visible to doctor/zombie-sweep and a crash mid-run leaves at most
1163
+ // one tick of stale pane records.
1164
+ manifest = attachSurfaceSnapshot(manifest);
1092
1165
  manifest = writeProgress(manifest, tasks, "team-runner", input.executeWorkers, input.runtimeConfig);
1093
1166
  await saveRunManifestAsync(manifest);
1094
1167
  }
@@ -1098,11 +1171,39 @@ async function executeTeamRunCore(
1098
1171
  // + health snapshot, performs the joint atomic manifest+tasks save, and
1099
1172
  // returns the terminal { manifest, tasks } result. Sync the locals back
1100
1173
  // from ctx so the finally block observes consistent state.
1174
+ // Last degrade drain: a pane that died during the closeout still gets its
1175
+ // headless replay attempt if any scheduler work remains, else it stays
1176
+ // needs_attention for resume — never silently dropped from the manifest.
1177
+ const finalDegraded = surfaceController.takeDegraded();
1178
+ if (finalDegraded.length > 0) {
1179
+ const finalReplay = planHeadlessRedeplays({
1180
+ tasks: ctx.tasks,
1181
+ degraded: finalDegraded,
1182
+ handledTaskIds: surfaceLossHandled,
1183
+ });
1184
+ ctx.tasks = finalReplay.tasks;
1185
+ }
1186
+ ctx.manifest = attachSurfaceSnapshot(ctx.manifest);
1101
1187
  const finalResult = await finalizeRun(ctx);
1102
1188
  manifest = ctx.manifest;
1103
1189
  tasks = ctx.tasks;
1104
1190
  return finalResult;
1105
1191
  } finally {
1192
+ // Task 5 (tab-layout §5): tab chỉ đóng khi RUN END — mọi đường thoát
1193
+ // (completed/failed/cancelled/throw) đều đóng toàn tab của run qua
1194
+ // provider singleton cùng instance mà spawn đã dùng. Best-effort: lỗi đã
1195
+ // được log bên trong closeTabForRun, teardown không được sập vì mux chết.
1196
+ const surfaceProviderKind = surfaceController.snapshot().provider;
1197
+ if (surfaceProviderKind) {
1198
+ try {
1199
+ await surfaceController.closeRunTabs(surfaceProviderForCleanup(surfaceProviderKind));
1200
+ } catch (error) {
1201
+ logInternalError("team-runner.close-run-tabs", error, `runId=${manifest.runId}`);
1202
+ }
1203
+ }
1204
+ // MuxSurface A1: drop the run's registry entry FIRST — no in-flight
1205
+ // notify after teardown may resurrect policy state into the next run.
1206
+ clearSurfaceRuntimeController(manifest.runId);
1106
1207
  // #3: drainPendingUnits returns settled outcomes, but the finally block
1107
1208
  // only needs the drain side-effect (abort + await + clear); the return
1108
1209
  // value is intentionally unused here.
@@ -95,6 +95,19 @@ export const PiTeamsRuntimeConfigSchema = Type.Object(
95
95
  ),
96
96
  excludeContextBash: Type.Optional(Type.Boolean()),
97
97
  agentExtensions: Type.Optional(Type.Array(Type.String({ minLength: 1 }), { sensitive: true })),
98
+ // Mux-surface policy (spec v0.7 §8.2.4): no sensitive marks — surface
99
+ // picks where worker processes live, not what they may do.
100
+ surface: Type.Optional(
101
+ Type.Object(
102
+ {
103
+ mode: Type.Optional(
104
+ Type.Union([Type.Literal("auto"), Type.Literal("tmux"), Type.Literal("herdr"), Type.Literal("off")]),
105
+ ),
106
+ visibleAgents: Type.Optional(Type.Array(Type.String({ minLength: 1 }))),
107
+ },
108
+ { additionalProperties: false },
109
+ ),
110
+ ),
98
111
  isolationPolicy: Type.Optional(
99
112
  Type.Object(
100
113
  {
@@ -341,6 +354,16 @@ export const PiTeamsNestingConfigSchema = Type.Object(
341
354
  { additionalProperties: false },
342
355
  );
343
356
 
357
+ /** State-layer persistence schema (perf round 2, Task 3). Opt-in only.
358
+ * Does affect write cost, but never correctness: tasks.json is
359
+ * reconstructible from the fsync'd event log, so no privilege surface. */
360
+ export const PiTeamsPersistenceConfigSchema = Type.Object(
361
+ {
362
+ skipTasksFsync: Type.Optional(Type.Boolean()),
363
+ },
364
+ { additionalProperties: false },
365
+ );
366
+
344
367
  export const PiTeamsConfigSchema = Type.Object(
345
368
  {
346
369
  asyncByDefault: Type.Optional(Type.Boolean({ sensitive: true })),
@@ -365,6 +388,7 @@ export const PiTeamsConfigSchema = Type.Object(
365
388
  ui: Type.Optional(PiTeamsUiConfigSchema),
366
389
  broker: Type.Optional(CrewBrokerConfigSchema),
367
390
  nesting: Type.Optional(PiTeamsNestingConfigSchema),
391
+ persistence: Type.Optional(PiTeamsPersistenceConfigSchema),
368
392
  },
369
393
  { additionalProperties: false },
370
394
  );