pi-crew 0.9.34 → 0.9.35

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,520 @@
1
+ # PI-CREW PERFORMANCE OPTIMIZATIONS: DETAILED EXECUTION PLAN
2
+
3
+ **Created:** 2026-07-13
4
+ **Based on:** Performance review findings and planner output
5
+ **Scope:** 8 performance optimizations across 4 phases
6
+
7
+ ---
8
+
9
+ ## Executive Summary
10
+
11
+ This plan details the execution of 8 performance optimizations for pi-crew, organized into 4 phases with clear dependency chains, parallelization opportunities, and risk assessment. The optimizations range from quick wins (0.5 day) to major architectural changes (5-8 days), with a total estimated timeline of 4 weeks with 2 developers or 2 weeks with 4 developers.
12
+
13
+ ---
14
+
15
+ ## Phase 1: Quick Wins (Week 1)
16
+
17
+ **Goal:** Low-risk, high-impact optimizations that can be done in parallel
18
+ **Total Effort:** 3-5 days (parallel)
19
+
20
+ ### OPT-03: Guard collectedJsonEvents (0.5 day)
21
+
22
+ **Current State:**
23
+ - `task-runner.ts:301`: Already guarded with conditional allocation
24
+ - `live-session-runtime.ts:597`: Still allocates unconditionally
25
+
26
+ **Implementation:**
27
+ 1. Apply same pattern from `task-runner.ts` to `live-session-runtime.ts`
28
+ 2. Check yield configuration before allocating array
29
+ 3. Update all usage sites to handle `undefined` case
30
+
31
+ **Files to Modify:**
32
+ - `src/runtime/live-session-runtime.ts` (line 597)
33
+ - `src/runtime/task-runner.ts` (verify existing implementation)
34
+
35
+ **Validation:**
36
+ - Unit test: verify no allocation when yield disabled
37
+ - Memory profile: compare before/after for 100-task run
38
+ - Verify no behavioral changes
39
+
40
+ ### OPT-05: Cache compacted skill content (0.5 day)
41
+
42
+ **Current State:**
43
+ - `skill-instructions.ts:144`: Cache already exists with `SKILL_CACHE_MAX_ENTRIES = 128`
44
+ - `skill-instructions.ts:132`: Compacted content already cached
45
+
46
+ **Implementation:**
47
+ 1. Verify cache hit rates in production
48
+ 2. Tune `SKILL_CACHE_MAX_ENTRIES` if needed
49
+ 3. Consider cross-run cache (like `stableIOCache` in prompt-builder.ts)
50
+
51
+ **Files to Modify:**
52
+ - `src/runtime/skill-instructions.ts` (cache tuning)
53
+
54
+ **Validation:**
55
+ - Add cache hit/miss metrics
56
+ - Verify with 10-run session that cache hit rate > 80%
57
+ - Benchmark render time with/without cache
58
+
59
+ ### OPT-02: Convert saveRunManifest to async (1-2 days)
60
+
61
+ **Current State:**
62
+ - `state/state-store.ts:332`: Synchronous `saveRunManifest` with sync file I/O
63
+ - `state/state-store.ts:373`: Async `saveRunManifestAsync` exists but not widely used
64
+ - Multiple callers still use sync version in critical paths
65
+
66
+ **Implementation:**
67
+ 1. Audit all `saveRunManifest` call sites
68
+ 2. Convert critical path callers to use async version
69
+ 3. Verify no sync-only callers remain
70
+ 4. Update `task-runner.ts` line 1205 to use async version
71
+
72
+ **Files to Modify:**
73
+ - `src/runtime/team-runner.ts` (lines 640, 748, 776)
74
+ - `src/runtime/task-runner.ts` (line 1205)
75
+ - `src/runtime/stale-reconciler.ts` (lines 77, 332)
76
+ - `src/runtime/adaptive-plan.ts` (line 445)
77
+ - `src/runtime/background-runner.ts` (lines 624, 645, 690)
78
+ - `src/extension/team-tool.ts` (lines 305, 375)
79
+ - `src/extension/team-tool/goal-wrap.ts` (line 271)
80
+ - `src/extension/team-tool/api.ts` (lines 172, 247)
81
+ - `src/extension/team-tool/goal.ts` (line 194)
82
+
83
+ **Validation:**
84
+ - Grep for all `saveRunManifest` calls, ensure none remain sync
85
+ - Unit test: async save with concurrent reads
86
+ - Integration test: run with `--trace-warnings` to detect event loop blocking
87
+ - Verify no race conditions in concurrent writes
88
+
89
+ ### OPT-06: Optimize transcript reads (1-2 days)
90
+
91
+ **Current State:**
92
+ - `child-pi.ts:413`: Uses sync `writeSync` for transcript writes
93
+ - `live-session-runtime.ts:152`: Uses sync `appendFileSync`
94
+
95
+ **Implementation:**
96
+ 1. Convert `appendTranscript` to use async `writeFile`
97
+ 2. Keep fd open per-task instead of open/write/close per line
98
+ 3. Add path validation caching (like OPT-05 pattern)
99
+ 4. Ensure transcript writes are best-effort (loss acceptable)
100
+
101
+ **Files to Modify:**
102
+ - `src/runtime/child-pi.ts` (line 413)
103
+ - `src/runtime/live-session-runtime.ts` (line 152)
104
+ - `src/runtime/task-runner.ts` (line 414)
105
+
106
+ **Validation:**
107
+ - Benchmark: 1000-line transcript write time
108
+ - Verify no event loop blocking with `--trace-warnings`
109
+ - Test transcript integrity after async writes
110
+
111
+ ---
112
+
113
+ ## Phase 2: Independent Optimizations (Week 2)
114
+
115
+ **Goal:** Medium-effort optimizations that don't depend on Phase 1
116
+ **Total Effort:** 5-8 days (parallel)
117
+
118
+ ### OPT-04: Code-aware token estimation (2-3 days)
119
+
120
+ **Current State:**
121
+ - `utils/token-counter.ts:46-80`: Uses alpha/4 + punct heuristic
122
+ - Already better than naive char/4, but still ~10-15% off for code
123
+
124
+ **Implementation:**
125
+ 1. Add language detection (heuristic: look for keywords like `function`, `const`, `=>`)
126
+ 2. Apply different weighting for code vs prose
127
+ 3. Benchmark against real tokenizer outputs
128
+ 4. Update `tool-output-pruner.ts` to use new estimator
129
+
130
+ **Files to Modify:**
131
+ - `src/utils/token-counter.ts` (lines 46-80)
132
+ - `src/runtime/tool-output-pruner.ts` (update to use new estimator)
133
+
134
+ **Validation:**
135
+ - Benchmark against tiktoken/actual tokenizer for 100 code samples
136
+ - Ensure no regression for existing prose estimation
137
+ - Update `tool-output-pruner.ts` to use new estimator
138
+ - Add unit tests for code vs prose detection
139
+
140
+ ### OPT-01: Streaming dispatch (3-5 days)
141
+
142
+ **Current State:**
143
+ - `team-runner.ts:1265-1340`: Tasks grouped into `readyBatch`, then dispatched via `mapConcurrent`
144
+ - Loop waits for ALL tasks in a batch to complete before checking for new ready tasks
145
+ - Creates unnecessary latency when tasks finish at different times
146
+
147
+ **Implementation:**
148
+ 1. Refactor main execution loop in `executeTeamRunCore`
149
+ 2. Implement task completion events to trigger immediate dispatch
150
+ 3. Maintain correct ordering for dependency graph
151
+ 4. Handle race conditions in concurrent dispatch + completion callbacks
152
+ 5. Ensure proper state serialization
153
+
154
+ **Files to Modify:**
155
+ - `src/runtime/team-runner.ts` (main execution loop)
156
+ - `src/runtime/task-graph-scheduler.ts` (readiness updates)
157
+ - `src/runtime/parallel-utils.ts` (concurrent dispatch)
158
+
159
+ **Validation:**
160
+ - Unit test: DAG scheduler with dynamic readiness updates
161
+ - Integration test: 10-task workflow with varying task durations
162
+ - Measure: Time from first task ready to all tasks completed
163
+ - Stress test: 100 concurrent tasks with complex dependencies
164
+
165
+ ---
166
+
167
+ ## Phase 3: Structural Changes (Week 3-4)
168
+
169
+ **Goal:** High-effort architectural changes that depend on earlier phases
170
+ **Total Effort:** 8-13 days (parallel)
171
+
172
+ ### OPT-08: In-memory task state (5-8 days)
173
+
174
+ **Dependencies:** Requires OPT-02 (async saveRunManifest)
175
+
176
+ **Current State:**
177
+ - `state-store.ts`: Every state change reads full manifest + tasks from disk
178
+ - `task-runner/state-helpers.ts:29-141`: `persistSingleTaskUpdate` does 100-iteration sync CAS loop
179
+ - Performance review F1/F2/F4 all relate to this issue
180
+
181
+ **Implementation:**
182
+ 1. Design in-memory state model with write-through
183
+ 2. Implement CAS (compare-and-swap) for concurrent updates
184
+ 3. Add crash recovery for in-memory state
185
+ 4. Migrate all callers to new API
186
+ 5. Ensure bulletproof crash recovery
187
+
188
+ **Files to Modify:**
189
+ - `src/state/state-store.ts` (in-memory state model)
190
+ - `src/runtime/task-runner/state-helpers.ts` (CAS implementation)
191
+ - `src/runtime/team-runner.ts` (state persistence)
192
+ - `src/runtime/task-runner.ts` (state updates)
193
+
194
+ **Validation:**
195
+ - Stress test: 100 concurrent tasks writing state
196
+ - Kill test: random process termination during write
197
+ - Verify no data loss after crash recovery
198
+ - Benchmark: state update latency before/after
199
+
200
+ ### OPT-07: Live-session migration (3-5 days)
201
+
202
+ **Dependencies:** Requires OPT-01 (streaming dispatch)
203
+
204
+ **Current State:**
205
+ - `live-session-runtime.ts`: Separate execution path from child-process
206
+ - Has its own `collectedJsonEvents` allocation (see OPT-03)
207
+ - Different state management than child-process path
208
+
209
+ **Implementation:**
210
+ 1. Align live-session with new streaming dispatch architecture
211
+ 2. Refactor `live-session-runtime.ts` to use shared patterns
212
+ 3. Ensure behavioral parity with child-process path
213
+ 4. Extensive integration testing
214
+
215
+ **Files to Modify:**
216
+ - `src/runtime/live-session-runtime.ts` (major refactoring)
217
+ - `src/runtime/task-runner/live-executor.ts` (alignment)
218
+ - `src/runtime/team-runner.ts` (integration)
219
+
220
+ **Validation:**
221
+ - Full integration test suite for live-session
222
+ - Manual testing with real Pi session
223
+ - Performance comparison before/after
224
+ - Verify no breaking changes
225
+
226
+ ---
227
+
228
+ ## Dependency Graph
229
+
230
+ ```
231
+ Phase 1 (Week 1):
232
+ OPT-03 ─┐
233
+ OPT-05 ─┤
234
+ OPT-02 ─┼── (no dependencies)
235
+ OPT-06 ─┘
236
+
237
+ Phase 2 (Week 2):
238
+ OPT-04 ─── (no dependencies)
239
+ OPT-01 ─── (no dependencies)
240
+
241
+ Phase 3 (Week 3-4):
242
+ OPT-08 ─── depends on OPT-02
243
+ OPT-07 ─── depends on OPT-01
244
+ ```
245
+
246
+ ---
247
+
248
+ ## Risk Assessment
249
+
250
+ | Optimization | Risk Level | Mitigation |
251
+ |--------------|------------|------------|
252
+ | OPT-01 | MEDIUM | Extensive integration testing, gradual rollout |
253
+ | OPT-02 | LOW | Incremental migration, existing async version |
254
+ | OPT-03 | VERY LOW | Pattern already proven |
255
+ | OPT-04 | LOW | Additive change, benchmark validation |
256
+ | OPT-05 | VERY LOW | Already implemented, just tuning |
257
+ | OPT-06 | LOW | Best-effort writes, no consistency requirements |
258
+ | OPT-07 | HIGH | Extensive testing, manual validation |
259
+ | OPT-08 | VERY HIGH | Stress testing, crash recovery validation |
260
+
261
+ ---
262
+
263
+ ## Potential Conflicts
264
+
265
+ 1. **OPT-01 vs OPT-07:** Both affect task dispatch. Ensure streaming dispatch design accommodates live-session runtime.
266
+
267
+ 2. **OPT-02 vs OPT-08:** Both affect state persistence. Ensure async save works with in-memory write-through.
268
+
269
+ 3. **OPT-04 vs existing callers:** Token estimation is used in `tool-output-pruner.ts`. Ensure new estimator doesn't break pruning logic.
270
+
271
+ 4. **Cross-cutting concerns:** All optimizations must maintain cross-platform compatibility (Windows/macOS/Linux).
272
+
273
+ ---
274
+
275
+ ## Parallelization Opportunities
276
+
277
+ **Maximum Parallelism:**
278
+ - Phase 1: 4 developers can work simultaneously
279
+ - Phase 2: 2 developers can work simultaneously
280
+ - Phase 3: 2 developers can work simultaneously
281
+
282
+ **Critical Path:** OPT-02 → OPT-08 (8-10 days) or OPT-01 → OPT-07 (6-10 days)
283
+
284
+ **Total Timeline:** 4 weeks with 2 developers, 2 weeks with 4 developers
285
+
286
+ ---
287
+
288
+ ## Validation Strategy
289
+
290
+ ### Unit Tests
291
+ Each optimization should have unit tests for the specific change:
292
+ - OPT-03: Test conditional allocation
293
+ - OPT-05: Test cache hit/miss metrics
294
+ - OPT-02: Test async save with concurrent reads
295
+ - OPT-06: Test async transcript writes
296
+ - OPT-04: Test code vs prose detection
297
+ - OPT-01: Test DAG scheduler with dynamic readiness
298
+ - OPT-08: Test CAS operations
299
+ - OPT-07: Test live-session integration
300
+
301
+ ### Integration Tests
302
+ Full workflow tests before/after each phase:
303
+ - Run complete team workflows
304
+ - Verify state persistence
305
+ - Test error handling and recovery
306
+
307
+ ### Performance Tests
308
+ - Micro-benchmarks for individual functions
309
+ - End-to-end benchmarks for 10/50/100-task runs
310
+ - Event loop blocking detection (`--trace-warnings`)
311
+ - Memory usage profiling
312
+
313
+ ### Stress Tests
314
+ - OPT-08: Concurrent writes + crash recovery
315
+ - OPT-01: High concurrency with complex dependencies
316
+ - All optimizations: Long-running sessions
317
+
318
+ ### Manual Testing
319
+ - OPT-07: Real Pi session testing
320
+ - All optimizations: Cross-platform verification
321
+
322
+ ---
323
+
324
+ ## Success Metrics
325
+
326
+ 1. **Event Loop Blocking:** Reduce from ~50ms/task to <5ms/task
327
+ 2. **Task Dispatch Latency:** Reduce from batch-complete to task-ready (target: 50% reduction)
328
+ 3. **Token Estimation Accuracy:** Improve from ±15% to ±5%
329
+ 4. **Cache Hit Rates:** Skill cache >80%, config cache >90%
330
+ 5. **Memory Usage:** Stable for long runs (no linear growth)
331
+ 6. **State Update Latency:** Reduce from disk-read to in-memory (target: 10x improvement)
332
+
333
+ ---
334
+
335
+ ## Recommendations
336
+
337
+ 1. **Start with Phase 1:** Quick wins provide immediate value with minimal risk
338
+ 2. **Defer OPT-08:** In-memory task state is high-risk; consider as future project
339
+ 3. **Monitor OPT-07:** Live-session migration is complex; may need to split into sub-tasks
340
+ 4. **Benchmark everything:** Establish baseline before starting, measure after each phase
341
+ 5. **Cross-platform testing:** Ensure all changes work on Windows/macOS/Linux
342
+ 6. **Gradual rollout:** Implement optimizations incrementally, not all at once
343
+
344
+ ---
345
+
346
+ ## Files to Modify (Summary)
347
+
348
+ ### Phase 1:
349
+ - `src/runtime/live-session-runtime.ts` (OPT-03)
350
+ - `src/runtime/skill-instructions.ts` (OPT-05)
351
+ - `src/state/state-store.ts` (OPT-02)
352
+ - `src/runtime/team-runner.ts` (OPT-02)
353
+ - `src/runtime/task-runner.ts` (OPT-02)
354
+ - `src/runtime/stale-reconciler.ts` (OPT-02)
355
+ - `src/runtime/adaptive-plan.ts` (OPT-02)
356
+ - `src/runtime/background-runner.ts` (OPT-02)
357
+ - `src/extension/team-tool.ts` (OPT-02)
358
+ - `src/extension/team-tool/goal-wrap.ts` (OPT-02)
359
+ - `src/extension/team-tool/api.ts` (OPT-02)
360
+ - `src/extension/team-tool/goal.ts` (OPT-02)
361
+ - `src/runtime/child-pi.ts` (OPT-06)
362
+ - `src/runtime/task-runner.ts` (OPT-06)
363
+
364
+ ### Phase 2:
365
+ - `src/utils/token-counter.ts` (OPT-04)
366
+ - `src/runtime/tool-output-pruner.ts` (OPT-04)
367
+ - `src/runtime/team-runner.ts` (OPT-01)
368
+ - `src/runtime/task-graph-scheduler.ts` (OPT-01)
369
+ - `src/runtime/parallel-utils.ts` (OPT-01)
370
+
371
+ ### Phase 3:
372
+ - `src/state/state-store.ts` (OPT-08)
373
+ - `src/runtime/task-runner/state-helpers.ts` (OPT-08)
374
+ - `src/runtime/live-session-runtime.ts` (OPT-07)
375
+ - `src/runtime/task-runner/live-executor.ts` (OPT-07)
376
+ - `src/runtime/team-runner.ts` (OPT-07)
377
+
378
+ ---
379
+
380
+ ## Remaining Risks
381
+
382
+ 1. **OPT-01 complexity:** Streaming dispatch may introduce subtle race conditions
383
+ 2. **OPT-08 crash recovery:** In-memory state must survive process crashes
384
+ 3. **OPT-07 behavioral parity:** Live-session must behave identically to child-process
385
+ 4. **Cross-platform:** Ensure all changes work on Windows/macOS/Linux
386
+ 5. **Performance regression:** New optimizations must not introduce performance regressions
387
+ 6. **Breaking changes:** All optimizations must maintain backward compatibility
388
+
389
+ ---
390
+
391
+ ## Next Steps
392
+
393
+ 1. Review this plan with the team
394
+ 2. Assign owners for each optimization
395
+ 3. Create detailed design documents for Phase 2 and Phase 3
396
+ 4. Establish baseline metrics before starting
397
+ 5. Begin Phase 1 execution
398
+ 6. Set up performance monitoring and alerting
399
+ 7. Create rollback procedures for each optimization
400
+
401
+ ---
402
+
403
+ ## Appendix: Detailed Technical Specifications
404
+
405
+ ### OPT-01: Streaming Dispatch Technical Details
406
+
407
+ **Current Architecture:**
408
+ ```
409
+ while (tasks.some(queued)) {
410
+ readyBatch = getReadyTasks(tasks);
411
+ if (readyBatch.length === 0) break;
412
+ results = await mapConcurrent(readyBatch, ...);
413
+ tasks = mergeResults(tasks, results);
414
+ }
415
+ ```
416
+
417
+ **Proposed Architecture:**
418
+ ```
419
+ const pendingPromises = new Map();
420
+ while (tasks.some(queued) || pendingPromises.size > 0) {
421
+ readyBatch = getReadyTasks(tasks);
422
+ for (task of readyBatch) {
423
+ const promise = runTask(task).then(result => {
424
+ pendingPromises.delete(task.id);
425
+ tasks = mergeResult(tasks, result);
426
+ // Trigger next iteration
427
+ });
428
+ pendingPromises.set(task.id, promise);
429
+ }
430
+ await Promise.race([...pendingPromises.values()]);
431
+ }
432
+ ```
433
+
434
+ ### OPT-02: Async Migration Strategy
435
+
436
+ **Migration Pattern:**
437
+ ```typescript
438
+ // Before
439
+ saveRunManifest(manifest);
440
+
441
+ // After
442
+ await saveRunManifestAsync(manifest);
443
+ ```
444
+
445
+ **Critical Path Analysis:**
446
+ - `team-runner.ts:640` - Plan approval
447
+ - `team-runner.ts:748` - Budget persistence
448
+ - `team-runner.ts:776` - Goal achievement
449
+ - `task-runner.ts:1205` - Task completion
450
+
451
+ ### OPT-03: Guard Pattern
452
+
453
+ **Current Implementation:**
454
+ ```typescript
455
+ const collectedJsonEvents: Record<string, unknown>[] = [];
456
+ ```
457
+
458
+ **Proposed Implementation:**
459
+ ```typescript
460
+ const collectYieldEvents = runtimeKind !== "child-process" &&
461
+ (input.runtimeConfig?.yield?.enabled ?? DEFAULT_YIELD_CONFIG.enabled);
462
+ const collectedJsonEvents: Record<string, unknown>[] | undefined =
463
+ collectYieldEvents ? [] : undefined;
464
+ ```
465
+
466
+ ### OPT-04: Code-Aware Estimation
467
+
468
+ **Heuristic Detection:**
469
+ ```typescript
470
+ function isCodeContent(text: string): boolean {
471
+ const codeIndicators = [
472
+ /function\s*\w*\s*\(/,
473
+ /const\s+\w+\s*=/,
474
+ /=>\s*{/,
475
+ /import\s+.*from\s+['"]/,
476
+ /export\s+(default\s+)?(function|class|const)/,
477
+ /\w+\.\w+\(.*\)/, // Method calls
478
+ /[{}\[\]();]/, // Code punctuation
479
+ ];
480
+
481
+ let score = 0;
482
+ for (const indicator of codeIndicators) {
483
+ if (indicator.test(text)) score++;
484
+ }
485
+
486
+ return score >= 3; // Threshold for code detection
487
+ }
488
+ ```
489
+
490
+ ### OPT-08: In-Memory State Model
491
+
492
+ **Architecture:**
493
+ ```typescript
494
+ class InMemoryTaskState {
495
+ private state: Map<string, TeamTaskState> = new Map();
496
+ private manifest: TeamRunManifest;
497
+
498
+ async updateTask(taskId: string, update: Partial<TeamTaskState>): Promise<void> {
499
+ const current = this.state.get(taskId);
500
+ if (!current) throw new Error(`Task ${taskId} not found`);
501
+
502
+ const merged = { ...current, ...update };
503
+ this.state.set(taskId, merged);
504
+
505
+ // Write-through to disk
506
+ await this.flushToDisk();
507
+ }
508
+
509
+ private async flushToDisk(): Promise<void> {
510
+ await saveRunManifestAsync(this.manifest);
511
+ await saveRunTasksAsync(this.manifest, Array.from(this.state.values()));
512
+ }
513
+ }
514
+ ```
515
+
516
+ ---
517
+
518
+ **Document Version:** 1.0
519
+ **Last Updated:** 2026-07-13
520
+ **Author:** Performance Optimization Task Force
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-crew",
3
- "version": "0.9.34",
3
+ "version": "0.9.35",
4
4
  "description": "Pi extension for coordinated AI teams, workflows, worktrees, and async task orchestration",
5
5
  "author": "baphuongna",
6
6
  "license": "MIT",
@@ -11,7 +11,7 @@ import { touchWorkerHeartbeat } from "../../runtime/worker-heartbeat.ts";
11
11
  import type { TeamToolParamsValue } from "../../schema/team-tool-schema.ts";
12
12
  import { canTransitionTaskStatus, isTeamTaskStatus } from "../../state/contracts.ts";
13
13
  import { appendEvent, readEvents, readEventsCursor } from "../../state/event-log.ts";
14
- import { withRunLockSync } from "../../state/locks.ts";
14
+ import { withRunLock, withRunLockSync } from "../../state/locks.ts";
15
15
  import {
16
16
  acknowledgeMailboxMessage,
17
17
  appendFollowUpMessage,
@@ -24,7 +24,7 @@ import {
24
24
  readMailboxMessage,
25
25
  validateMailbox,
26
26
  } from "../../state/mailbox.ts";
27
- import { loadRunManifestById, saveRunManifest, saveRunTasks, updateRunStatus } from "../../state/state-store.ts";
27
+ import { loadRunManifestById, saveRunManifestAsync, saveRunTasks, updateRunStatus } from "../../state/state-store.ts";
28
28
  import { claimTask, releaseTaskClaim, transitionClaimedTaskStatus } from "../../state/task-claims.ts";
29
29
  import { appendLiveAgentControlRequest } from "../../subagents/live/control.ts";
30
30
  import {
@@ -145,8 +145,8 @@ export async function handleApi(params: TeamToolParamsValue, ctx: TeamContext):
145
145
  true,
146
146
  );
147
147
  try {
148
- return withRunLockSync(loaded.manifest, () => {
149
- const current = loadRunManifestById(ctx.cwd, loaded.manifest.runId) ?? loaded; // NOTE: inside withRunLockSync - consistent read
148
+ return await withRunLock(loaded.manifest, async () => {
149
+ const current = loadRunManifestById(ctx.cwd, loaded.manifest.runId) ?? loaded; // NOTE: inside withRunLock - consistent read
150
150
  const approval = current.manifest.planApproval;
151
151
  if (!approval?.required || approval.status !== "pending")
152
152
  return result(
@@ -169,7 +169,7 @@ export async function handleApi(params: TeamToolParamsValue, ctx: TeamContext):
169
169
  updatedAt: now,
170
170
  },
171
171
  };
172
- saveRunManifest(manifest);
172
+ await saveRunManifestAsync(manifest);
173
173
  appendEvent(manifest.eventsPath, {
174
174
  type: "plan.approved",
175
175
  runId: manifest.runId,
@@ -210,8 +210,8 @@ export async function handleApi(params: TeamToolParamsValue, ctx: TeamContext):
210
210
  true,
211
211
  );
212
212
  try {
213
- return withRunLockSync(loaded.manifest, () => {
214
- const current = loadRunManifestById(ctx.cwd, loaded.manifest.runId) ?? loaded; // NOTE: inside withRunLockSync - consistent read
213
+ return await withRunLock(loaded.manifest, async () => {
214
+ const current = loadRunManifestById(ctx.cwd, loaded.manifest.runId) ?? loaded; // NOTE: inside withRunLock - consistent read
215
215
  const approval = current.manifest.planApproval;
216
216
  if (!approval?.required || approval.status !== "pending")
217
217
  return result(
@@ -244,7 +244,7 @@ export async function handleApi(params: TeamToolParamsValue, ctx: TeamContext):
244
244
  updatedAt: now,
245
245
  },
246
246
  };
247
- saveRunManifest(manifest);
247
+ await saveRunManifestAsync(manifest);
248
248
  saveRunTasks(manifest, tasks);
249
249
  appendEvent(manifest.eventsPath, {
250
250
  type: "plan.cancelled",
@@ -24,7 +24,7 @@ import { snapshotManifests } from "../../runtime/verification-integrity.ts";
24
24
  import type { TeamToolParamsValue } from "../../schema/team-tool-schema.ts";
25
25
  import { atomicWriteJson } from "../../state/atomic-write.ts";
26
26
  import { appendEvent } from "../../state/event-log.ts";
27
- import { createRunPaths, saveRunManifest } from "../../state/state-store.ts";
27
+ import { createRunPaths, saveRunManifestAsync } from "../../state/state-store.ts";
28
28
  import type { GoalLoopState, TeamRunManifest } from "../../state/types.ts";
29
29
  import { spawnBackgroundTeamRun } from "../../subagents/async-entry.ts";
30
30
  import { logInternalError } from "../../utils/internal-error.ts";
@@ -268,7 +268,7 @@ export async function startGoalWrappedRun(
268
268
  ownerSessionId,
269
269
  runKind: "goal-loop",
270
270
  };
271
- saveRunManifest(goalLoopManifest);
271
+ await saveRunManifestAsync(goalLoopManifest);
272
272
  appendEvent(paths.eventsPath, {
273
273
  type: "goal.loop_start",
274
274
  runId: goalId,
@@ -19,7 +19,7 @@ import { snapshotManifests } from "../../runtime/verification-integrity.ts";
19
19
  import { acquireWorkspaceLock, isWorkspaceBusy, type WorkspaceLockHandle } from "../../runtime/workspace-lock.ts";
20
20
  import type { TeamToolParamsValue } from "../../schema/team-tool-schema.ts";
21
21
  import { appendEvent } from "../../state/event-log.ts";
22
- import { createRunPaths, saveRunManifest } from "../../state/state-store.ts";
22
+ import { createRunPaths, saveRunManifestAsync } from "../../state/state-store.ts";
23
23
  import type { GoalLoopState, GoalLoopStatus, TeamRunManifest } from "../../state/types.ts";
24
24
  import { spawnBackgroundTeamRun } from "../../subagents/async-entry.ts";
25
25
  import { logInternalError } from "../../utils/internal-error.ts";
@@ -191,7 +191,7 @@ async function handleStart(input: GoalSubActionInput): Promise<ReturnType<typeof
191
191
  ownerSessionId,
192
192
  runKind: "goal-loop",
193
193
  };
194
- saveRunManifest(goalLoopManifest);
194
+ await saveRunManifestAsync(goalLoopManifest);
195
195
  appendEvent(paths.eventsPath, {
196
196
  type: "goal.loop_start",
197
197
  runId: goalId,