@mjasnikovs/pi-task 0.38.8 → 0.38.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,9 @@
1
1
  import type { ExtensionAPI, ExtensionCommandContext } from '@earendil-works/pi-coding-agent';
2
2
  import { type GateDeps } from './task-gates.js';
3
3
  import { type FinalGateStageDeps } from './run-final-gate.js';
4
+ import { type RequirementEntry } from './requirements.js';
5
+ import { type ScoredPlan } from './coverage-loop.js';
6
+ import { type DanglingRef } from './artifact-closure.js';
4
7
  /**
5
8
  * Injectable seams so the planner and loop are testable without spawning pi.
6
9
  * `runChild` is the planning-only seam used by planAuto; everything else is one of
@@ -75,6 +78,84 @@ export declare function attachSpecRefs(titles: string[], refs: string[]): string
75
78
  * authoritative spec ref still rides on THIS step's own title via attachSpecRefs.
76
79
  */
77
80
  export declare function buildScopeFence(titles: string[], currentIndex: number): string;
81
+ /**
82
+ * What ORIENT establishes about the feature before anyone is asked anything: the
83
+ * spec text the planning children will actually read, and the requirement ledger
84
+ * derived from it. Every later stage reads these; none of them writes one.
85
+ */
86
+ export interface OrientedFeature {
87
+ /** The inlined spec, with phantom runtime specifiers struck out. */
88
+ featureForModel: string;
89
+ /** Manifest/config already on disk, fed to every triage call. '' when absent. */
90
+ existingFilesBlock: string;
91
+ /** Grounded requirement units extracted from the spec. */
92
+ reqEntries: RequirementEntry[];
93
+ /** How many of those a single task could own (cross-cutting ones excluded). */
94
+ ownableRequirements: number;
95
+ /** The task-count floor those ownable requirements imply. 0 ⇒ no channel. */
96
+ coarseFloor: number;
97
+ }
98
+ /**
99
+ * ORIENT — read the feature, strike what must never reach a planning child, and
100
+ * derive the requirement ledger. Depends on nothing but the feature and the tree,
101
+ * which is why it runs before clarify: the plan-shape fork below needs a real
102
+ * count to judge with. Every fallible part is best-effort; a fault degrades the
103
+ * channel it belongs to and never fails planning.
104
+ */
105
+ export declare function orientFeature(cwd: string, feature: string, deps: AutoDeps): Promise<OrientedFeature>;
106
+ /**
107
+ * ELICIT — clarify, sequential & adaptive: ask one question at a time, feeding every
108
+ * answer back into the next call so later questions react to earlier ones (e.g. a
109
+ * framework choice reshapes what gets asked). Each question is shown exactly like
110
+ * /task's grill dialog: a binary fork offers two options (A/B), otherwise the model's
111
+ * recommendation is shown as the input placeholder and in the title. Nothing is
112
+ * pre-filled into the editor — submitting an empty field is what accepts the
113
+ * recommendation; typing overrides it. Each generated question first runs the
114
+ * answer-side TRIAGE (triageClarifyQuestion): a question the inlined spec already
115
+ * settles is auto-resolved and never shown — only genuine open forks reach the user.
116
+ * The model emits NONE when nothing remains.
117
+ *
118
+ * The ONLY stage that talks to the user, and so the only one that can be dismissed:
119
+ * `null` means the user cancelled and the cancellation has already been announced.
120
+ * Every other outcome is a transcript, possibly empty.
121
+ */
122
+ export declare function elicitClarifications(ctx: ExtensionCommandContext, cwd: string, deps: AutoDeps, oriented: OrientedFeature): Promise<string | null>;
123
+ /**
124
+ * What DECOMPOSE settles: the task list, plus the two things the coverage loop
125
+ * needs to ask for a better one. `decomposePrompt` and `parsePlan` are returned
126
+ * rather than rebuilt because COVER re-prompts with the identical prompt and must
127
+ * reconcile the reply identically — rebuilding either is how the two paths drift.
128
+ */
129
+ export interface DecomposedPlan {
130
+ /** The reconciled, de-batched task titles. May be empty. */
131
+ planTitles: string[];
132
+ /** The exact prompt that produced them, for COVER's re-prompt. */
133
+ decomposePrompt: string;
134
+ /** Parse + fidelity-reconcile + de-batch, applied to EVERY decompose output. */
135
+ parsePlan: (raw: string) => string[];
136
+ }
137
+ /**
138
+ * DECOMPOSE — turn the settled feature into a task list, then defend that list's
139
+ * SHAPE: a plan under the granularity floor is sent back once to be split, and a
140
+ * suspect (empty or tiny) plan is regenerated on its own separate budget. Neither
141
+ * guard can block planning — a plan that survives both falls through to the judge.
142
+ */
143
+ export declare function decomposePlan(cwd: string, deps: AutoDeps, oriented: OrientedFeature, clarifications: string): Promise<DecomposedPlan>;
144
+ /** What COVER settles: the plan that ships, and the accounting behind it. */
145
+ export interface CoveredPlan {
146
+ /** The best-covered plan seen across the rounds — adoption is monotone. */
147
+ best: ScoredPlan;
148
+ /** That plan's titles, which is what everything downstream persists. */
149
+ planTitles: string[];
150
+ }
151
+ /**
152
+ * COVER — judge the plan against the feature and re-prompt for a better one, up to
153
+ * a bounded number of rounds. Adoption is MONOTONE: a retry that drops a
154
+ * requirement the current plan owns is rejected, so coverage can only hold or grow,
155
+ * and the plan at exhaustion is the best one seen rather than the last one drawn.
156
+ * Best-effort throughout — a fault degrades a signal, it never blocks planning.
157
+ */
158
+ export declare function coverPlan(ctx: ExtensionCommandContext, cwd: string, deps: AutoDeps, oriented: OrientedFeature, clarifications: string, decomposed: DecomposedPlan, specDangling: DanglingRef[]): Promise<CoveredPlan>;
78
159
  /** Plan phase: clarify → decompose → write AUTO file. Returns the new id, or null. */
79
160
  export declare function planAuto(ctx: ExtensionCommandContext, cwd: string, feature: string, deps: AutoDeps): Promise<string | null>;
80
161
  export declare function requestAutoCancel(): void;
@@ -409,21 +409,14 @@ async function schedulePendingRepairs(cwd, id, afterIndex, ctx, deps) {
409
409
  // the plan is best-effort here; the underlying debt is already recorded
410
410
  }
411
411
  }
412
- /** Plan phase: clarify → decompose → write AUTO file. Returns the new id, or null. */
413
- export async function planAuto(ctx, cwd, feature, deps) {
414
- // clarify sequential & adaptive: ask one question at a time, feeding every
415
- // answer back into the next call so later questions react to earlier ones
416
- // (e.g. a framework choice reshapes what gets asked). Each question is shown
417
- // exactly like /task's grill dialog: a binary fork offers two options (A/B),
418
- // otherwise the model's recommendation is shown as the input placeholder and
419
- // in the title. Nothing is pre-filled into the editor — submitting an empty
420
- // field is what accepts the recommendation (see the typed.length === 0 branch
421
- // below); typing overrides it. Each generated question first runs the
422
- // answer-side TRIAGE (triageClarifyQuestion): a question the inlined spec
423
- // already settles is auto-resolved and never shown — only genuine open forks
424
- // reach the user. The model emits NONE when nothing remains.
425
- const theme = ctx.ui.theme;
426
- const ui = new SessionUI(ctx);
412
+ /**
413
+ * ORIENT read the feature, strike what must never reach a planning child, and
414
+ * derive the requirement ledger. Depends on nothing but the feature and the tree,
415
+ * which is why it runs before clarify: the plan-shape fork below needs a real
416
+ * count to judge with. Every fallible part is best-effort; a fault degrades the
417
+ * channel it belongs to and never fails planning.
418
+ */
419
+ export async function orientFeature(cwd, feature, deps) {
427
420
  // Inline any @file spec the user referenced so clarify/decompose reason over
428
421
  // the real content, not a one-line "Implement @file" that reads as trivial.
429
422
  const rawFeatureForModel = await expandFeatureMentions(cwd, feature);
@@ -456,12 +449,8 @@ export async function planAuto(ctx, cwd, feature, deps) {
456
449
  // into the decompose prompt as a ledger (structure-mirroring can't discharge
457
450
  // them) and drive the per-requirement coverage accounting below.
458
451
  //
459
- // Runs BEFORE clarify (it depends only on the inlined feature, never on the
460
- // answers) so the plan-shape gate below has a real count to judge with: the
461
- // host must not seize the granularity fork on a spec that has no breakdown to
462
- // speak of. Best-effort:
463
- // a fault leaves reqEntries empty and the whole channel degrades to the old
464
- // behavior (one-liners / doc-less features naturally yield few or none).
452
+ // Best-effort: a fault leaves reqEntries empty and the whole channel degrades to
453
+ // the old behavior (one-liners / doc-less features naturally yield few or none).
465
454
  let reqEntries = [];
466
455
  try {
467
456
  // Recall floor: the obligation-marked passages ride into the prompt as a
@@ -499,6 +488,28 @@ export async function planAuto(ctx, cwd, feature, deps) {
499
488
  logPlanDebug(cwd, `granularity floor: ${ownableRequirements} ownable requirement(s) ⇒ at least `
500
489
  + `${coarseFloor} task(s)`);
501
490
  }
491
+ return { featureForModel, existingFilesBlock, reqEntries, ownableRequirements, coarseFloor };
492
+ }
493
+ /**
494
+ * ELICIT — clarify, sequential & adaptive: ask one question at a time, feeding every
495
+ * answer back into the next call so later questions react to earlier ones (e.g. a
496
+ * framework choice reshapes what gets asked). Each question is shown exactly like
497
+ * /task's grill dialog: a binary fork offers two options (A/B), otherwise the model's
498
+ * recommendation is shown as the input placeholder and in the title. Nothing is
499
+ * pre-filled into the editor — submitting an empty field is what accepts the
500
+ * recommendation; typing overrides it. Each generated question first runs the
501
+ * answer-side TRIAGE (triageClarifyQuestion): a question the inlined spec already
502
+ * settles is auto-resolved and never shown — only genuine open forks reach the user.
503
+ * The model emits NONE when nothing remains.
504
+ *
505
+ * The ONLY stage that talks to the user, and so the only one that can be dismissed:
506
+ * `null` means the user cancelled and the cancellation has already been announced.
507
+ * Every other outcome is a transcript, possibly empty.
508
+ */
509
+ export async function elicitClarifications(ctx, cwd, deps, oriented) {
510
+ const { featureForModel, existingFilesBlock, ownableRequirements } = oriented;
511
+ const theme = ctx.ui.theme;
512
+ const ui = new SessionUI(ctx);
502
513
  const answers = [];
503
514
  // Plain text of every question already shown, for the duplicate backstop.
504
515
  const askedQuestions = [];
@@ -615,26 +626,16 @@ export async function planAuto(ctx, cwd, feature, deps) {
615
626
  if (answers.length === 0) {
616
627
  ctx.ui.notify('No clarifying questions needed — planning tasks…', 'info');
617
628
  }
618
- const clarifications = answers.join('\n');
619
- // Artifact-production closure, plan side (mx5 run 13, PROMPT 2): runtime
620
- // files the spec REFERENCES (server snippets, prose "serve the built
621
- // index.html") that neither its file tree, its parsed build outputs, nor the
622
- // existing scaffold produce. Sentence-grounded coverage credited the SERVING
623
- // side and reported "0 unowned" while nothing ever CREATED the file so
624
- // these ride the coverage loop's `missing` list as unowned areas until some
625
- // task title claims the artifact (grounded in titles, which the coverage-map
626
- // model cannot fake the run-12 lesson). Deterministic and best-effort.
627
- let specDangling = [];
628
- try {
629
- specDangling = findSpecDanglingArtifacts(featureForModel, rel => existsSync(path.join(cwd, rel)));
630
- if (specDangling.length > 0) {
631
- logPlanDebug(cwd, `artifact closure: ${specDangling.length} dangling runtime artifact(s) in the `
632
- + `spec: ${specDangling.map(d => d.path).join(', ')}`);
633
- }
634
- }
635
- catch {
636
- // best-effort channel
637
- }
629
+ return answers.join('\n');
630
+ }
631
+ /**
632
+ * DECOMPOSE turn the settled feature into a task list, then defend that list's
633
+ * SHAPE: a plan under the granularity floor is sent back once to be split, and a
634
+ * suspect (empty or tiny) plan is regenerated on its own separate budget. Neither
635
+ * guard can block planning a plan that survives both falls through to the judge.
636
+ */
637
+ export async function decomposePlan(cwd, deps, oriented, clarifications) {
638
+ const { featureForModel, reqEntries, ownableRequirements, coarseFloor } = oriented;
638
639
  // Tests-in-the-same-change cadence (mx5 run 14, PROMPT item 6): when the
639
640
  // decisions mandate it, a whole-project batch test task contradicts them —
640
641
  // run 14 shipped one anyway (TASK_0037, 4.7h, yolo-accepted FAIL) because
@@ -729,6 +730,19 @@ export async function planAuto(ctx, cwd, feature, deps) {
729
730
  if (retryTitles.length > planTitles.length)
730
731
  planTitles = retryTitles;
731
732
  }
733
+ return { planTitles, decomposePrompt, parsePlan };
734
+ }
735
+ /**
736
+ * COVER — judge the plan against the feature and re-prompt for a better one, up to
737
+ * a bounded number of rounds. Adoption is MONOTONE: a retry that drops a
738
+ * requirement the current plan owns is rejected, so coverage can only hold or grow,
739
+ * and the plan at exhaustion is the best one seen rather than the last one drawn.
740
+ * Best-effort throughout — a fault degrades a signal, it never blocks planning.
741
+ */
742
+ export async function coverPlan(ctx, cwd, deps, oriented, clarifications, decomposed, specDangling) {
743
+ const { featureForModel, reqEntries } = oriented;
744
+ const { decomposePrompt, parsePlan } = decomposed;
745
+ let planTitles = decomposed.planTitles;
732
746
  // Coverage gate: a stochastic degenerate completion (live mx5: ONE task +
733
747
  // natural EOS for an 18KB design doc) is nonempty, so the length guard below
734
748
  // never fires and the whole run "completes" after one task. Judge the list
@@ -905,6 +919,44 @@ export async function planAuto(ctx, cwd, feature, deps) {
905
919
  + 'task. To give it one, stop now and add it to the plan in .pi-tasks/; otherwise '
906
920
  + 'it proceeds.', 'warning');
907
921
  }
922
+ return { best, planTitles };
923
+ }
924
+ /** Plan phase: clarify → decompose → write AUTO file. Returns the new id, or null. */
925
+ export async function planAuto(ctx, cwd, feature, deps) {
926
+ // ORIENT. Reads the feature and the tree, asks nobody anything.
927
+ const oriented = await orientFeature(cwd, feature, deps);
928
+ // The floor and the ownable count are DECOMPOSE's to enforce; what the tail of
929
+ // this function still reads is the spec text and the requirement ledger.
930
+ const { featureForModel, reqEntries } = oriented;
931
+ // ELICIT. The only stage that talks to the user.
932
+ const clarifications = await elicitClarifications(ctx, cwd, deps, oriented);
933
+ if (clarifications === null)
934
+ return null; // dismissed; already announced
935
+ // Artifact-production closure, plan side (mx5 run 13, PROMPT 2): runtime
936
+ // files the spec REFERENCES (server snippets, prose "serve the built
937
+ // index.html") that neither its file tree, its parsed build outputs, nor the
938
+ // existing scaffold produce. Sentence-grounded coverage credited the SERVING
939
+ // side and reported "0 unowned" while nothing ever CREATED the file — so
940
+ // these ride the coverage loop's `missing` list as unowned areas until some
941
+ // task title claims the artifact (grounded in titles, which the coverage-map
942
+ // model cannot fake — the run-12 lesson). Deterministic and best-effort.
943
+ let specDangling = [];
944
+ try {
945
+ specDangling = findSpecDanglingArtifacts(featureForModel, rel => existsSync(path.join(cwd, rel)));
946
+ if (specDangling.length > 0) {
947
+ logPlanDebug(cwd, `artifact closure: ${specDangling.length} dangling runtime artifact(s) in the `
948
+ + `spec: ${specDangling.map(d => d.path).join(', ')}`);
949
+ }
950
+ }
951
+ catch {
952
+ // best-effort channel
953
+ }
954
+ // DECOMPOSE. Produces the task list and the means to ask for a better one.
955
+ const decomposedPlan = await decomposePlan(cwd, deps, oriented, clarifications);
956
+ // COVER. Judges the plan and re-prompts for a better one, monotonically.
957
+ const covered = await coverPlan(ctx, cwd, deps, oriented, clarifications, decomposedPlan, specDangling);
958
+ const best = covered.best;
959
+ const planTitles = covered.planTitles;
908
960
  // Carry what no single task owns (goal A(b)/(c)): cross-cutting requirements
909
961
  // become `.pi-tasks/requirements.md`, injected VERBATIM into every task's
910
962
  // refine/compose (run 11: §10's test-first cadence had no carrier — the "spec
@@ -942,6 +994,20 @@ export async function planAuto(ctx, cwd, feature, deps) {
942
994
  ctx.ui.notify(`/task-auto: carrying ${parts.join(', ')} requirement(s) into every task`
943
995
  + ' — see .pi-tasks/requirements.md.', 'info');
944
996
  }
997
+ // Thread the feature's spec doc(s) into every title so each per-task
998
+ // pipeline — which only ever sees its title — reads the real spec instead of
999
+ // a lossy one-line paraphrase of it.
1000
+ const refs = await readableMentions(cwd, feature);
1001
+ const titles = attachSpecRefs(planTitles, refs);
1002
+ if (titles.length === 0) {
1003
+ announceDone(ctx, '/task-auto: no tasks produced from the feature.', 'warning');
1004
+ return null;
1005
+ }
1006
+ // The two grounded extractions below run AFTER the empty-plan guard on purpose.
1007
+ // They each spawn a child and each APPEND to a run-level artifact; on the
1008
+ // empty-plan path the plan is discarded one line later, so running them first
1009
+ // burned two model calls and left contracts.md / launch-contract.md carrying
1010
+ // facts for a run that never produced a task.
945
1011
  // Cross-slice contract registry (mx5 run 8, F3): now that the plan is settled,
946
1012
  // extract the interface facts MORE THAN ONE slice must agree on — endpoint paths,
947
1013
  // exported signatures, file layouts, env var names the DESIGN pins — into a
@@ -983,15 +1049,6 @@ export async function planAuto(ctx, cwd, feature, deps) {
983
1049
  ground: emitted => keepGroundedScripts(emitted, featureForModel),
984
1050
  append: appendDeclaredScripts
985
1051
  });
986
- // Thread the feature's spec doc(s) into every title so each per-task
987
- // pipeline — which only ever sees its title — reads the real spec instead of
988
- // a lossy one-line paraphrase of it.
989
- const refs = await readableMentions(cwd, feature);
990
- const titles = attachSpecRefs(planTitles, refs);
991
- if (titles.length === 0) {
992
- announceDone(ctx, '/task-auto: no tasks produced from the feature.', 'warning');
993
- return null;
994
- }
995
1052
  // Persist the TASK-MAPPED requirements keyed by the (spec-ref-attached) title
996
1053
  // each task will carry (mx5 run 16: only cross-cutting entries travelled;
997
1054
  // the 33 mapped ones shaped the title list and vanished — TASK_0008 narrowed
@@ -30,7 +30,8 @@ import type { ExtensionCommandContext } from '@earendil-works/pi-coding-agent';
30
30
  import type { CommitResult } from './auto-commit.js';
31
31
  import type { FinalGateOutcome } from './final-gate.js';
32
32
  import type { FinalGateFixFn } from './gate-deps.js';
33
- import { type AcceptDebt } from './accept-debt.js';
33
+ import { type AcceptDebt, type DebtOrigin } from './accept-debt.js';
34
+ import { type OwnedRequirement } from './requirements.js';
34
35
  /**
35
36
  * The seams this stage drives. A strict subset of what /task-auto builds, and
36
37
  * deliberately narrow: every optional dep absent degrades to a documented earlier
@@ -86,6 +87,20 @@ export interface FinalGateStageDeps {
86
87
  * file) — the same auditability contract the per-task records carry. Best-effort:
87
88
  * absent in tests → skipped; a failure never breaks the gate. */
88
89
  record?: (cwd: string, taskId: string, line: string) => Promise<void>;
90
+ /**
91
+ * Write one ACCEPT debt to the ledger. The run-level twin of `GateDeps.recordDebt`,
92
+ * and injectable for the same reason: without it a test that only wants to observe
93
+ * WHICH debts a scenario carries has to write, and then read back, the real ledger
94
+ * on disk. Absent → the real `accept-debt.ts` writer, so production wiring and the
95
+ * prior behaviour are unchanged.
96
+ */
97
+ recordDebt?: (cwd: string, taskId: string, reason: string, origin: DebtOrigin) => Promise<void>;
98
+ /**
99
+ * Read the owned-requirement ledger, to report obligations a task DETACHED and no
100
+ * later task claimed. Absent → the real `requirements.ts` reader (prior behaviour);
101
+ * injectable so that check is testable without seeding a real ledger file.
102
+ */
103
+ ownedRequirements?: (cwd: string) => Promise<OwnedRequirement[]>;
89
104
  }
90
105
  /** Inputs that vary per caller. */
91
106
  export interface FinalGateStageParams {
@@ -73,7 +73,7 @@ export async function runFinalGateStage(active, deps, p) {
73
73
  * What they no longer each restate is WHERE the debt goes: the run's own id, under
74
74
  * origin 'final-gate', which is what the next run's gate re-checks.
75
75
  */
76
- const carryDebt = (reason) => recordDebt(cwd, id, reason, 'final-gate');
76
+ const carryDebt = (reason) => (deps.recordDebt ?? recordDebt)(cwd, id, reason, 'final-gate');
77
77
  // Set when the gate finished having observed NOTHING dynamic. Declared out here so
78
78
  // the run-completion announcement can say so.
79
79
  let unobservedNote = null;
@@ -147,7 +147,7 @@ export async function runFinalGateStage(active, deps, p) {
147
147
  // satisfy it, nexttask 2) and no later task claimed. Detach never deletes the
148
148
  // quote, so the run ends holding it — say so, or the resolution would be a quieter
149
149
  // version of the deletion it exists to prevent.
150
- const unclaimed = unclaimedPendingRequirements(await readOwnedRequirements(cwd).catch(() => []));
150
+ const unclaimed = unclaimedPendingRequirements(await (deps.ownedRequirements ?? readOwnedRequirements)(cwd).catch(() => []));
151
151
  for (const o of unclaimed) {
152
152
  await recGate(`owned requirement UNCLAIMED — "${o.quote.slice(0, 200)}"`
153
153
  + ` [frozen in "${o.title.slice(0, 60)}"; no task claimed`
@@ -1,5 +1,7 @@
1
1
  /** The env var the orchestrator stamps with the per-run id children inherit. */
2
2
  export declare const RESEARCH_RUN_ID_ENV = "PI_TASK_RUN_ID";
3
+ /** Is this a transient filesystem error worth retrying inside the caller's deadline? */
4
+ export declare function isRetryableFsError(err: unknown): boolean;
3
5
  export declare function researchCacheFile(cwd: string): string;
4
6
  /**
5
7
  * The current run's id, or undefined when caching is off (the orchestrator did not
@@ -117,6 +117,12 @@ const LOCK_SUFFIX = '.lock';
117
117
  const LOCK_TIMEOUT_MS = 2_000;
118
118
  /** Poll interval while the lock is held by someone else. */
119
119
  const LOCK_POLL_MS = 10;
120
+ /**
121
+ * How long the atomic replace retries a transient refusal. Short: the lock is HELD for
122
+ * every one of these milliseconds, so this trades a bounded stall for not losing a
123
+ * write the lock was taken to protect.
124
+ */
125
+ const RENAME_TIMEOUT_MS = 500;
120
126
  /**
121
127
  * A lock older than this is treated as abandoned and removed. Two writers can both
122
128
  * decide that and both proceed, which degrades exactly to the pre-lock behaviour (one
@@ -124,6 +130,32 @@ const LOCK_POLL_MS = 10;
124
130
  * the run. Sized far above the critical section, so a live holder is never stolen from.
125
131
  */
126
132
  const LOCK_STALE_MS = 30_000;
133
+ /**
134
+ * Filesystem errors that mean "try again in a moment", not "this will never work".
135
+ *
136
+ * POSIX gives `mkdir` two clean answers when someone else holds the lock: it succeeds,
137
+ * or it is EEXIST. Windows has a third. A directory that another process is removing
138
+ * enters a DELETE-PENDING state, and a create against it fails with EPERM/EACCES/EBUSY
139
+ * instead of EEXIST — so the exact moment the previous writer released the lock is a
140
+ * window in which the next writer's `mkdir` fails with a code that used to be read as
141
+ * fatal. The store was then silently skipped: no throw, no log, exit code 0, one entry
142
+ * missing. That is the whole of CI's `39 of 40` on windows-latest; the same run's
143
+ * ubuntu half is green because POSIX never produces the code.
144
+ *
145
+ * `rename` over an existing file has the same shape on Windows — it fails while any
146
+ * other handle is open on the target, including a scanner's — so the cache write
147
+ * retries on this set too.
148
+ */
149
+ const RETRYABLE_FS_CODES = new Set(['EEXIST', 'EPERM', 'EACCES', 'EBUSY', 'ENOTEMPTY']);
150
+ /** Is this a transient filesystem error worth retrying inside the caller's deadline? */
151
+ export function isRetryableFsError(err) {
152
+ const code = err?.code;
153
+ return code !== undefined && RETRYABLE_FS_CODES.has(code);
154
+ }
155
+ /** Does this error mean the lock directory is genuinely held right now? */
156
+ function isHeldError(err) {
157
+ return err?.code === 'EEXIST';
158
+ }
127
159
  /**
128
160
  * Schema marker for per-entry package provenance. A file without it was written by a
129
161
  * version that stored no `pkg` on its entries, so its docs entries are indistinguishable
@@ -307,15 +339,35 @@ export async function lookupResearch(cwd, runId, key) {
307
339
  }
308
340
  /** Write the cache file atomic-ish, so a concurrent reader never sees it half-written. */
309
341
  async function writeCacheFile(cwd, out) {
342
+ const tmp = `${researchCacheFile(cwd)}.${process.pid}.${Math.random().toString(36).slice(2, 8)}.tmp`;
310
343
  try {
311
344
  await fsp.mkdir(tasksDir(cwd), { recursive: true });
312
- const tmp = `${researchCacheFile(cwd)}.${process.pid}.${Math.random().toString(36).slice(2, 8)}.tmp`;
313
345
  await fsp.writeFile(tmp, JSON.stringify(out), 'utf8');
314
- await fsp.rename(tmp, researchCacheFile(cwd));
346
+ // The replace, not the write, is the part Windows can transiently refuse — any
347
+ // other open handle on the target (a reader, a scanner) fails it with EPERM.
348
+ // The whole point of the lock above is that this write is not lost, so a
349
+ // transient refusal is retried inside the same bounded budget rather than
350
+ // swallowed. The lock is still held throughout.
351
+ const deadline = Date.now() + RENAME_TIMEOUT_MS;
352
+ for (;;) {
353
+ try {
354
+ await fsp.rename(tmp, researchCacheFile(cwd));
355
+ return;
356
+ }
357
+ catch (err) {
358
+ if (!isRetryableFsError(err) || Date.now() >= deadline)
359
+ throw err;
360
+ await new Promise(resolve => setTimeout(resolve, LOCK_POLL_MS));
361
+ }
362
+ }
315
363
  }
316
364
  catch {
317
365
  // best-effort cache
318
366
  }
367
+ finally {
368
+ // Never leave a .tmp behind for a replace that never happened.
369
+ await fsp.rm(tmp, { force: true }).catch(() => { });
370
+ }
319
371
  }
320
372
  /**
321
373
  * Serialises this process's own writers per cache file, so the 4-6 parallel tool calls
@@ -337,8 +389,18 @@ async function acquireLock(lockPath, deadline) {
337
389
  return true;
338
390
  }
339
391
  catch (err) {
340
- if (err.code !== 'EEXIST')
392
+ if (!isRetryableFsError(err))
341
393
  return false;
394
+ // Not EEXIST but still retryable ⇒ Windows delete-pending (see
395
+ // RETRYABLE_FS_CODES). There is no lock to inspect for staleness: the
396
+ // directory is on its way out, so wait one poll and try to create it again
397
+ // rather than reporting a hold that nobody has.
398
+ if (!isHeldError(err)) {
399
+ if (Date.now() >= deadline)
400
+ return false;
401
+ await new Promise(resolve => setTimeout(resolve, LOCK_POLL_MS));
402
+ continue;
403
+ }
342
404
  try {
343
405
  const st = await fsp.stat(lockPath);
344
406
  if (Date.now() - st.mtimeMs > LOCK_STALE_MS) {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@mjasnikovs/pi-task",
3
- "version": "0.38.8",
3
+ "version": "0.38.10",
4
4
  "description": "Deterministic task planning and spec-orchestration for local models — crash-safe /task pipelines with verify/enforce gates, a real-time remote web view, and web/docs/fetch/worker subagent tools.",
5
5
  "type": "module",
6
6
  "main": "./dist/index.js",