@mjasnikovs/pi-task 0.38.15 → 0.38.17

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (86) hide show
  1. package/dist/config/config.d.ts +26 -0
  2. package/dist/config/config.js +68 -17
  3. package/dist/shared/child-process.js +9 -16
  4. package/dist/task/accept-debt.d.ts +7 -5
  5. package/dist/task/accept-debt.js +18 -14
  6. package/dist/task/artifact-closure.js +18 -63
  7. package/dist/task/auto-orchestrator.js +211 -218
  8. package/dist/task/autofix-ledger.d.ts +113 -0
  9. package/dist/task/autofix-ledger.js +152 -0
  10. package/dist/task/boot-probe.d.ts +109 -1
  11. package/dist/task/boot-probe.js +139 -23
  12. package/dist/task/child-runner.d.ts +50 -6
  13. package/dist/task/child-runner.js +48 -69
  14. package/dist/task/command-run.d.ts +49 -6
  15. package/dist/task/command-run.js +154 -18
  16. package/dist/task/coverage-loop.d.ts +11 -0
  17. package/dist/task/coverage-loop.js +16 -0
  18. package/dist/task/external-context.d.ts +9 -12
  19. package/dist/task/external-context.js +5 -5
  20. package/dist/task/failure-classifier.d.ts +9 -1
  21. package/dist/task/failure-classifier.js +9 -0
  22. package/dist/task/final-gate-fix.d.ts +22 -26
  23. package/dist/task/final-gate-fix.js +16 -31
  24. package/dist/task/final-gate.d.ts +10 -2
  25. package/dist/task/final-gate.js +55 -89
  26. package/dist/task/fix-child.d.ts +64 -0
  27. package/dist/task/fix-child.js +66 -0
  28. package/dist/task/gate-deps.js +20 -13
  29. package/dist/task/lint-fix.d.ts +7 -0
  30. package/dist/task/lint-fix.js +45 -9
  31. package/dist/task/orchestrator.d.ts +33 -24
  32. package/dist/task/orchestrator.js +75 -46
  33. package/dist/task/phases.d.ts +120 -34
  34. package/dist/task/phases.js +221 -134
  35. package/dist/task/plan-orchestrator.js +2 -2
  36. package/dist/task/plan-rounds.d.ts +86 -0
  37. package/dist/task/plan-rounds.js +105 -0
  38. package/dist/task/plan-session.d.ts +31 -21
  39. package/dist/task/plan-session.js +97 -120
  40. package/dist/task/qa-transcript.d.ts +100 -0
  41. package/dist/task/qa-transcript.js +99 -0
  42. package/dist/task/question-source.d.ts +117 -0
  43. package/dist/task/question-source.js +174 -0
  44. package/dist/task/repo-health-check.d.ts +21 -21
  45. package/dist/task/repo-health-check.js +43 -112
  46. package/dist/task/run-end.d.ts +77 -0
  47. package/dist/task/run-end.js +37 -0
  48. package/dist/task/run-final-gate.js +71 -79
  49. package/dist/task/serve-entry.js +6 -57
  50. package/dist/task/shipped-source.d.ts +67 -0
  51. package/dist/task/shipped-source.js +144 -0
  52. package/dist/task/task-gates.d.ts +9 -1
  53. package/dist/task/task-gates.js +27 -6
  54. package/dist/task/terminal-outcome.d.ts +1 -1
  55. package/dist/task/terminal-outcome.js +12 -0
  56. package/dist/task/verify-work.d.ts +46 -0
  57. package/dist/task/verify-work.js +51 -3
  58. package/dist/workers/brave-search.d.ts +7 -0
  59. package/dist/workers/brave-search.js +36 -55
  60. package/dist/workers/ddg-search.d.ts +1 -1
  61. package/dist/workers/ddg-search.js +27 -47
  62. package/dist/workers/docs-core.d.ts +71 -1
  63. package/dist/workers/docs-core.js +131 -71
  64. package/dist/workers/exa-search.d.ts +2 -2
  65. package/dist/workers/exa-search.js +53 -68
  66. package/dist/workers/html-clean.js +67 -88
  67. package/dist/workers/http-request.d.ts +74 -0
  68. package/dist/workers/http-request.js +103 -0
  69. package/dist/workers/npm-version.js +37 -42
  70. package/dist/workers/pi-worker-core.d.ts +13 -2
  71. package/dist/workers/pi-worker-core.js +35 -25
  72. package/dist/workers/pi-worker-docs.d.ts +1 -1
  73. package/dist/workers/pi-worker-docs.js +49 -68
  74. package/dist/workers/pi-worker-fetch.d.ts +1 -1
  75. package/dist/workers/pi-worker-fetch.js +20 -21
  76. package/dist/workers/pi-worker-search.js +6 -4
  77. package/dist/workers/pi-worker.js +5 -4
  78. package/dist/workers/search-core.d.ts +1 -1
  79. package/dist/workers/search-core.js +36 -42
  80. package/dist/workers/search-types.d.ts +13 -0
  81. package/dist/workers/search-types.js +27 -0
  82. package/dist/workers/shared.d.ts +51 -11
  83. package/dist/workers/shared.js +0 -0
  84. package/dist/workers/worker-channels.d.ts +60 -0
  85. package/dist/workers/worker-channels.js +98 -0
  86. package/package.json +1 -1
@@ -0,0 +1,117 @@
1
+ /**
2
+ * Where the NEXT question comes from — the other half of `question-dialog.ts`.
3
+ *
4
+ * `question-dialog.ts` unified the ANSWER side of the adaptive dialogs (the picker
5
+ * cards, the reply mapping) and its own docstring makes the argument for doing so:
6
+ * *"It was written three times… the two mirrors were never converted, and they had
7
+ * already drifted apart in three ways… The next edit to any of them is where the
8
+ * bug lands."* The QUESTION side — generate → parse → pick the real question →
9
+ * dedupe → spend one corrective re-prompt → yield or exhaust — was left behind,
10
+ * and it had drifted between the two loops that use the SAME parser on the SAME
11
+ * prompt format (`plan-prompts.ts` says `parseClarifyList` parses it "UNCHANGED";
12
+ * `auto-prompts.ts` specifies the identical shape).
13
+ *
14
+ * The five drifts, all in clarify's favour of being wrong:
15
+ *
16
+ * 1. **Which entry is the question.** `parseClarifyList` pushes an entry for
17
+ * EVERY numbered line, and the local model writes numbered analysis notes
18
+ * before the question it was asked for (measured live). `pickQuestion` prefers
19
+ * the first entry carrying a `SUGGESTED:` line; clarify took `parsed[0]`
20
+ * blindly, showing the note as the question and losing the recommendation
21
+ * attached further down.
22
+ * 2. **NONE vs unparseable.** The parser returns `[]` for both. Clarify's
23
+ * `if (parsed.length === 0) break` ended the whole clarify — and decomposed the
24
+ * feature with ZERO clarifications — on a formatting slip.
25
+ * 3. **A re-typed sentinel.** `isNoneReply`'s regex was a byte-identical second
26
+ * copy of the parser's own.
27
+ * 4. **Missing SUGGESTED** bought one corrective re-prompt in plan and none in
28
+ * clarify, so clarify showed a card-less question.
29
+ * 5. **The deferral guard.** It exists because an accepted "clarify with the user
30
+ * before proceeding" rode into `/task`'s handoff AS AN AUTHORITATIVE DECISION
31
+ * and produced a task whose VERIFY asserted no source file had changed.
32
+ * Clarify's answers ride into the decompose prompt and the AUTO file with
33
+ * exactly the same authority, and had no guard.
34
+ *
35
+ * NOT unified here: grill's generation loop. It uses a different parser
36
+ * (`parseGrillQuestions`, which yields bare strings) and has no `SUGGESTED` at
37
+ * generation time at all — grill's recommendation comes from `phaseAutoAnswer`
38
+ * one step later, so every quality rule below is inapplicable to it. Folding it in
39
+ * would mean a generic over the parsed shape with one consumer opting out of the
40
+ * entire rule table: a wider interface for less behaviour.
41
+ */
42
+ import { type ClarifyQuestion } from './parsers.js';
43
+ /**
44
+ * The cap on distinct questions ONE adaptive dialog may ask.
45
+ *
46
+ * Clarify and plan each declared their own `8`, linked only by a comment saying
47
+ * "matches /task-auto's MAX_CLARIFY_QUESTIONS, for the same reason". They bound
48
+ * the same thing for the same reason; this is that reason, once.
49
+ */
50
+ export declare const MAX_DIALOG_QUESTIONS = 8;
51
+ /** True when the reply is the deliberate "nothing left to ask" sentinel, as
52
+ * opposed to output the parser simply could not read. */
53
+ export declare function isNoneReply(raw: string): boolean;
54
+ /**
55
+ * Which of the parsed entries is the actual question.
56
+ *
57
+ * `parseClarifyList` turns EVERY numbered line into an entry, and the local model
58
+ * sometimes writes a numbered analysis note or two before the question it was
59
+ * asked for (measured live: the first numbered line was a note like
60
+ * "1. gateDebugWriter in orchestrator.ts — wraps a raw append function"). Taking
61
+ * entry 0 blindly then shows the note as the question and loses the SUGGESTED line
62
+ * attached further down. The prompt requires exactly one SUGGESTED, and the parser
63
+ * attaches it to the entry it follows.
64
+ */
65
+ export declare function pickQuestion<T extends {
66
+ suggested?: string;
67
+ }>(parsed: T[]): T | undefined;
68
+ /**
69
+ * One QUALITY rule: a defect in an otherwise-usable question that is worth exactly
70
+ * one corrective re-prompt.
71
+ *
72
+ * `detect` returns the hint to re-prompt with, or null to pass. `repair` is what
73
+ * to do when the SAME defect survives the re-prompt — a question with a bad
74
+ * default still beats no question, so a rule degrades rather than discards.
75
+ */
76
+ export interface QuestionRule {
77
+ id: string;
78
+ detect: (q: ClarifyQuestion, plain: string) => string | null;
79
+ /** Applied only when the defect survived its one re-prompt. */
80
+ repair?: (q: ClarifyQuestion) => ClarifyQuestion;
81
+ }
82
+ export interface QuestionSourceDeps {
83
+ /**
84
+ * Ask the model for the next question. `hint` is the corrective re-prompt to
85
+ * prepend, or null. The caller closes over its own transcript and prompt — the
86
+ * source never builds one.
87
+ */
88
+ generate: (hint: string | null) => Promise<string>;
89
+ /** The corrective re-prompt for a reply the parser could not read. */
90
+ formatHint: string;
91
+ /** Ordered quality rules; each may fire at most once per question. */
92
+ rules?: ReadonlyArray<QuestionRule>;
93
+ cap?: number;
94
+ log?: (msg: string) => void;
95
+ }
96
+ export type NextQuestion = {
97
+ kind: 'question';
98
+ q: ClarifyQuestion;
99
+ plain: string;
100
+ index: number;
101
+ } | {
102
+ kind: 'exhausted';
103
+ why: 'none' | 'cap' | 'dups' | 'unparseable';
104
+ };
105
+ /**
106
+ * A deep module over a state machine that was five mutable locals per site.
107
+ *
108
+ * The interface is one method. Behind it: the cap, the duplicate backstop and its
109
+ * strike budget, the NONE-vs-unparseable distinction, `pickQuestion`, the one-shot
110
+ * budget shared by every quality rule, and the hint precedence between a format
111
+ * re-prompt and a duplicate re-prompt.
112
+ */
113
+ export declare function makeQuestionSource(deps: QuestionSourceDeps): {
114
+ next: () => Promise<NextQuestion>;
115
+ asked: () => ReadonlyArray<string>;
116
+ reopen: () => void;
117
+ };
@@ -0,0 +1,174 @@
1
+ /**
2
+ * Where the NEXT question comes from — the other half of `question-dialog.ts`.
3
+ *
4
+ * `question-dialog.ts` unified the ANSWER side of the adaptive dialogs (the picker
5
+ * cards, the reply mapping) and its own docstring makes the argument for doing so:
6
+ * *"It was written three times… the two mirrors were never converted, and they had
7
+ * already drifted apart in three ways… The next edit to any of them is where the
8
+ * bug lands."* The QUESTION side — generate → parse → pick the real question →
9
+ * dedupe → spend one corrective re-prompt → yield or exhaust — was left behind,
10
+ * and it had drifted between the two loops that use the SAME parser on the SAME
11
+ * prompt format (`plan-prompts.ts` says `parseClarifyList` parses it "UNCHANGED";
12
+ * `auto-prompts.ts` specifies the identical shape).
13
+ *
14
+ * The five drifts, all in clarify's favour of being wrong:
15
+ *
16
+ * 1. **Which entry is the question.** `parseClarifyList` pushes an entry for
17
+ * EVERY numbered line, and the local model writes numbered analysis notes
18
+ * before the question it was asked for (measured live). `pickQuestion` prefers
19
+ * the first entry carrying a `SUGGESTED:` line; clarify took `parsed[0]`
20
+ * blindly, showing the note as the question and losing the recommendation
21
+ * attached further down.
22
+ * 2. **NONE vs unparseable.** The parser returns `[]` for both. Clarify's
23
+ * `if (parsed.length === 0) break` ended the whole clarify — and decomposed the
24
+ * feature with ZERO clarifications — on a formatting slip.
25
+ * 3. **A re-typed sentinel.** `isNoneReply`'s regex was a byte-identical second
26
+ * copy of the parser's own.
27
+ * 4. **Missing SUGGESTED** bought one corrective re-prompt in plan and none in
28
+ * clarify, so clarify showed a card-less question.
29
+ * 5. **The deferral guard.** It exists because an accepted "clarify with the user
30
+ * before proceeding" rode into `/task`'s handoff AS AN AUTHORITATIVE DECISION
31
+ * and produced a task whose VERIFY asserted no source file had changed.
32
+ * Clarify's answers ride into the decompose prompt and the AUTO file with
33
+ * exactly the same authority, and had no guard.
34
+ *
35
+ * NOT unified here: grill's generation loop. It uses a different parser
36
+ * (`parseGrillQuestions`, which yields bare strings) and has no `SUGGESTED` at
37
+ * generation time at all — grill's recommendation comes from `phaseAutoAnswer`
38
+ * one step later, so every quality rule below is inapplicable to it. Folding it in
39
+ * would mean a generic over the parsed shape with one consumer opting out of the
40
+ * entire rule table: a wider interface for less behaviour.
41
+ */
42
+ import { parseClarifyList } from './parsers.js';
43
+ import { DUP_REPROMPT_HINT, isDuplicateQuestion, MAX_DUP_STRIKES } from './question-dedup.js';
44
+ import { stripInlineMarkdown } from './inline-markdown.js';
45
+ /**
46
+ * The cap on distinct questions ONE adaptive dialog may ask.
47
+ *
48
+ * Clarify and plan each declared their own `8`, linked only by a comment saying
49
+ * "matches /task-auto's MAX_CLARIFY_QUESTIONS, for the same reason". They bound
50
+ * the same thing for the same reason; this is that reason, once.
51
+ */
52
+ export const MAX_DIALOG_QUESTIONS = 8;
53
+ /** True when the reply is the deliberate "nothing left to ask" sentinel, as
54
+ * opposed to output the parser simply could not read. */
55
+ export function isNoneReply(raw) {
56
+ return /^\s*NONE\s*$/m.test(raw);
57
+ }
58
+ /**
59
+ * Which of the parsed entries is the actual question.
60
+ *
61
+ * `parseClarifyList` turns EVERY numbered line into an entry, and the local model
62
+ * sometimes writes a numbered analysis note or two before the question it was
63
+ * asked for (measured live: the first numbered line was a note like
64
+ * "1. gateDebugWriter in orchestrator.ts — wraps a raw append function"). Taking
65
+ * entry 0 blindly then shows the note as the question and loses the SUGGESTED line
66
+ * attached further down. The prompt requires exactly one SUGGESTED, and the parser
67
+ * attaches it to the entry it follows.
68
+ */
69
+ export function pickQuestion(parsed) {
70
+ return parsed.find(q => q.suggested !== undefined && q.suggested.length > 0) ?? parsed[0];
71
+ }
72
+ /**
73
+ * A deep module over a state machine that was five mutable locals per site.
74
+ *
75
+ * The interface is one method. Behind it: the cap, the duplicate backstop and its
76
+ * strike budget, the NONE-vs-unparseable distinction, `pickQuestion`, the one-shot
77
+ * budget shared by every quality rule, and the hint precedence between a format
78
+ * re-prompt and a duplicate re-prompt.
79
+ */
80
+ export function makeQuestionSource(deps) {
81
+ const cap = deps.cap ?? MAX_DIALOG_QUESTIONS;
82
+ const rules = deps.rules ?? [];
83
+ const asked = [];
84
+ let dupStrikes = 0;
85
+ let dupHint = null;
86
+ // The one-shot budget is per QUESTION, not per dialog: a fresh draw starts
87
+ // with every rule available again. `hint` being non-null is also what spends
88
+ // it — a rule may not fire while another rule's re-prompt is in flight, which
89
+ // is what stops two rules ping-ponging a stateless child forever.
90
+ let hint = null;
91
+ async function next() {
92
+ for (;;) {
93
+ if (asked.length >= cap) {
94
+ deps.log?.(`question cap (${cap}) reached`);
95
+ return { kind: 'exhausted', why: 'cap' };
96
+ }
97
+ const raw = await deps.generate(hint ?? dupHint);
98
+ const parsed = parseClarifyList(raw);
99
+ if (parsed.length === 0) {
100
+ // `[]` means BOTH "deliberate NONE" and "could not parse". Ending
101
+ // the dialog on the second is how a formatting slip becomes a
102
+ // feature decomposed with zero clarifications.
103
+ if (!isNoneReply(raw) && hint === null) {
104
+ deps.log?.('unparseable question reply — one format re-prompt');
105
+ hint = deps.formatHint;
106
+ continue;
107
+ }
108
+ // A SECOND unreadable reply is not a NONE. Recording it as one is
109
+ // the very conflation this module exists to end — it would put
110
+ // "model has no further questions" on the trail for a run where
111
+ // the model produced two malformed replies.
112
+ if (!isNoneReply(raw)) {
113
+ deps.log?.('second unparseable reply — giving up on this draw');
114
+ hint = null;
115
+ return { kind: 'exhausted', why: 'unparseable' };
116
+ }
117
+ deps.log?.('model has no further questions (NONE)');
118
+ hint = null;
119
+ return { kind: 'exhausted', why: 'none' };
120
+ }
121
+ let picked = pickQuestion(parsed);
122
+ const plain = stripInlineMarkdown(picked.question);
123
+ // The duplicate backstop runs BEFORE any quality re-prompt: a question
124
+ // about to be discarded as a re-ask must not first buy itself an extra
125
+ // child call to be polished.
126
+ if (isDuplicateQuestion(asked, plain)) {
127
+ dupStrikes++;
128
+ deps.log?.(`duplicate question, strike ${dupStrikes}/${MAX_DUP_STRIKES}`);
129
+ hint = null;
130
+ if (dupStrikes >= MAX_DUP_STRIKES)
131
+ return { kind: 'exhausted', why: 'dups' };
132
+ dupHint = DUP_REPROMPT_HINT;
133
+ continue;
134
+ }
135
+ let reprompted = false;
136
+ for (const rule of rules) {
137
+ const h = rule.detect(picked, plain);
138
+ if (h === null)
139
+ continue;
140
+ if (hint === null) {
141
+ deps.log?.(`${rule.id} — one re-prompt`);
142
+ hint = h;
143
+ reprompted = true;
144
+ break;
145
+ }
146
+ // Survived its re-prompt: degrade rather than discard.
147
+ if (rule.repair) {
148
+ deps.log?.(`${rule.id} — survived the re-prompt, repaired`);
149
+ picked = rule.repair(picked);
150
+ }
151
+ }
152
+ if (reprompted)
153
+ continue;
154
+ hint = null;
155
+ dupStrikes = 0;
156
+ dupHint = null;
157
+ asked.push(plain);
158
+ return { kind: 'question', q: picked, plain, index: asked.length };
159
+ }
160
+ }
161
+ /**
162
+ * Clear the DUP strike budget after the caller supplies new context.
163
+ *
164
+ * `/task-plan` lets the user ask the model a question or state a decision
165
+ * mid-session; that is new context, so a generator that had struck out may now
166
+ * have something novel. The plan loop reset its own `dupStrikes` at two sites
167
+ * for exactly this; the budget lives here now, so the reset has to.
168
+ */
169
+ function reopen() {
170
+ dupStrikes = 0;
171
+ dupHint = null;
172
+ }
173
+ return { next, asked: () => asked, reopen };
174
+ }
@@ -1,3 +1,4 @@
1
+ import { type CommandRunner } from './command-run.js';
1
2
  export interface HealthOutcome {
2
3
  /** true → every discovered static check passed, or there was nothing to run.
3
4
  * false → a discovered command actually ran and exited non-zero. */
@@ -29,38 +30,37 @@ export declare function discoverHealthCommands(cwd: string): {
29
30
  ecosystem: string | null;
30
31
  cmds: HealthCommand[];
31
32
  };
33
+ /** Progress hook: called with each command's label as it STARTS, so a caller can
34
+ * keep a live status line naming what is currently running. */
35
+ export type HealthProgress = (command: string) => void;
32
36
  /**
33
37
  * Run the discovered static checks whole-repo and let the real exit codes decide.
34
- * Deterministic and synchronous under the hood (a wrapper keeps the caller async).
35
38
  *
36
39
  * - No manifest / no static command → ok (nothing can regress).
37
- * - A command that CANNOT run (ENOENT / null exit tool not installed) → skipped,
40
+ * - A command that CANNOT run (ENOENT / null exit / 127 inside the chain) → skipped,
38
41
  * treated as an environment gap, not a fault.
39
42
  * - A command that ran and exited non-zero → the first such failure is returned.
40
43
  *
41
- * A generous per-command timeout guards against a wedged tool; a timeout is treated
42
- * as an inconclusive skip, not a fault (it is an environment problem, not the code's).
44
+ * This module owns DISCOVERY and its own output policy; running a command and
45
+ * deciding what its ending MEANS is `command-run.ts`'s. It used to own those too
46
+ * `HealthRun`, `classifyHealthRun` and `spawnHealthCommand` were a second statement
47
+ * of the gap ladder, with no injectable runner, so every classification case in the
48
+ * suite spawned a real shell. `command-run.ts`'s own header notes that this module
49
+ * "had solved exactly this shape years earlier" and the gate never adopted it; this
50
+ * is the adoption, in the other direction.
43
51
  *
44
- * SYNCHRONOUS it blocks the event loop for as long as the project's own lint takes
45
- * (MEASURED: 15s on mx5, 69s on aiz-client), so nothing can render or animate while
46
- * it runs. Gate callers must use {@link runRepoHealthCheckAsync} instead; this stays
47
- * for callers that genuinely have no async seam.
48
- */
49
- export declare function runRepoHealthCheck(cwd: string, timeoutMs?: number): HealthOutcome;
50
- /** Progress hook: called with each command's label as it STARTS, so a caller can
51
- * keep a live status line naming what is currently running. */
52
- export type HealthProgress = (command: string) => void;
53
- /**
54
- * Same check, same verdicts, without blocking the event loop.
52
+ * `captureHealthOutput` stays this module's own: 40 lines of a linter's report is a
53
+ * real difference from `outputTail`'s 400 characters, and that is a parameter, not a
54
+ * thing to unify.
55
55
  *
56
- * The gate runs this immediately after the implementation turn ends, when the impl
57
- * widget has just been cleared the sync version froze the whole TUI there for the
58
- * duration of the project's lint (MEASURED: 0 of 686 expected 100ms timer ticks
59
- * fired during a 69s aiz-client run), so no spinner, clock or queued notify could
60
- * paint. `onCommand` lets the caller name the running command in a live status line.
56
+ * `onCommand` lets the caller name the running command in a live status line — the
57
+ * gate runs this immediately after the implementation turn ends, when the impl
58
+ * widget has just been cleared.
61
59
  */
62
- export declare function runRepoHealthCheckAsync(cwd: string, opts?: {
60
+ export declare function runRepoHealthCheck(cwd: string, opts?: {
63
61
  timeoutMs?: number;
64
62
  signal?: AbortSignal;
65
63
  onCommand?: HealthProgress;
64
+ /** The spawner. Injected so a verdict is testable without a real shell. */
65
+ run?: CommandRunner;
66
66
  }): Promise<HealthOutcome>;
@@ -29,10 +29,10 @@
29
29
  * environment gap, not a code fault, so that command is SKIPPED — only a command that
30
30
  * actually ran and returned non-zero fails the check.
31
31
  */
32
- import { spawn, spawnSync } from 'node:child_process';
33
32
  import { existsSync, readFileSync } from 'node:fs';
34
33
  import * as path from 'node:path';
35
- import { resolveRunner, runnerEnv, isCommandNotFound } from './runner-resolve.js';
34
+ import { resolveRunner, runnerEnv } from './runner-resolve.js';
35
+ import { classifyCommandRun, spawnCommand } from './command-run.js';
36
36
  /** How much of a failing command's output to keep — bounded so a wedged tool that
37
37
  * spews megabytes cannot bloat the trail. stderr leads (a crash trace lives there). */
38
38
  const HEALTH_OUTPUT_MAX_LINES = 40;
@@ -106,141 +106,72 @@ export function discoverHealthCommands(cwd) {
106
106
  }
107
107
  return { ecosystem: null, cmds: [] };
108
108
  }
109
- /**
110
- * Verdict for ONE finished command: 'skip' (environment gap — cannot conclude),
111
- * 'pass', or the FAIL outcome. Shared by the sync and async runners so their
112
- * semantics cannot drift apart — the async runner exists only to stop blocking the
113
- * event loop, and a behaviour difference between the two would be a silent gate
114
- * change rather than a UI fix.
115
- */
116
- function classifyHealthRun(bin, args, ecosystem, r) {
117
- // Tool missing (ENOENT) or killed by timeout → cannot conclude; skip it.
118
- if (r.failedToStart || r.status === null)
119
- return 'skip';
120
- // "Command not found" INSIDE the script chain (e.g. `bun run lint` before
121
- // node_modules exists — seen live failing TASK_0001's first verify). Same
122
- // environment gap as ENOENT, just surfaced through the runner's shell —
123
- // as exit 127 where a posix shell ran it, else by the runner's own wording
124
- // (Windows bun reports the miss itself and exits 1).
125
- if (isCommandNotFound(r.status, `${r.stdout ?? ''}\n${r.stderr ?? ''}`))
126
- return 'skip';
127
- if (r.status !== 0) {
128
- return {
129
- ok: false,
130
- reason: `\`${bin} ${args.join(' ')}\` exited ${r.status}`,
131
- ecosystem,
132
- output: captureHealthOutput(r.stdout, r.stderr)
133
- };
134
- }
135
- return 'pass';
136
- }
137
109
  /** The nothing-to-run outcome, shared by both runners. */
138
110
  function noCommandOutcome(ecosystem) {
139
111
  return { ok: true, reason: 'no repo-wide static-analysis command found', ecosystem, output: '' };
140
112
  }
141
113
  /**
142
114
  * Run the discovered static checks whole-repo and let the real exit codes decide.
143
- * Deterministic and synchronous under the hood (a wrapper keeps the caller async).
144
115
  *
145
116
  * - No manifest / no static command → ok (nothing can regress).
146
- * - A command that CANNOT run (ENOENT / null exit tool not installed) → skipped,
117
+ * - A command that CANNOT run (ENOENT / null exit / 127 inside the chain) → skipped,
147
118
  * treated as an environment gap, not a fault.
148
119
  * - A command that ran and exited non-zero → the first such failure is returned.
149
120
  *
150
- * A generous per-command timeout guards against a wedged tool; a timeout is treated
151
- * as an inconclusive skip, not a fault (it is an environment problem, not the code's).
121
+ * This module owns DISCOVERY and its own output policy; running a command and
122
+ * deciding what its ending MEANS is `command-run.ts`'s. It used to own those too
123
+ * `HealthRun`, `classifyHealthRun` and `spawnHealthCommand` were a second statement
124
+ * of the gap ladder, with no injectable runner, so every classification case in the
125
+ * suite spawned a real shell. `command-run.ts`'s own header notes that this module
126
+ * "had solved exactly this shape years earlier" and the gate never adopted it; this
127
+ * is the adoption, in the other direction.
152
128
  *
153
- * SYNCHRONOUS it blocks the event loop for as long as the project's own lint takes
154
- * (MEASURED: 15s on mx5, 69s on aiz-client), so nothing can render or animate while
155
- * it runs. Gate callers must use {@link runRepoHealthCheckAsync} instead; this stays
156
- * for callers that genuinely have no async seam.
129
+ * `captureHealthOutput` stays this module's own: 40 lines of a linter's report is a
130
+ * real difference from `outputTail`'s 400 characters, and that is a parameter, not a
131
+ * thing to unify.
132
+ *
133
+ * `onCommand` lets the caller name the running command in a live status line — the
134
+ * gate runs this immediately after the implementation turn ends, when the impl
135
+ * widget has just been cleared.
157
136
  */
158
- export function runRepoHealthCheck(cwd, timeoutMs = 600_000) {
137
+ export async function runRepoHealthCheck(cwd, opts = {}) {
159
138
  const { ecosystem, cmds } = discoverHealthCommands(cwd);
160
139
  if (!ecosystem || cmds.length === 0)
161
140
  return noCommandOutcome(ecosystem);
141
+ const run = opts.run ?? spawnCommand;
162
142
  for (const [bin, args] of cmds) {
143
+ opts.onCommand?.(`${bin} ${args.join(' ')}`);
163
144
  // Runner resolution (mx5 run 16): a PATH-stripped environment must not
164
145
  // silently skip the statics when the runner sits at a known install
165
146
  // location; the resolved dir also rides on PATH for the script chain.
166
147
  const runner = resolveRunner(bin);
167
- const r = spawnSync(runner.bin, args, {
148
+ const r = await run({
168
149
  cwd,
169
- encoding: 'utf8',
170
- timeout: timeoutMs,
171
- env: runnerEnv(runner)
150
+ bin: runner.bin,
151
+ args,
152
+ timeoutMs: opts.timeoutMs ?? 600_000,
153
+ env: runnerEnv(runner),
154
+ ...(opts.signal === undefined ? {} : { signal: opts.signal })
172
155
  });
173
- const verdict = classifyHealthRun(bin, args, ecosystem, {
174
- failedToStart: r.error !== undefined,
175
- status: r.status,
176
- stdout: r.stdout ?? '',
177
- stderr: r.stderr ?? ''
178
- });
179
- if (verdict === 'skip' || verdict === 'pass')
156
+ // The DECISION comes from the shared ladder; the OUTPUT is this module's own
157
+ // policy. `captureHealthOutput` keeps 40 lines of a linter's report where the
158
+ // ladder's `tail` keeps 400 characters, and that difference is real — a
159
+ // truncated lint report is unactionable. So the run is classified, not
160
+ // consumed: the verdict decides, the raw streams are what we show.
161
+ // `runtimeGap: false` — this ladder is NARROWER than the gate's. The
162
+ // browser/runtime row was written for the gate's TEST commands; here the
163
+ // commands are lint and typecheck, and its pattern matches ordinary
164
+ // English, so a genuine report quoting "browsers are not installed" would
165
+ // skip the static check and certify the repo healthy.
166
+ const verdict = classifyCommandRun(r, [], { runtimeGap: false });
167
+ if (verdict.outcome !== 'fail')
180
168
  continue;
181
- return verdict;
182
- }
183
- return { ok: true, reason: `${ecosystem}: static checks passed`, ecosystem, output: '' };
184
- }
185
- /** Spawn one health command without blocking the event loop. Mirrors spawnSync's
186
- * result shape (status null when killed, failedToStart on ENOENT). */
187
- function spawnHealthCommand(bin, args, cwd, timeoutMs, signal) {
188
- return new Promise(resolve => {
189
- const runner = resolveRunner(bin);
190
- let stdout = '';
191
- let stderr = '';
192
- let settled = false;
193
- const child = spawn(runner.bin, args, { cwd, env: runnerEnv(runner) });
194
- const done = (r) => {
195
- if (settled)
196
- return;
197
- settled = true;
198
- clearTimeout(timer);
199
- signal?.removeEventListener('abort', onAbort);
200
- resolve(r);
201
- };
202
- const kill = () => {
203
- try {
204
- child.kill('SIGKILL');
205
- }
206
- catch {
207
- /* already gone */
208
- }
169
+ return {
170
+ ok: false,
171
+ reason: `\`${bin} ${args.join(' ')}\` exited ${verdict.status}`,
172
+ ecosystem,
173
+ output: captureHealthOutput(r.stdout, r.stderr)
209
174
  };
210
- const timer = setTimeout(kill, timeoutMs);
211
- timer.unref?.();
212
- const onAbort = () => kill();
213
- signal?.addEventListener('abort', onAbort, { once: true });
214
- child.stdout?.on('data', (d) => {
215
- stdout += d.toString();
216
- });
217
- child.stderr?.on('data', (d) => {
218
- stderr += d.toString();
219
- });
220
- child.on('error', () => done({ failedToStart: true, status: null, stdout, stderr }));
221
- child.on('close', (code) => done({ failedToStart: false, status: code, stdout, stderr }));
222
- });
223
- }
224
- /**
225
- * Same check, same verdicts, without blocking the event loop.
226
- *
227
- * The gate runs this immediately after the implementation turn ends, when the impl
228
- * widget has just been cleared — the sync version froze the whole TUI there for the
229
- * duration of the project's lint (MEASURED: 0 of 686 expected 100ms timer ticks
230
- * fired during a 69s aiz-client run), so no spinner, clock or queued notify could
231
- * paint. `onCommand` lets the caller name the running command in a live status line.
232
- */
233
- export async function runRepoHealthCheckAsync(cwd, opts = {}) {
234
- const { ecosystem, cmds } = discoverHealthCommands(cwd);
235
- if (!ecosystem || cmds.length === 0)
236
- return noCommandOutcome(ecosystem);
237
- for (const [bin, args] of cmds) {
238
- opts.onCommand?.(`${bin} ${args.join(' ')}`);
239
- const r = await spawnHealthCommand(bin, args, cwd, opts.timeoutMs ?? 600_000, opts.signal);
240
- const verdict = classifyHealthRun(bin, args, ecosystem, r);
241
- if (verdict === 'skip' || verdict === 'pass')
242
- continue;
243
- return verdict;
244
175
  }
245
176
  return { ok: true, reason: `${ecosystem}: static checks passed`, ecosystem, output: '' };
246
177
  }
@@ -0,0 +1,77 @@
1
+ /**
2
+ * run-end — how a single /task run ENDED, named once.
3
+ *
4
+ * `TaskRunner.run` returned `void` and never threw, so `runSingleTask` learned
5
+ * what it had just done by RE-READING the task file's front matter, narrowing
6
+ * that to `ok: boolean`, and smuggling the rest out of the `withSession` closure
7
+ * through three mutable captures. Both commands then re-derived a cause the
8
+ * runner already had: `classifyFailure` names the ending exactly, and
9
+ * `handleFailure` threw the name away.
10
+ *
11
+ * The live consequence was a wrong report. `/task-cancel` during a gated run
12
+ * writes `cancelled` to the file; `ok` is `state === 'completed'`, so it is
13
+ * false; `!res.ok` calls `markResumable`, which overwrites `cancelled` with
14
+ * `failed`, and announces a red *"stopped — fix and run /task-resume"*.
15
+ * `/task-auto` hits the same arm for the same input, because its cancel branch
16
+ * only consults a module global that `/task-cancel` never sets.
17
+ *
18
+ * The file is still written — it is what a RESUME reads. What changed is that it
19
+ * is no longer the channel this process uses to talk to itself.
20
+ */
21
+ /** Why a run stopped. Exactly one of these is true of any finished run. */
22
+ export type RunEnd =
23
+ /** Every phase ran and the spec was delivered. */
24
+ {
25
+ kind: 'completed';
26
+ }
27
+ /** The user cancelled — via /task-cancel, ESC, or an aborted signal. */
28
+ | {
29
+ kind: 'cancelled';
30
+ }
31
+ /** A phase threw. `reason` is `FailureClass.reason`, already trimmed. */
32
+ | {
33
+ kind: 'failed';
34
+ reason?: string;
35
+ }
36
+ /** The implementation turn was interrupted and left resumable. */
37
+ | {
38
+ kind: 'interrupted';
39
+ }
40
+ /** No fresh session could be started, so nothing ran at all. */
41
+ | {
42
+ kind: 'no-session';
43
+ };
44
+ export type RunEndKind = RunEnd['kind'];
45
+ /**
46
+ * What a command does about each ending.
47
+ *
48
+ * `resumable` and `announce` are the two facts the two hand-written ladders
49
+ * disagreed about, and the disagreement is where the cancel bug lived: a
50
+ * `cancelled` run was falling into the `failed` arm and being marked resumable.
51
+ * The WORDING stays per-command — `/task` says "resume with /task-resume" where
52
+ * `/task-auto` says "/task-auto-resume" — so only the policy is shared.
53
+ */
54
+ export interface RunEndPolicy {
55
+ /** Mark the task resumable (overwrites its state with `failed`). */
56
+ resumable: boolean;
57
+ /**
58
+ * Does this ending FAIL the plan that contains the task?
59
+ *
60
+ * Strictly narrower than `resumable`, and the distinction is load-bearing: a
61
+ * declined-steer interrupt leaves the inner task resumable but the PLAN in
62
+ * progress, so `/task-auto-resume` re-delivers that task's spec. A fault fails
63
+ * the plan too. Only `/task-auto` reads this — a bare `/task` has no plan.
64
+ */
65
+ failsRun: boolean;
66
+ /** The notify level for this ending. */
67
+ level: 'info' | 'warning' | 'error';
68
+ }
69
+ /**
70
+ * The one table. A run that the USER stopped is not resumable-as-failed: its
71
+ * file already says `cancelled`, and rewriting that to `failed` both lies in the
72
+ * ledger and turns a deliberate stop into a red error the user has to read as a
73
+ * fault.
74
+ */
75
+ export declare const RUN_END_POLICY: Record<RunEndKind, RunEndPolicy>;
76
+ /** Did the run deliver a spec? The single question the old `ok` boolean answered. */
77
+ export declare function runSucceeded(end: RunEnd): boolean;