@mjasnikovs/pi-task 0.38.15 → 0.38.17
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/config/config.d.ts +26 -0
- package/dist/config/config.js +68 -17
- package/dist/shared/child-process.js +9 -16
- package/dist/task/accept-debt.d.ts +7 -5
- package/dist/task/accept-debt.js +18 -14
- package/dist/task/artifact-closure.js +18 -63
- package/dist/task/auto-orchestrator.js +211 -218
- package/dist/task/autofix-ledger.d.ts +113 -0
- package/dist/task/autofix-ledger.js +152 -0
- package/dist/task/boot-probe.d.ts +109 -1
- package/dist/task/boot-probe.js +139 -23
- package/dist/task/child-runner.d.ts +50 -6
- package/dist/task/child-runner.js +48 -69
- package/dist/task/command-run.d.ts +49 -6
- package/dist/task/command-run.js +154 -18
- package/dist/task/coverage-loop.d.ts +11 -0
- package/dist/task/coverage-loop.js +16 -0
- package/dist/task/external-context.d.ts +9 -12
- package/dist/task/external-context.js +5 -5
- package/dist/task/failure-classifier.d.ts +9 -1
- package/dist/task/failure-classifier.js +9 -0
- package/dist/task/final-gate-fix.d.ts +22 -26
- package/dist/task/final-gate-fix.js +16 -31
- package/dist/task/final-gate.d.ts +10 -2
- package/dist/task/final-gate.js +55 -89
- package/dist/task/fix-child.d.ts +64 -0
- package/dist/task/fix-child.js +66 -0
- package/dist/task/gate-deps.js +20 -13
- package/dist/task/lint-fix.d.ts +7 -0
- package/dist/task/lint-fix.js +45 -9
- package/dist/task/orchestrator.d.ts +33 -24
- package/dist/task/orchestrator.js +75 -46
- package/dist/task/phases.d.ts +120 -34
- package/dist/task/phases.js +221 -134
- package/dist/task/plan-orchestrator.js +2 -2
- package/dist/task/plan-rounds.d.ts +86 -0
- package/dist/task/plan-rounds.js +105 -0
- package/dist/task/plan-session.d.ts +31 -21
- package/dist/task/plan-session.js +97 -120
- package/dist/task/qa-transcript.d.ts +100 -0
- package/dist/task/qa-transcript.js +99 -0
- package/dist/task/question-source.d.ts +117 -0
- package/dist/task/question-source.js +174 -0
- package/dist/task/repo-health-check.d.ts +21 -21
- package/dist/task/repo-health-check.js +43 -112
- package/dist/task/run-end.d.ts +77 -0
- package/dist/task/run-end.js +37 -0
- package/dist/task/run-final-gate.js +71 -79
- package/dist/task/serve-entry.js +6 -57
- package/dist/task/shipped-source.d.ts +67 -0
- package/dist/task/shipped-source.js +144 -0
- package/dist/task/task-gates.d.ts +9 -1
- package/dist/task/task-gates.js +27 -6
- package/dist/task/terminal-outcome.d.ts +1 -1
- package/dist/task/terminal-outcome.js +12 -0
- package/dist/task/verify-work.d.ts +46 -0
- package/dist/task/verify-work.js +51 -3
- package/dist/workers/brave-search.d.ts +7 -0
- package/dist/workers/brave-search.js +36 -55
- package/dist/workers/ddg-search.d.ts +1 -1
- package/dist/workers/ddg-search.js +27 -47
- package/dist/workers/docs-core.d.ts +71 -1
- package/dist/workers/docs-core.js +131 -71
- package/dist/workers/exa-search.d.ts +2 -2
- package/dist/workers/exa-search.js +53 -68
- package/dist/workers/html-clean.js +67 -88
- package/dist/workers/http-request.d.ts +74 -0
- package/dist/workers/http-request.js +103 -0
- package/dist/workers/npm-version.js +37 -42
- package/dist/workers/pi-worker-core.d.ts +13 -2
- package/dist/workers/pi-worker-core.js +35 -25
- package/dist/workers/pi-worker-docs.d.ts +1 -1
- package/dist/workers/pi-worker-docs.js +49 -68
- package/dist/workers/pi-worker-fetch.d.ts +1 -1
- package/dist/workers/pi-worker-fetch.js +20 -21
- package/dist/workers/pi-worker-search.js +6 -4
- package/dist/workers/pi-worker.js +5 -4
- package/dist/workers/search-core.d.ts +1 -1
- package/dist/workers/search-core.js +36 -42
- package/dist/workers/search-types.d.ts +13 -0
- package/dist/workers/search-types.js +27 -0
- package/dist/workers/shared.d.ts +51 -11
- package/dist/workers/shared.js +0 -0
- package/dist/workers/worker-channels.d.ts +60 -0
- package/dist/workers/worker-channels.js +98 -0
- package/package.json +1 -1
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Where the NEXT question comes from — the other half of `question-dialog.ts`.
|
|
3
|
+
*
|
|
4
|
+
* `question-dialog.ts` unified the ANSWER side of the adaptive dialogs (the picker
|
|
5
|
+
* cards, the reply mapping) and its own docstring makes the argument for doing so:
|
|
6
|
+
* *"It was written three times… the two mirrors were never converted, and they had
|
|
7
|
+
* already drifted apart in three ways… The next edit to any of them is where the
|
|
8
|
+
* bug lands."* The QUESTION side — generate → parse → pick the real question →
|
|
9
|
+
* dedupe → spend one corrective re-prompt → yield or exhaust — was left behind,
|
|
10
|
+
* and it had drifted between the two loops that use the SAME parser on the SAME
|
|
11
|
+
* prompt format (`plan-prompts.ts` says `parseClarifyList` parses it "UNCHANGED";
|
|
12
|
+
* `auto-prompts.ts` specifies the identical shape).
|
|
13
|
+
*
|
|
14
|
+
* The five drifts, all in clarify's favour of being wrong:
|
|
15
|
+
*
|
|
16
|
+
* 1. **Which entry is the question.** `parseClarifyList` pushes an entry for
|
|
17
|
+
* EVERY numbered line, and the local model writes numbered analysis notes
|
|
18
|
+
* before the question it was asked for (measured live). `pickQuestion` prefers
|
|
19
|
+
* the first entry carrying a `SUGGESTED:` line; clarify took `parsed[0]`
|
|
20
|
+
* blindly, showing the note as the question and losing the recommendation
|
|
21
|
+
* attached further down.
|
|
22
|
+
* 2. **NONE vs unparseable.** The parser returns `[]` for both. Clarify's
|
|
23
|
+
* `if (parsed.length === 0) break` ended the whole clarify — and decomposed the
|
|
24
|
+
* feature with ZERO clarifications — on a formatting slip.
|
|
25
|
+
* 3. **A re-typed sentinel.** `isNoneReply`'s regex was a byte-identical second
|
|
26
|
+
* copy of the parser's own.
|
|
27
|
+
* 4. **Missing SUGGESTED** bought one corrective re-prompt in plan and none in
|
|
28
|
+
* clarify, so clarify showed a card-less question.
|
|
29
|
+
* 5. **The deferral guard.** It exists because an accepted "clarify with the user
|
|
30
|
+
* before proceeding" rode into `/task`'s handoff AS AN AUTHORITATIVE DECISION
|
|
31
|
+
* and produced a task whose VERIFY asserted no source file had changed.
|
|
32
|
+
* Clarify's answers ride into the decompose prompt and the AUTO file with
|
|
33
|
+
* exactly the same authority, and had no guard.
|
|
34
|
+
*
|
|
35
|
+
* NOT unified here: grill's generation loop. It uses a different parser
|
|
36
|
+
* (`parseGrillQuestions`, which yields bare strings) and has no `SUGGESTED` at
|
|
37
|
+
* generation time at all — grill's recommendation comes from `phaseAutoAnswer`
|
|
38
|
+
* one step later, so every quality rule below is inapplicable to it. Folding it in
|
|
39
|
+
* would mean a generic over the parsed shape with one consumer opting out of the
|
|
40
|
+
* entire rule table: a wider interface for less behaviour.
|
|
41
|
+
*/
|
|
42
|
+
import { type ClarifyQuestion } from './parsers.js';
|
|
43
|
+
/**
|
|
44
|
+
* The cap on distinct questions ONE adaptive dialog may ask.
|
|
45
|
+
*
|
|
46
|
+
* Clarify and plan each declared their own `8`, linked only by a comment saying
|
|
47
|
+
* "matches /task-auto's MAX_CLARIFY_QUESTIONS, for the same reason". They bound
|
|
48
|
+
* the same thing for the same reason; this is that reason, once.
|
|
49
|
+
*/
|
|
50
|
+
export declare const MAX_DIALOG_QUESTIONS = 8;
|
|
51
|
+
/** True when the reply is the deliberate "nothing left to ask" sentinel, as
|
|
52
|
+
* opposed to output the parser simply could not read. */
|
|
53
|
+
export declare function isNoneReply(raw: string): boolean;
|
|
54
|
+
/**
|
|
55
|
+
* Which of the parsed entries is the actual question.
|
|
56
|
+
*
|
|
57
|
+
* `parseClarifyList` turns EVERY numbered line into an entry, and the local model
|
|
58
|
+
* sometimes writes a numbered analysis note or two before the question it was
|
|
59
|
+
* asked for (measured live: the first numbered line was a note like
|
|
60
|
+
* "1. gateDebugWriter in orchestrator.ts — wraps a raw append function"). Taking
|
|
61
|
+
* entry 0 blindly then shows the note as the question and loses the SUGGESTED line
|
|
62
|
+
* attached further down. The prompt requires exactly one SUGGESTED, and the parser
|
|
63
|
+
* attaches it to the entry it follows.
|
|
64
|
+
*/
|
|
65
|
+
export declare function pickQuestion<T extends {
|
|
66
|
+
suggested?: string;
|
|
67
|
+
}>(parsed: T[]): T | undefined;
|
|
68
|
+
/**
|
|
69
|
+
* One QUALITY rule: a defect in an otherwise-usable question that is worth exactly
|
|
70
|
+
* one corrective re-prompt.
|
|
71
|
+
*
|
|
72
|
+
* `detect` returns the hint to re-prompt with, or null to pass. `repair` is what
|
|
73
|
+
* to do when the SAME defect survives the re-prompt — a question with a bad
|
|
74
|
+
* default still beats no question, so a rule degrades rather than discards.
|
|
75
|
+
*/
|
|
76
|
+
export interface QuestionRule {
|
|
77
|
+
id: string;
|
|
78
|
+
detect: (q: ClarifyQuestion, plain: string) => string | null;
|
|
79
|
+
/** Applied only when the defect survived its one re-prompt. */
|
|
80
|
+
repair?: (q: ClarifyQuestion) => ClarifyQuestion;
|
|
81
|
+
}
|
|
82
|
+
export interface QuestionSourceDeps {
|
|
83
|
+
/**
|
|
84
|
+
* Ask the model for the next question. `hint` is the corrective re-prompt to
|
|
85
|
+
* prepend, or null. The caller closes over its own transcript and prompt — the
|
|
86
|
+
* source never builds one.
|
|
87
|
+
*/
|
|
88
|
+
generate: (hint: string | null) => Promise<string>;
|
|
89
|
+
/** The corrective re-prompt for a reply the parser could not read. */
|
|
90
|
+
formatHint: string;
|
|
91
|
+
/** Ordered quality rules; each may fire at most once per question. */
|
|
92
|
+
rules?: ReadonlyArray<QuestionRule>;
|
|
93
|
+
cap?: number;
|
|
94
|
+
log?: (msg: string) => void;
|
|
95
|
+
}
|
|
96
|
+
export type NextQuestion = {
|
|
97
|
+
kind: 'question';
|
|
98
|
+
q: ClarifyQuestion;
|
|
99
|
+
plain: string;
|
|
100
|
+
index: number;
|
|
101
|
+
} | {
|
|
102
|
+
kind: 'exhausted';
|
|
103
|
+
why: 'none' | 'cap' | 'dups' | 'unparseable';
|
|
104
|
+
};
|
|
105
|
+
/**
|
|
106
|
+
* A deep module over a state machine that was five mutable locals per site.
|
|
107
|
+
*
|
|
108
|
+
* The interface is one method. Behind it: the cap, the duplicate backstop and its
|
|
109
|
+
* strike budget, the NONE-vs-unparseable distinction, `pickQuestion`, the one-shot
|
|
110
|
+
* budget shared by every quality rule, and the hint precedence between a format
|
|
111
|
+
* re-prompt and a duplicate re-prompt.
|
|
112
|
+
*/
|
|
113
|
+
export declare function makeQuestionSource(deps: QuestionSourceDeps): {
|
|
114
|
+
next: () => Promise<NextQuestion>;
|
|
115
|
+
asked: () => ReadonlyArray<string>;
|
|
116
|
+
reopen: () => void;
|
|
117
|
+
};
|
|
@@ -0,0 +1,174 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Where the NEXT question comes from — the other half of `question-dialog.ts`.
|
|
3
|
+
*
|
|
4
|
+
* `question-dialog.ts` unified the ANSWER side of the adaptive dialogs (the picker
|
|
5
|
+
* cards, the reply mapping) and its own docstring makes the argument for doing so:
|
|
6
|
+
* *"It was written three times… the two mirrors were never converted, and they had
|
|
7
|
+
* already drifted apart in three ways… The next edit to any of them is where the
|
|
8
|
+
* bug lands."* The QUESTION side — generate → parse → pick the real question →
|
|
9
|
+
* dedupe → spend one corrective re-prompt → yield or exhaust — was left behind,
|
|
10
|
+
* and it had drifted between the two loops that use the SAME parser on the SAME
|
|
11
|
+
* prompt format (`plan-prompts.ts` says `parseClarifyList` parses it "UNCHANGED";
|
|
12
|
+
* `auto-prompts.ts` specifies the identical shape).
|
|
13
|
+
*
|
|
14
|
+
* The five drifts, all in clarify's favour of being wrong:
|
|
15
|
+
*
|
|
16
|
+
* 1. **Which entry is the question.** `parseClarifyList` pushes an entry for
|
|
17
|
+
* EVERY numbered line, and the local model writes numbered analysis notes
|
|
18
|
+
* before the question it was asked for (measured live). `pickQuestion` prefers
|
|
19
|
+
* the first entry carrying a `SUGGESTED:` line; clarify took `parsed[0]`
|
|
20
|
+
* blindly, showing the note as the question and losing the recommendation
|
|
21
|
+
* attached further down.
|
|
22
|
+
* 2. **NONE vs unparseable.** The parser returns `[]` for both. Clarify's
|
|
23
|
+
* `if (parsed.length === 0) break` ended the whole clarify — and decomposed the
|
|
24
|
+
* feature with ZERO clarifications — on a formatting slip.
|
|
25
|
+
* 3. **A re-typed sentinel.** `isNoneReply`'s regex was a byte-identical second
|
|
26
|
+
* copy of the parser's own.
|
|
27
|
+
* 4. **Missing SUGGESTED** bought one corrective re-prompt in plan and none in
|
|
28
|
+
* clarify, so clarify showed a card-less question.
|
|
29
|
+
* 5. **The deferral guard.** It exists because an accepted "clarify with the user
|
|
30
|
+
* before proceeding" rode into `/task`'s handoff AS AN AUTHORITATIVE DECISION
|
|
31
|
+
* and produced a task whose VERIFY asserted no source file had changed.
|
|
32
|
+
* Clarify's answers ride into the decompose prompt and the AUTO file with
|
|
33
|
+
* exactly the same authority, and had no guard.
|
|
34
|
+
*
|
|
35
|
+
* NOT unified here: grill's generation loop. It uses a different parser
|
|
36
|
+
* (`parseGrillQuestions`, which yields bare strings) and has no `SUGGESTED` at
|
|
37
|
+
* generation time at all — grill's recommendation comes from `phaseAutoAnswer`
|
|
38
|
+
* one step later, so every quality rule below is inapplicable to it. Folding it in
|
|
39
|
+
* would mean a generic over the parsed shape with one consumer opting out of the
|
|
40
|
+
* entire rule table: a wider interface for less behaviour.
|
|
41
|
+
*/
|
|
42
|
+
import { parseClarifyList } from './parsers.js';
|
|
43
|
+
import { DUP_REPROMPT_HINT, isDuplicateQuestion, MAX_DUP_STRIKES } from './question-dedup.js';
|
|
44
|
+
import { stripInlineMarkdown } from './inline-markdown.js';
|
|
45
|
+
/**
|
|
46
|
+
* The cap on distinct questions ONE adaptive dialog may ask.
|
|
47
|
+
*
|
|
48
|
+
* Clarify and plan each declared their own `8`, linked only by a comment saying
|
|
49
|
+
* "matches /task-auto's MAX_CLARIFY_QUESTIONS, for the same reason". They bound
|
|
50
|
+
* the same thing for the same reason; this is that reason, once.
|
|
51
|
+
*/
|
|
52
|
+
export const MAX_DIALOG_QUESTIONS = 8;
|
|
53
|
+
/** True when the reply is the deliberate "nothing left to ask" sentinel, as
|
|
54
|
+
* opposed to output the parser simply could not read. */
|
|
55
|
+
export function isNoneReply(raw) {
|
|
56
|
+
return /^\s*NONE\s*$/m.test(raw);
|
|
57
|
+
}
|
|
58
|
+
/**
|
|
59
|
+
* Which of the parsed entries is the actual question.
|
|
60
|
+
*
|
|
61
|
+
* `parseClarifyList` turns EVERY numbered line into an entry, and the local model
|
|
62
|
+
* sometimes writes a numbered analysis note or two before the question it was
|
|
63
|
+
* asked for (measured live: the first numbered line was a note like
|
|
64
|
+
* "1. gateDebugWriter in orchestrator.ts — wraps a raw append function"). Taking
|
|
65
|
+
* entry 0 blindly then shows the note as the question and loses the SUGGESTED line
|
|
66
|
+
* attached further down. The prompt requires exactly one SUGGESTED, and the parser
|
|
67
|
+
* attaches it to the entry it follows.
|
|
68
|
+
*/
|
|
69
|
+
export function pickQuestion(parsed) {
|
|
70
|
+
return parsed.find(q => q.suggested !== undefined && q.suggested.length > 0) ?? parsed[0];
|
|
71
|
+
}
|
|
72
|
+
/**
|
|
73
|
+
* A deep module over a state machine that was five mutable locals per site.
|
|
74
|
+
*
|
|
75
|
+
* The interface is one method. Behind it: the cap, the duplicate backstop and its
|
|
76
|
+
* strike budget, the NONE-vs-unparseable distinction, `pickQuestion`, the one-shot
|
|
77
|
+
* budget shared by every quality rule, and the hint precedence between a format
|
|
78
|
+
* re-prompt and a duplicate re-prompt.
|
|
79
|
+
*/
|
|
80
|
+
export function makeQuestionSource(deps) {
|
|
81
|
+
const cap = deps.cap ?? MAX_DIALOG_QUESTIONS;
|
|
82
|
+
const rules = deps.rules ?? [];
|
|
83
|
+
const asked = [];
|
|
84
|
+
let dupStrikes = 0;
|
|
85
|
+
let dupHint = null;
|
|
86
|
+
// The one-shot budget is per QUESTION, not per dialog: a fresh draw starts
|
|
87
|
+
// with every rule available again. `hint` being non-null is also what spends
|
|
88
|
+
// it — a rule may not fire while another rule's re-prompt is in flight, which
|
|
89
|
+
// is what stops two rules ping-ponging a stateless child forever.
|
|
90
|
+
let hint = null;
|
|
91
|
+
async function next() {
|
|
92
|
+
for (;;) {
|
|
93
|
+
if (asked.length >= cap) {
|
|
94
|
+
deps.log?.(`question cap (${cap}) reached`);
|
|
95
|
+
return { kind: 'exhausted', why: 'cap' };
|
|
96
|
+
}
|
|
97
|
+
const raw = await deps.generate(hint ?? dupHint);
|
|
98
|
+
const parsed = parseClarifyList(raw);
|
|
99
|
+
if (parsed.length === 0) {
|
|
100
|
+
// `[]` means BOTH "deliberate NONE" and "could not parse". Ending
|
|
101
|
+
// the dialog on the second is how a formatting slip becomes a
|
|
102
|
+
// feature decomposed with zero clarifications.
|
|
103
|
+
if (!isNoneReply(raw) && hint === null) {
|
|
104
|
+
deps.log?.('unparseable question reply — one format re-prompt');
|
|
105
|
+
hint = deps.formatHint;
|
|
106
|
+
continue;
|
|
107
|
+
}
|
|
108
|
+
// A SECOND unreadable reply is not a NONE. Recording it as one is
|
|
109
|
+
// the very conflation this module exists to end — it would put
|
|
110
|
+
// "model has no further questions" on the trail for a run where
|
|
111
|
+
// the model produced two malformed replies.
|
|
112
|
+
if (!isNoneReply(raw)) {
|
|
113
|
+
deps.log?.('second unparseable reply — giving up on this draw');
|
|
114
|
+
hint = null;
|
|
115
|
+
return { kind: 'exhausted', why: 'unparseable' };
|
|
116
|
+
}
|
|
117
|
+
deps.log?.('model has no further questions (NONE)');
|
|
118
|
+
hint = null;
|
|
119
|
+
return { kind: 'exhausted', why: 'none' };
|
|
120
|
+
}
|
|
121
|
+
let picked = pickQuestion(parsed);
|
|
122
|
+
const plain = stripInlineMarkdown(picked.question);
|
|
123
|
+
// The duplicate backstop runs BEFORE any quality re-prompt: a question
|
|
124
|
+
// about to be discarded as a re-ask must not first buy itself an extra
|
|
125
|
+
// child call to be polished.
|
|
126
|
+
if (isDuplicateQuestion(asked, plain)) {
|
|
127
|
+
dupStrikes++;
|
|
128
|
+
deps.log?.(`duplicate question, strike ${dupStrikes}/${MAX_DUP_STRIKES}`);
|
|
129
|
+
hint = null;
|
|
130
|
+
if (dupStrikes >= MAX_DUP_STRIKES)
|
|
131
|
+
return { kind: 'exhausted', why: 'dups' };
|
|
132
|
+
dupHint = DUP_REPROMPT_HINT;
|
|
133
|
+
continue;
|
|
134
|
+
}
|
|
135
|
+
let reprompted = false;
|
|
136
|
+
for (const rule of rules) {
|
|
137
|
+
const h = rule.detect(picked, plain);
|
|
138
|
+
if (h === null)
|
|
139
|
+
continue;
|
|
140
|
+
if (hint === null) {
|
|
141
|
+
deps.log?.(`${rule.id} — one re-prompt`);
|
|
142
|
+
hint = h;
|
|
143
|
+
reprompted = true;
|
|
144
|
+
break;
|
|
145
|
+
}
|
|
146
|
+
// Survived its re-prompt: degrade rather than discard.
|
|
147
|
+
if (rule.repair) {
|
|
148
|
+
deps.log?.(`${rule.id} — survived the re-prompt, repaired`);
|
|
149
|
+
picked = rule.repair(picked);
|
|
150
|
+
}
|
|
151
|
+
}
|
|
152
|
+
if (reprompted)
|
|
153
|
+
continue;
|
|
154
|
+
hint = null;
|
|
155
|
+
dupStrikes = 0;
|
|
156
|
+
dupHint = null;
|
|
157
|
+
asked.push(plain);
|
|
158
|
+
return { kind: 'question', q: picked, plain, index: asked.length };
|
|
159
|
+
}
|
|
160
|
+
}
|
|
161
|
+
/**
|
|
162
|
+
* Clear the DUP strike budget after the caller supplies new context.
|
|
163
|
+
*
|
|
164
|
+
* `/task-plan` lets the user ask the model a question or state a decision
|
|
165
|
+
* mid-session; that is new context, so a generator that had struck out may now
|
|
166
|
+
* have something novel. The plan loop reset its own `dupStrikes` at two sites
|
|
167
|
+
* for exactly this; the budget lives here now, so the reset has to.
|
|
168
|
+
*/
|
|
169
|
+
function reopen() {
|
|
170
|
+
dupStrikes = 0;
|
|
171
|
+
dupHint = null;
|
|
172
|
+
}
|
|
173
|
+
return { next, asked: () => asked, reopen };
|
|
174
|
+
}
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { type CommandRunner } from './command-run.js';
|
|
1
2
|
export interface HealthOutcome {
|
|
2
3
|
/** true → every discovered static check passed, or there was nothing to run.
|
|
3
4
|
* false → a discovered command actually ran and exited non-zero. */
|
|
@@ -29,38 +30,37 @@ export declare function discoverHealthCommands(cwd: string): {
|
|
|
29
30
|
ecosystem: string | null;
|
|
30
31
|
cmds: HealthCommand[];
|
|
31
32
|
};
|
|
33
|
+
/** Progress hook: called with each command's label as it STARTS, so a caller can
|
|
34
|
+
* keep a live status line naming what is currently running. */
|
|
35
|
+
export type HealthProgress = (command: string) => void;
|
|
32
36
|
/**
|
|
33
37
|
* Run the discovered static checks whole-repo and let the real exit codes decide.
|
|
34
|
-
* Deterministic and synchronous under the hood (a wrapper keeps the caller async).
|
|
35
38
|
*
|
|
36
39
|
* - No manifest / no static command → ok (nothing can regress).
|
|
37
|
-
* - A command that CANNOT run (ENOENT / null exit
|
|
40
|
+
* - A command that CANNOT run (ENOENT / null exit / 127 inside the chain) → skipped,
|
|
38
41
|
* treated as an environment gap, not a fault.
|
|
39
42
|
* - A command that ran and exited non-zero → the first such failure is returned.
|
|
40
43
|
*
|
|
41
|
-
*
|
|
42
|
-
*
|
|
44
|
+
* This module owns DISCOVERY and its own output policy; running a command and
|
|
45
|
+
* deciding what its ending MEANS is `command-run.ts`'s. It used to own those too —
|
|
46
|
+
* `HealthRun`, `classifyHealthRun` and `spawnHealthCommand` were a second statement
|
|
47
|
+
* of the gap ladder, with no injectable runner, so every classification case in the
|
|
48
|
+
* suite spawned a real shell. `command-run.ts`'s own header notes that this module
|
|
49
|
+
* "had solved exactly this shape years earlier" and the gate never adopted it; this
|
|
50
|
+
* is the adoption, in the other direction.
|
|
43
51
|
*
|
|
44
|
-
*
|
|
45
|
-
*
|
|
46
|
-
*
|
|
47
|
-
* for callers that genuinely have no async seam.
|
|
48
|
-
*/
|
|
49
|
-
export declare function runRepoHealthCheck(cwd: string, timeoutMs?: number): HealthOutcome;
|
|
50
|
-
/** Progress hook: called with each command's label as it STARTS, so a caller can
|
|
51
|
-
* keep a live status line naming what is currently running. */
|
|
52
|
-
export type HealthProgress = (command: string) => void;
|
|
53
|
-
/**
|
|
54
|
-
* Same check, same verdicts, without blocking the event loop.
|
|
52
|
+
* `captureHealthOutput` stays this module's own: 40 lines of a linter's report is a
|
|
53
|
+
* real difference from `outputTail`'s 400 characters, and that is a parameter, not a
|
|
54
|
+
* thing to unify.
|
|
55
55
|
*
|
|
56
|
-
*
|
|
57
|
-
*
|
|
58
|
-
*
|
|
59
|
-
* fired during a 69s aiz-client run), so no spinner, clock or queued notify could
|
|
60
|
-
* paint. `onCommand` lets the caller name the running command in a live status line.
|
|
56
|
+
* `onCommand` lets the caller name the running command in a live status line — the
|
|
57
|
+
* gate runs this immediately after the implementation turn ends, when the impl
|
|
58
|
+
* widget has just been cleared.
|
|
61
59
|
*/
|
|
62
|
-
export declare function
|
|
60
|
+
export declare function runRepoHealthCheck(cwd: string, opts?: {
|
|
63
61
|
timeoutMs?: number;
|
|
64
62
|
signal?: AbortSignal;
|
|
65
63
|
onCommand?: HealthProgress;
|
|
64
|
+
/** The spawner. Injected so a verdict is testable without a real shell. */
|
|
65
|
+
run?: CommandRunner;
|
|
66
66
|
}): Promise<HealthOutcome>;
|
|
@@ -29,10 +29,10 @@
|
|
|
29
29
|
* environment gap, not a code fault, so that command is SKIPPED — only a command that
|
|
30
30
|
* actually ran and returned non-zero fails the check.
|
|
31
31
|
*/
|
|
32
|
-
import { spawn, spawnSync } from 'node:child_process';
|
|
33
32
|
import { existsSync, readFileSync } from 'node:fs';
|
|
34
33
|
import * as path from 'node:path';
|
|
35
|
-
import { resolveRunner, runnerEnv
|
|
34
|
+
import { resolveRunner, runnerEnv } from './runner-resolve.js';
|
|
35
|
+
import { classifyCommandRun, spawnCommand } from './command-run.js';
|
|
36
36
|
/** How much of a failing command's output to keep — bounded so a wedged tool that
|
|
37
37
|
* spews megabytes cannot bloat the trail. stderr leads (a crash trace lives there). */
|
|
38
38
|
const HEALTH_OUTPUT_MAX_LINES = 40;
|
|
@@ -106,141 +106,72 @@ export function discoverHealthCommands(cwd) {
|
|
|
106
106
|
}
|
|
107
107
|
return { ecosystem: null, cmds: [] };
|
|
108
108
|
}
|
|
109
|
-
/**
|
|
110
|
-
* Verdict for ONE finished command: 'skip' (environment gap — cannot conclude),
|
|
111
|
-
* 'pass', or the FAIL outcome. Shared by the sync and async runners so their
|
|
112
|
-
* semantics cannot drift apart — the async runner exists only to stop blocking the
|
|
113
|
-
* event loop, and a behaviour difference between the two would be a silent gate
|
|
114
|
-
* change rather than a UI fix.
|
|
115
|
-
*/
|
|
116
|
-
function classifyHealthRun(bin, args, ecosystem, r) {
|
|
117
|
-
// Tool missing (ENOENT) or killed by timeout → cannot conclude; skip it.
|
|
118
|
-
if (r.failedToStart || r.status === null)
|
|
119
|
-
return 'skip';
|
|
120
|
-
// "Command not found" INSIDE the script chain (e.g. `bun run lint` before
|
|
121
|
-
// node_modules exists — seen live failing TASK_0001's first verify). Same
|
|
122
|
-
// environment gap as ENOENT, just surfaced through the runner's shell —
|
|
123
|
-
// as exit 127 where a posix shell ran it, else by the runner's own wording
|
|
124
|
-
// (Windows bun reports the miss itself and exits 1).
|
|
125
|
-
if (isCommandNotFound(r.status, `${r.stdout ?? ''}\n${r.stderr ?? ''}`))
|
|
126
|
-
return 'skip';
|
|
127
|
-
if (r.status !== 0) {
|
|
128
|
-
return {
|
|
129
|
-
ok: false,
|
|
130
|
-
reason: `\`${bin} ${args.join(' ')}\` exited ${r.status}`,
|
|
131
|
-
ecosystem,
|
|
132
|
-
output: captureHealthOutput(r.stdout, r.stderr)
|
|
133
|
-
};
|
|
134
|
-
}
|
|
135
|
-
return 'pass';
|
|
136
|
-
}
|
|
137
109
|
/** The nothing-to-run outcome, shared by both runners. */
|
|
138
110
|
function noCommandOutcome(ecosystem) {
|
|
139
111
|
return { ok: true, reason: 'no repo-wide static-analysis command found', ecosystem, output: '' };
|
|
140
112
|
}
|
|
141
113
|
/**
|
|
142
114
|
* Run the discovered static checks whole-repo and let the real exit codes decide.
|
|
143
|
-
* Deterministic and synchronous under the hood (a wrapper keeps the caller async).
|
|
144
115
|
*
|
|
145
116
|
* - No manifest / no static command → ok (nothing can regress).
|
|
146
|
-
* - A command that CANNOT run (ENOENT / null exit
|
|
117
|
+
* - A command that CANNOT run (ENOENT / null exit / 127 inside the chain) → skipped,
|
|
147
118
|
* treated as an environment gap, not a fault.
|
|
148
119
|
* - A command that ran and exited non-zero → the first such failure is returned.
|
|
149
120
|
*
|
|
150
|
-
*
|
|
151
|
-
*
|
|
121
|
+
* This module owns DISCOVERY and its own output policy; running a command and
|
|
122
|
+
* deciding what its ending MEANS is `command-run.ts`'s. It used to own those too —
|
|
123
|
+
* `HealthRun`, `classifyHealthRun` and `spawnHealthCommand` were a second statement
|
|
124
|
+
* of the gap ladder, with no injectable runner, so every classification case in the
|
|
125
|
+
* suite spawned a real shell. `command-run.ts`'s own header notes that this module
|
|
126
|
+
* "had solved exactly this shape years earlier" and the gate never adopted it; this
|
|
127
|
+
* is the adoption, in the other direction.
|
|
152
128
|
*
|
|
153
|
-
*
|
|
154
|
-
*
|
|
155
|
-
*
|
|
156
|
-
*
|
|
129
|
+
* `captureHealthOutput` stays this module's own: 40 lines of a linter's report is a
|
|
130
|
+
* real difference from `outputTail`'s 400 characters, and that is a parameter, not a
|
|
131
|
+
* thing to unify.
|
|
132
|
+
*
|
|
133
|
+
* `onCommand` lets the caller name the running command in a live status line — the
|
|
134
|
+
* gate runs this immediately after the implementation turn ends, when the impl
|
|
135
|
+
* widget has just been cleared.
|
|
157
136
|
*/
|
|
158
|
-
export function runRepoHealthCheck(cwd,
|
|
137
|
+
export async function runRepoHealthCheck(cwd, opts = {}) {
|
|
159
138
|
const { ecosystem, cmds } = discoverHealthCommands(cwd);
|
|
160
139
|
if (!ecosystem || cmds.length === 0)
|
|
161
140
|
return noCommandOutcome(ecosystem);
|
|
141
|
+
const run = opts.run ?? spawnCommand;
|
|
162
142
|
for (const [bin, args] of cmds) {
|
|
143
|
+
opts.onCommand?.(`${bin} ${args.join(' ')}`);
|
|
163
144
|
// Runner resolution (mx5 run 16): a PATH-stripped environment must not
|
|
164
145
|
// silently skip the statics when the runner sits at a known install
|
|
165
146
|
// location; the resolved dir also rides on PATH for the script chain.
|
|
166
147
|
const runner = resolveRunner(bin);
|
|
167
|
-
const r =
|
|
148
|
+
const r = await run({
|
|
168
149
|
cwd,
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
150
|
+
bin: runner.bin,
|
|
151
|
+
args,
|
|
152
|
+
timeoutMs: opts.timeoutMs ?? 600_000,
|
|
153
|
+
env: runnerEnv(runner),
|
|
154
|
+
...(opts.signal === undefined ? {} : { signal: opts.signal })
|
|
172
155
|
});
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
156
|
+
// The DECISION comes from the shared ladder; the OUTPUT is this module's own
|
|
157
|
+
// policy. `captureHealthOutput` keeps 40 lines of a linter's report where the
|
|
158
|
+
// ladder's `tail` keeps 400 characters, and that difference is real — a
|
|
159
|
+
// truncated lint report is unactionable. So the run is classified, not
|
|
160
|
+
// consumed: the verdict decides, the raw streams are what we show.
|
|
161
|
+
// `runtimeGap: false` — this ladder is NARROWER than the gate's. The
|
|
162
|
+
// browser/runtime row was written for the gate's TEST commands; here the
|
|
163
|
+
// commands are lint and typecheck, and its pattern matches ordinary
|
|
164
|
+
// English, so a genuine report quoting "browsers are not installed" would
|
|
165
|
+
// skip the static check and certify the repo healthy.
|
|
166
|
+
const verdict = classifyCommandRun(r, [], { runtimeGap: false });
|
|
167
|
+
if (verdict.outcome !== 'fail')
|
|
180
168
|
continue;
|
|
181
|
-
return
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
* result shape (status null when killed, failedToStart on ENOENT). */
|
|
187
|
-
function spawnHealthCommand(bin, args, cwd, timeoutMs, signal) {
|
|
188
|
-
return new Promise(resolve => {
|
|
189
|
-
const runner = resolveRunner(bin);
|
|
190
|
-
let stdout = '';
|
|
191
|
-
let stderr = '';
|
|
192
|
-
let settled = false;
|
|
193
|
-
const child = spawn(runner.bin, args, { cwd, env: runnerEnv(runner) });
|
|
194
|
-
const done = (r) => {
|
|
195
|
-
if (settled)
|
|
196
|
-
return;
|
|
197
|
-
settled = true;
|
|
198
|
-
clearTimeout(timer);
|
|
199
|
-
signal?.removeEventListener('abort', onAbort);
|
|
200
|
-
resolve(r);
|
|
201
|
-
};
|
|
202
|
-
const kill = () => {
|
|
203
|
-
try {
|
|
204
|
-
child.kill('SIGKILL');
|
|
205
|
-
}
|
|
206
|
-
catch {
|
|
207
|
-
/* already gone */
|
|
208
|
-
}
|
|
169
|
+
return {
|
|
170
|
+
ok: false,
|
|
171
|
+
reason: `\`${bin} ${args.join(' ')}\` exited ${verdict.status}`,
|
|
172
|
+
ecosystem,
|
|
173
|
+
output: captureHealthOutput(r.stdout, r.stderr)
|
|
209
174
|
};
|
|
210
|
-
const timer = setTimeout(kill, timeoutMs);
|
|
211
|
-
timer.unref?.();
|
|
212
|
-
const onAbort = () => kill();
|
|
213
|
-
signal?.addEventListener('abort', onAbort, { once: true });
|
|
214
|
-
child.stdout?.on('data', (d) => {
|
|
215
|
-
stdout += d.toString();
|
|
216
|
-
});
|
|
217
|
-
child.stderr?.on('data', (d) => {
|
|
218
|
-
stderr += d.toString();
|
|
219
|
-
});
|
|
220
|
-
child.on('error', () => done({ failedToStart: true, status: null, stdout, stderr }));
|
|
221
|
-
child.on('close', (code) => done({ failedToStart: false, status: code, stdout, stderr }));
|
|
222
|
-
});
|
|
223
|
-
}
|
|
224
|
-
/**
|
|
225
|
-
* Same check, same verdicts, without blocking the event loop.
|
|
226
|
-
*
|
|
227
|
-
* The gate runs this immediately after the implementation turn ends, when the impl
|
|
228
|
-
* widget has just been cleared — the sync version froze the whole TUI there for the
|
|
229
|
-
* duration of the project's lint (MEASURED: 0 of 686 expected 100ms timer ticks
|
|
230
|
-
* fired during a 69s aiz-client run), so no spinner, clock or queued notify could
|
|
231
|
-
* paint. `onCommand` lets the caller name the running command in a live status line.
|
|
232
|
-
*/
|
|
233
|
-
export async function runRepoHealthCheckAsync(cwd, opts = {}) {
|
|
234
|
-
const { ecosystem, cmds } = discoverHealthCommands(cwd);
|
|
235
|
-
if (!ecosystem || cmds.length === 0)
|
|
236
|
-
return noCommandOutcome(ecosystem);
|
|
237
|
-
for (const [bin, args] of cmds) {
|
|
238
|
-
opts.onCommand?.(`${bin} ${args.join(' ')}`);
|
|
239
|
-
const r = await spawnHealthCommand(bin, args, cwd, opts.timeoutMs ?? 600_000, opts.signal);
|
|
240
|
-
const verdict = classifyHealthRun(bin, args, ecosystem, r);
|
|
241
|
-
if (verdict === 'skip' || verdict === 'pass')
|
|
242
|
-
continue;
|
|
243
|
-
return verdict;
|
|
244
175
|
}
|
|
245
176
|
return { ok: true, reason: `${ecosystem}: static checks passed`, ecosystem, output: '' };
|
|
246
177
|
}
|
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* run-end — how a single /task run ENDED, named once.
|
|
3
|
+
*
|
|
4
|
+
* `TaskRunner.run` returned `void` and never threw, so `runSingleTask` learned
|
|
5
|
+
* what it had just done by RE-READING the task file's front matter, narrowing
|
|
6
|
+
* that to `ok: boolean`, and smuggling the rest out of the `withSession` closure
|
|
7
|
+
* through three mutable captures. Both commands then re-derived a cause the
|
|
8
|
+
* runner already had: `classifyFailure` names the ending exactly, and
|
|
9
|
+
* `handleFailure` threw the name away.
|
|
10
|
+
*
|
|
11
|
+
* The live consequence was a wrong report. `/task-cancel` during a gated run
|
|
12
|
+
* writes `cancelled` to the file; `ok` is `state === 'completed'`, so it is
|
|
13
|
+
* false; `!res.ok` calls `markResumable`, which overwrites `cancelled` with
|
|
14
|
+
* `failed`, and announces a red *"stopped — fix and run /task-resume"*.
|
|
15
|
+
* `/task-auto` hits the same arm for the same input, because its cancel branch
|
|
16
|
+
* only consults a module global that `/task-cancel` never sets.
|
|
17
|
+
*
|
|
18
|
+
* The file is still written — it is what a RESUME reads. What changed is that it
|
|
19
|
+
* is no longer the channel this process uses to talk to itself.
|
|
20
|
+
*/
|
|
21
|
+
/** Why a run stopped. Exactly one of these is true of any finished run. */
|
|
22
|
+
export type RunEnd =
|
|
23
|
+
/** Every phase ran and the spec was delivered. */
|
|
24
|
+
{
|
|
25
|
+
kind: 'completed';
|
|
26
|
+
}
|
|
27
|
+
/** The user cancelled — via /task-cancel, ESC, or an aborted signal. */
|
|
28
|
+
| {
|
|
29
|
+
kind: 'cancelled';
|
|
30
|
+
}
|
|
31
|
+
/** A phase threw. `reason` is `FailureClass.reason`, already trimmed. */
|
|
32
|
+
| {
|
|
33
|
+
kind: 'failed';
|
|
34
|
+
reason?: string;
|
|
35
|
+
}
|
|
36
|
+
/** The implementation turn was interrupted and left resumable. */
|
|
37
|
+
| {
|
|
38
|
+
kind: 'interrupted';
|
|
39
|
+
}
|
|
40
|
+
/** No fresh session could be started, so nothing ran at all. */
|
|
41
|
+
| {
|
|
42
|
+
kind: 'no-session';
|
|
43
|
+
};
|
|
44
|
+
export type RunEndKind = RunEnd['kind'];
|
|
45
|
+
/**
|
|
46
|
+
* What a command does about each ending.
|
|
47
|
+
*
|
|
48
|
+
* `resumable` and `announce` are the two facts the two hand-written ladders
|
|
49
|
+
* disagreed about, and the disagreement is where the cancel bug lived: a
|
|
50
|
+
* `cancelled` run was falling into the `failed` arm and being marked resumable.
|
|
51
|
+
* The WORDING stays per-command — `/task` says "resume with /task-resume" where
|
|
52
|
+
* `/task-auto` says "/task-auto-resume" — so only the policy is shared.
|
|
53
|
+
*/
|
|
54
|
+
export interface RunEndPolicy {
|
|
55
|
+
/** Mark the task resumable (overwrites its state with `failed`). */
|
|
56
|
+
resumable: boolean;
|
|
57
|
+
/**
|
|
58
|
+
* Does this ending FAIL the plan that contains the task?
|
|
59
|
+
*
|
|
60
|
+
* Strictly narrower than `resumable`, and the distinction is load-bearing: a
|
|
61
|
+
* declined-steer interrupt leaves the inner task resumable but the PLAN in
|
|
62
|
+
* progress, so `/task-auto-resume` re-delivers that task's spec. A fault fails
|
|
63
|
+
* the plan too. Only `/task-auto` reads this — a bare `/task` has no plan.
|
|
64
|
+
*/
|
|
65
|
+
failsRun: boolean;
|
|
66
|
+
/** The notify level for this ending. */
|
|
67
|
+
level: 'info' | 'warning' | 'error';
|
|
68
|
+
}
|
|
69
|
+
/**
|
|
70
|
+
* The one table. A run that the USER stopped is not resumable-as-failed: its
|
|
71
|
+
* file already says `cancelled`, and rewriting that to `failed` both lies in the
|
|
72
|
+
* ledger and turns a deliberate stop into a red error the user has to read as a
|
|
73
|
+
* fault.
|
|
74
|
+
*/
|
|
75
|
+
export declare const RUN_END_POLICY: Record<RunEndKind, RunEndPolicy>;
|
|
76
|
+
/** Did the run deliver a spec? The single question the old `ok` boolean answered. */
|
|
77
|
+
export declare function runSucceeded(end: RunEnd): boolean;
|