pi-gauntlet 4.13.1 → 5.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +10 -0
- package/README.md +12 -8
- package/agents/code-reviewer.md +1 -1
- package/agents/conformance-reviewer.md +7 -6
- package/agents/implementer.md +4 -3
- package/agents/spec-reviewer.md +4 -3
- package/package.json +1 -1
- package/skills/chase-bug/SKILL.md +320 -0
- package/skills/dispatching-parallel-agents/SKILL.md +1 -1
- package/skills/requesting-code-review/SKILL.md +2 -0
- package/skills/requesting-code-review/code-reviewer.md +5 -2
- package/skills/subagent-driven-development/SKILL.md +13 -9
- package/skills/subagent-driven-development/code-quality-reviewer-prompt.md +1 -0
- package/skills/subagent-driven-development/implementer-prompt.md +7 -1
- package/skills/subagent-driven-development/spec-reviewer-prompt.md +2 -0
- package/skills/test-driven-development/SKILL.md +3 -3
- package/skills/verification-before-completion/reference/conformance-check.md +14 -8
- package/skills/writing-plans/SKILL.md +5 -1
- package/skills/writing-skills/SKILL.md +3 -3
- package/skills/systematic-debugging/SKILL.md +0 -151
- package/skills/systematic-debugging/condition-based-waiting-example.ts +0 -158
- package/skills/systematic-debugging/condition-based-waiting.md +0 -115
- package/skills/systematic-debugging/defense-in-depth.md +0 -122
- package/skills/systematic-debugging/find-polluter.sh +0 -63
- package/skills/systematic-debugging/reference/rationalizations.md +0 -61
- package/skills/systematic-debugging/root-cause-tracing.md +0 -169
|
@@ -53,11 +53,11 @@ Before the first task, enter the implement phase: `phase_tracker({ action: "star
|
|
|
53
53
|
|
|
54
54
|
For each task in `plan_tracker`:
|
|
55
55
|
|
|
56
|
-
1. **Dispatch implementer.** Pass the full task text + scene-setting context. Don't make the subagent re-read the plan.
|
|
56
|
+
1. **Dispatch implementer.** Pass the full task text + scene-setting context + the task's plan-declared test commands as `SCOPED_TEST_COMMANDS` (or `none`). Don't make the subagent re-read the plan.
|
|
57
57
|
2. **Handle implementer status** (see below).
|
|
58
58
|
3. **Dispatch spec reviewer.** Verify the diff matches the spec — nothing missing, nothing extra.
|
|
59
59
|
4. If spec reviewer finds gaps → re-dispatch implementer to fix → re-review. Loop until ✅, within [Fix-Loop Rounds](#fix-loop-rounds).
|
|
60
|
-
5. **Dispatch code-quality reviewer.** Only after spec is ✅. Skip for doc-only tasks (every file in the task's `Files:` block documentation-only) — SR-only, same exemption as doc-only waves.
|
|
60
|
+
5. **Dispatch code-quality reviewer.** Only after spec is ✅. Skip for doc-only tasks (every file in the task's `Files:` block documentation-only) — SR-only, same exemption as doc-only waves. Pass `SCOPED_TEST_COMMANDS` = the task's plan-declared commands.
|
|
61
61
|
6. If quality reviewer finds issues → re-dispatch implementer → re-review. Loop until ✅, within [Fix-Loop Rounds](#fix-loop-rounds).
|
|
62
62
|
7. Mark task complete in `plan_tracker`.
|
|
63
63
|
|
|
@@ -71,6 +71,8 @@ One rule governs both review loops - spec-compliance and code-quality - in seque
|
|
|
71
71
|
|
|
72
72
|
**Fix fan-out.** When the triggering review's `Parallel-safe:` line certifies a `disjoint` group of ≥ 2 findings, dispatch that fix round per `dispatching-parallel-agents` "Fix fan-out"; the fan-out counts as **one** fix against this budget, its scoped test gate is the consuming task/wave's plan-declared commands, and one re-review of the integrated delta follows.
|
|
73
73
|
|
|
74
|
+
Every fix re-dispatch (implementer) and code-review re-review carries the consuming task/wave's `SCOPED_TEST_COMMANDS`; spec-reviewer re-reviews carry none - SR never executes.
|
|
75
|
+
|
|
74
76
|
**The sequence.** Each review that finds issues is a decision point: read the `TRAJECTORY:` line before dispatching anything (review 1 has no line - on issues, dispatch fix 1). Any clean review ends the loop.
|
|
75
77
|
|
|
76
78
|
1. **Review 1** (first review - no sentinel). Issues -> dispatch fix 1.
|
|
@@ -128,13 +130,13 @@ When in doubt, default. Don't downgrade reviewers — false negatives are expens
|
|
|
128
130
|
|
|
129
131
|
```ts
|
|
130
132
|
// implementer
|
|
131
|
-
subagent({ agent: "implementer", task: "<task text + context + status protocol>" })
|
|
133
|
+
subagent({ agent: "implementer", task: "<task text + context + SCOPED_TEST_COMMANDS + status protocol>" })
|
|
132
134
|
|
|
133
135
|
// spec compliance
|
|
134
136
|
subagent({ agent: "spec-reviewer", task: "<diff range + spec excerpt + ask: does this match?>" })
|
|
135
137
|
|
|
136
138
|
// code quality
|
|
137
|
-
subagent({ agent: "code-reviewer", task: "<diff range + ask: production-ready?>" })
|
|
139
|
+
subagent({ agent: "code-reviewer", task: "<diff range + SCOPED_TEST_COMMANDS (task commands; wave: union; whole-diff: none) + ask: production-ready?>" })
|
|
138
140
|
|
|
139
141
|
// closing-loop conformance (origin vs deliverable) — its OWN dispatch, never fused with code quality
|
|
140
142
|
// model: call gauntlet_setting({ key: "closureReview" }) first; use the returned model (omit model: if undefined to inherit) and maxFixRounds
|
|
@@ -167,10 +169,10 @@ Auto-selected at handoff by `writing-plans` (any wave with ≥2 tasks) when the
|
|
|
167
169
|
|
|
168
170
|
1. **Independence check.** Parse the wave's tasks' `Files:` blocks; assert pairwise-disjoint (mechanical). Runtime-resource disjointness (DB/schema, port, fixture, external service, shared temp path) is not machine-checkable here — trust the plan's wave grouping, which `writing-plans`' D5 contract guarantees. Either kind of overlap → the wave is mis-grouped; run those tasks as sequential single-task waves and note it.
|
|
169
171
|
2. **Fan out.** One parallel dispatch (shape below): `implementer` per task, `context: "fresh"`, `worktree: true`. Each returns a status + a patch.
|
|
170
|
-
3. **Status + spec review per task.** Parse each `DONE`/`BLOCKED`/etc. (see [Implementer Status](#implementer-status)) **first**. Then **dispatch a `spec-reviewer` per accepted patch** (`DONE`, or a `DONE_WITH_CONCERNS` you proceeded with) in one parallel fan-out — `context: "fresh"`, `cwd: <this worktree>`, **no `worktree` flag** (read-only) — each passed its task text, the returned **patch diff**, and the absolute spec path. Review is **diff-based**: the diff's hunks carry `file:line`, and test execution is
|
|
172
|
+
3. **Status + spec review per task.** Parse each `DONE`/`BLOCKED`/etc. (see [Implementer Status](#implementer-status)) **first**. Then **dispatch a `spec-reviewer` per accepted patch** (`DONE`, or a `DONE_WITH_CONCERNS` you proceeded with) in one parallel fan-out — `context: "fresh"`, `cwd: <this worktree>`, **no `worktree` flag** (read-only) — each passed its task text, the returned **patch diff**, and the absolute spec path. Review is **diff-based**: the diff's hunks carry `file:line`, and test execution is never the reviewer's job - in either mode (persona rule; the wave test gate in step 5 runs the wave's declared test commands). Inline verdicts are fine at normal wave sizes; large waves use `output:` + `outputMode: "file-only"` to keep verdicts out of your context. **Re-dispatch by cause:** `BLOCKED`/`NEEDS_CONTEXT` per the [Implementer Status](#implementer-status) matrix; a **spec gap** re-dispatches the implementer (fresh, `worktree: true`) carrying the prior patch + the reviewer's findings, the new patch superseding the old at step 4. Loop until accepted + spec ✅, within [Fix-Loop Rounds](#fix-loop-rounds), same as sequential.
|
|
171
173
|
4. **Integrate.** `git apply` each task's patch sequentially onto HEAD. Apply fails = textual conflict → drop that task, finish the rest, re-run the dropped task sequentially on the updated HEAD.
|
|
172
174
|
5. **Test gate.** Run the union of the wave's tasks' declared test commands on the integrated tree — the full verification set is the verify phase's job, run once. Failure = semantic conflict or bug → re-run the offending task sequentially, else fix per [When a Subagent Fails](#when-a-subagent-fails).
|
|
173
|
-
6. **Quality review.** Code-quality review on the integrated wave diff; loop fixes to ✅ within [Fix-Loop Rounds](#fix-loop-rounds), same as sequential. Skip for doc-only waves (SR-only per the commit precondition below).
|
|
175
|
+
6. **Quality review.** CR binds to the wave: exactly one **initial** code-review dispatch per code-touching wave, over the integrated wave diff - never per task within a wave, never batched across waves. Subsequent dispatches within the wave are re-reviews triggered only by findings, per Fix-Loop Rounds. Pass `SCOPED_TEST_COMMANDS` = the union of the wave's tasks' declared commands. Code-quality review on the integrated wave diff; loop fixes to ✅ within [Fix-Loop Rounds](#fix-loop-rounds), same as sequential. Skip for doc-only waves (SR-only per the commit precondition below).
|
|
174
176
|
7. **Commit the wave.** Leaves a clean tree; the next wave's children branch from this commit and so see the integrated work.
|
|
175
177
|
|
|
176
178
|
**Two-stage review is preserved:** spec review per task (pre-integration, dispatched `spec-reviewer` — not inline), quality review per wave (post-integration). A wave commit requires one spec-review verdict per accepted task, plus one code-review verdict on the integrated diff for waves that touch code. A doc-only wave (every task's `Files:` block documentation-only, per `writing-plans`' Wave Grouping) is SR-only — the CR gate does not apply.
|
|
@@ -189,8 +191,8 @@ subagent({
|
|
|
189
191
|
concurrency: 4, // default; cap = wave size
|
|
190
192
|
tasks: [
|
|
191
193
|
// do NOT set per-task cwd under worktree:true — it must equal the top-level cwd or the run errors
|
|
192
|
-
{ agent: "implementer", task: "<task text + owned files + status protocol>", output: "wave1-task1.md" },
|
|
193
|
-
{ agent: "implementer", task: "<task text + owned files + status protocol>", output: "wave1-task2.md" },
|
|
194
|
+
{ agent: "implementer", task: "<task text + owned files + SCOPED_TEST_COMMANDS + status protocol>", output: "wave1-task1.md" },
|
|
195
|
+
{ agent: "implementer", task: "<task text + owned files + SCOPED_TEST_COMMANDS + status protocol>", output: "wave1-task2.md" },
|
|
194
196
|
],
|
|
195
197
|
})
|
|
196
198
|
```
|
|
@@ -215,7 +217,7 @@ For the fan-out + worktree + patch-integration + conflict mechanics, see `dispat
|
|
|
215
217
|
## After All Tasks Complete
|
|
216
218
|
|
|
217
219
|
0. Call `phase_tracker({ action: "start", phase: "verify" })`. (The `implement` phase was started at execution start and auto-completes from `plan_tracker` once all tasks are done; this flow runs its own verify gate instead of `/skill:verification-before-completion`, so it must mark verify itself.)
|
|
218
|
-
1. **Run the whole-diff code review.** Dispatch `/skill:requesting-code-review` against the worktree's full diff vs `main` (already covered in [The Process](#the-process) step "After all tasks"). Address Critical and Moderate findings before handoff. (Consumers wanting an in-flow project-specific audit re-add it as an explicit step in the gauntlet overrides file (see Project overrides), or run `/self-audit` manually.)
|
|
220
|
+
1. **Run the whole-diff code review.** Dispatch `/skill:requesting-code-review` against the worktree's full diff vs `main` (already covered in [The Process](#the-process) step "After all tasks"). Address Critical and Moderate findings before handoff. (Consumers wanting an in-flow project-specific audit re-add it as an explicit step in the gauntlet overrides file (see Project overrides), or run `/self-audit` manually.) Pass `SCOPED_TEST_COMMANDS: none` - the verify phase's full run (step 2) is the orchestrator's.
|
|
219
221
|
2. **Run the full verification set — once.** Read the plan header's `**Verification:**` line and run it: tests + style + format (a single bundling entrypoint, or the listed individual commands). Green output is the fresh evidence verify requires; this is the only full run before conformance — task and wave gates ran scoped commands only. After conformance fix rounds land, re-run the set before re-dispatching the gate.
|
|
220
222
|
3. **Close the loop — conformance check.** The review in step 1 is plan-vs-code (single-step); it inherits any requirement the plan already dropped. Before marking verify complete, dispatch a fresh-context **`conformance-reviewer`** — its **own** dispatch, never fused into the step-1 review — to confront the deliverable (code **and** docs) against the *origin* — the spec **and** the original prompt — per `verification-before-completion/reference/conformance-check.md`. Pass the spec path, the verbatim original prompt, and the full diff. Follow that reference for the partition rule, concern decomposition, and fix-loop mechanics; do not reimplement them here. The fix loop may drive `plan_tracker` to surface fix-wave progress (task naming and lifecycle per conformance-check.md's fix loop / the Fix fan-out Progress rule); it never calls `phase_tracker`. Call `phase_tracker({ action: "complete", phase: "verify" })` only when the reference says the handoff is durably complete: either a current `CONFORMS` result, or a current `## Closure / conformance` inventory whose carried-open concerns all come from valid deferred gaps, including `recommended: fix` gaps carried open because a declared precondition made the fix loop unavailable (`maxFixRounds: 0`, or no eligible named-branch worktree). A started positive-cap fix loop that blocks, fails, or exhausts its rounds with an open `fix` gap is escalation, not completion; on escalation, do not complete verify, stop and report.
|
|
221
223
|
4. Summarize what was implemented (tasks completed, files changed, test counts, code-review verdict). Emit the `## Closure / conformance` block exactly as defined in `verification-before-completion/reference/conformance-check.md`: it must open with the two-line sentinel (`status: CONFORMS (0 open)` or `status: GAPS (N open)`, then `audited-base: <full HEAD SHA>`), then carry the exact durable concern schema by reference with no renamed or reformatted fields. `finishing-a-development-branch` Step 3.5 consumes that block verbatim.
|
|
@@ -238,6 +240,8 @@ For the fan-out + worktree + patch-integration + conflict mechanics, see `dispat
|
|
|
238
240
|
- Starting on main without explicit user consent
|
|
239
241
|
- Dispatching `code-reviewer` before every one of the wave's spec-review verdicts has landed (including fusing SR+CR into one parallel call)
|
|
240
242
|
- Dispatching fixes sequentially on a clean HEAD despite a ≥ 2-ID `disjoint` group in the review's `Parallel-safe:` line
|
|
243
|
+
- Dispatching `code-reviewer` per task inside a wave (CR binds to the integrated wave diff)
|
|
244
|
+
- Dispatching an implementer or code-reviewer without a `SCOPED_TEST_COMMANDS` value (commands or `none`)
|
|
241
245
|
- About to run the full verification entrypoint during the implement phase — task and wave gates run scoped, plan-declared commands only; the full set belongs to verify
|
|
242
246
|
|
|
243
247
|
## Integration
|
|
@@ -14,6 +14,7 @@ Dispatch a subagent with the code-reviewer template:
|
|
|
14
14
|
PLAN_OR_REQUIREMENTS: Task N from [plan-file]
|
|
15
15
|
BASE_SHA: [commit before task]
|
|
16
16
|
HEAD_SHA: [current commit]
|
|
17
|
+
SCOPED_TEST_COMMANDS: [the consuming task's plan-declared commands; wave reviews: the union of the wave's tasks' declared commands; `none` for the whole-diff verify-phase review]
|
|
17
18
|
```
|
|
18
19
|
|
|
19
20
|
**In addition to standard code quality concerns, the reviewer should check:**
|
|
@@ -31,13 +31,19 @@ Dispatch a subagent with this prompt:
|
|
|
31
31
|
Once you're clear on requirements:
|
|
32
32
|
1. Implement exactly what the task specifies
|
|
33
33
|
2. Write tests (following TDD — failing test first for production code)
|
|
34
|
-
3. Verify
|
|
34
|
+
3. Verify with the commands under SCOPED_TEST_COMMANDS (if `none`, state that)
|
|
35
35
|
4. Commit your work
|
|
36
36
|
5. Self-review (see below)
|
|
37
37
|
6. Report back
|
|
38
38
|
|
|
39
39
|
Work from: [directory]
|
|
40
40
|
|
|
41
|
+
SCOPED_TEST_COMMANDS: [the task's plan-declared test commands, verbatim | none]
|
|
42
|
+
|
|
43
|
+
Run ONLY these commands for verification. Never run a repo-wide suite,
|
|
44
|
+
linter, or type-checker. If the value is `none`, run nothing and say so
|
|
45
|
+
in your report.
|
|
46
|
+
|
|
41
47
|
**While you work:** If you encounter something unexpected or unclear, **ask questions**.
|
|
42
48
|
It's always OK to pause and clarify. Don't guess or make assumptions.
|
|
43
49
|
|
|
@@ -38,6 +38,8 @@ Dispatch a subagent with this prompt:
|
|
|
38
38
|
|
|
39
39
|
- **Read code and compare to spec: yes**
|
|
40
40
|
- **Edit, create, or delete any files: NO**
|
|
41
|
+
- **Run tests, linters, or type-checkers: NO.** Never run tests, linters, or type-checkers. Your evidence is the diff and the files you read.
|
|
42
|
+
- **Code-quality opinions (naming, design, complexity, test aesthetics, style): NO.** Those belong to code-reviewer. Report only spec-vs-implementation deltas.
|
|
41
43
|
- You are a reviewer. Your output is a written report listing what matches and what doesn't.
|
|
42
44
|
- If you find issues, describe them — do NOT fix them.
|
|
43
45
|
|
|
@@ -117,11 +117,11 @@ Don't add features, refactor other code, or "improve" beyond what the test requi
|
|
|
117
117
|
|
|
118
118
|
Run the test. Confirm:
|
|
119
119
|
- New test passes
|
|
120
|
-
-
|
|
120
|
+
- The task's scoped commands pass (full-suite verification belongs to the verify phase)
|
|
121
121
|
- Output is pristine (no errors, no warnings)
|
|
122
122
|
|
|
123
123
|
**Test fails?** Fix code, not test.
|
|
124
|
-
**
|
|
124
|
+
**Scoped commands fail?** Fix now — don't move on with broken tests.
|
|
125
125
|
|
|
126
126
|
### REFACTOR — Clean Up
|
|
127
127
|
|
|
@@ -176,7 +176,7 @@ Before marking work complete:
|
|
|
176
176
|
- [ ] Watched each test fail before implementing
|
|
177
177
|
- [ ] Each test failed for expected reason (feature missing, not typo)
|
|
178
178
|
- [ ] Wrote minimal code to pass each test
|
|
179
|
-
- [ ]
|
|
179
|
+
- [ ] The task's scoped commands pass (full suite belongs to the verify phase)
|
|
180
180
|
- [ ] Output pristine (no errors, warnings)
|
|
181
181
|
- [ ] Tests use real code (mocks only if unavoidable)
|
|
182
182
|
- [ ] Edge cases and errors covered
|
|
@@ -76,8 +76,9 @@ Default: **1 requirement source = 1 spec = code covering every requirement.**
|
|
|
76
76
|
The requirement source is whatever sits at the top of the priority table — a
|
|
77
77
|
ticket if there is one, otherwise the spec + original prompt. No ticket is fine;
|
|
78
78
|
spec + prompt is a first-class source, not a degraded one. "Every requirement" =
|
|
79
|
-
explicit acceptance criteria / spec clauses **+**
|
|
80
|
-
comments, or inline in the prompt
|
|
79
|
+
explicit acceptance criteria / spec clauses **+** quotable notes (written
|
|
80
|
+
sentences in the ticket body, comments, or inline in the prompt - quotable
|
|
81
|
+
verbatim, never derived inferences). Source and solution must end in sync.
|
|
81
82
|
|
|
82
83
|
Multi-spec effort → allowed **only if the spec explicitly says** it covers a
|
|
83
84
|
defined subset and names the deferred requirements. Silent partial coverage = failure.
|
|
@@ -152,7 +153,9 @@ Per round:
|
|
|
152
153
|
group of ≥ 2 gaps (per the report's `Parallel-safe:` line) fixes in one parallel
|
|
153
154
|
dispatch — one `implementer` per gap (fresh context, `worktree: true`, `cwd` =
|
|
154
155
|
the conformance worktree, task = the gap block verbatim with `touched-files` as
|
|
155
|
-
the ownership boundary)
|
|
156
|
+
the ownership boundary). The dispatch adds `SCOPED_TEST_COMMANDS` to the gap
|
|
157
|
+
block: the gap-relevant plan-declared commands, or `none` (the round's test
|
|
158
|
+
gate owns execution). `conflicts` pairs serialize. Gaps outside any ≥ 2-ID
|
|
156
159
|
`disjoint` group run sequentially as before. Then dispatch `spec-reviewer` per
|
|
157
160
|
gap on the gap-block reference contract below. Task lifecycle: mark `in_progress` at
|
|
158
161
|
dispatch; `complete` is deferred until the gap's patch is successfully
|
|
@@ -167,7 +170,7 @@ Per round:
|
|
|
167
170
|
integrated changes. A `BLOCKED`/`NEEDS_CONTEXT` return surfaces to the user.
|
|
168
171
|
4. **Test gate** on the integrated tree, using the project's canonical test
|
|
169
172
|
command. A failure re-enters the failure-handling rules above.
|
|
170
|
-
5. **`code-reviewer` once** on the round's cumulative fix delta (not per gap).
|
|
173
|
+
5. **`code-reviewer` once** on the round's cumulative fix delta (not per gap), with `SCOPED_TEST_COMMANDS` = the round's gap-relevant commands, or `none` (the round's test gate owns execution).
|
|
171
174
|
6. **Re-audit**: re-dispatch `conformance-reviewer` over the fixes **plus** the
|
|
172
175
|
regression guard (any prior-`DELIVERED` requirement whose `evidence` file
|
|
173
176
|
the fix diff touched). Pass the full prior conformance report (every row,
|
|
@@ -269,7 +272,9 @@ is an unmet-delivery fact, not
|
|
|
269
272
|
an external blocker** — never relabel missing implementation evidence as a
|
|
270
273
|
blocker. A malformed structured reviewer gap block — missing its stable `Gn`
|
|
271
274
|
label or any required field (`verdict`, `origin`, `evidence`, `remediation`,
|
|
272
|
-
`touched-files`, `touched-resources`, `recommended`) —
|
|
275
|
+
`touched-files`, `touched-resources`, `recommended`) — or, for any non-`UNAUTHORIZED`
|
|
276
|
+
gap (including re-audit blocks), an `origin` lacking a locator or a nonempty
|
|
277
|
+
quoted fragment — triggers a **fresh
|
|
273
278
|
audit**; a complete structured reviewer gap block does not — the orchestrator
|
|
274
279
|
decomposes it or emits the indivisible fallback.
|
|
275
280
|
|
|
@@ -411,7 +416,7 @@ must map each token to its titled concern or gap before asking for input.
|
|
|
411
416
|
```text
|
|
412
417
|
G1 - Source-image validation is incomplete
|
|
413
418
|
verdict: PARTIAL
|
|
414
|
-
origin: <requirement source and clause>
|
|
419
|
+
origin: <requirement source and clause> - "<quoted clause>"
|
|
415
420
|
evidence: <current file:line or observed state>
|
|
416
421
|
blocker: <specific blocker, or none>
|
|
417
422
|
touched-files: <paths or unknown>
|
|
@@ -421,7 +426,7 @@ G1 - Source-image validation is incomplete
|
|
|
421
426
|
G1/C1 - End-to-end OCR output has not been validated
|
|
422
427
|
unresolved: <plain statement of the concern>
|
|
423
428
|
impact: <why it matters to the current workflow>
|
|
424
|
-
origin: <requirement source and clause, narrowed from the gap origin; or none (scope creep) for UNAUTHORIZED
|
|
429
|
+
origin: <requirement source and clause, narrowed from the gap origin> - "<quoted clause>"; or none (scope creep) for UNAUTHORIZED
|
|
425
430
|
remediation: <concern-scoped remediation action>
|
|
426
431
|
evidence: <concern-specific evidence or blocker>
|
|
427
432
|
touched-files: <concern-scoped paths, narrowed from the gap; or unknown>
|
|
@@ -474,7 +479,8 @@ branch integration options.
|
|
|
474
479
|
## Checklist
|
|
475
480
|
|
|
476
481
|
- [ ] Located canonical requirements (spec → prompt → ticket fallback)
|
|
477
|
-
- [ ] Enumerated every requirement: explicit ACs / spec clauses +
|
|
482
|
+
- [ ] Enumerated every requirement: explicit ACs / spec clauses + quotable notes + inline prompt reqs (verbatim-quotable only)
|
|
483
|
+
- [ ] Every non-UNAUTHORIZED gap's origin carries locator + verbatim quote
|
|
478
484
|
- [ ] Checked spec ↔ prompt/ticket drift; reconciled any divergence
|
|
479
485
|
- [ ] Each requirement mapped to where it's satisfied (code/doc) + evidence
|
|
480
486
|
- [ ] Multi-spec? Subset declared in spec; deferred ACs noted as out of scope
|
|
@@ -128,7 +128,8 @@ If you can't list the files, the spec isn't ready. Send it back to `/skill:brain
|
|
|
128
128
|
Group tasks into **waves** so the executor can parallelize independent work (see `subagent-driven-development` Parallel-Wave Mode). A wave is a maximal set of tasks that (a) have no ordering dependency on each other, (b) own **pairwise-disjoint files**, and (c) contend on **no shared mutable runtime resource** (same DB/schema, port, fixture file, external service, shared temp path).
|
|
129
129
|
|
|
130
130
|
- Tasks nest under `## Wave N — <label>` headers; `### Task N` headers sit inside a wave.
|
|
131
|
-
- A wave with one task is legal
|
|
131
|
+
- Group independent tasks into the same wave by default. A wave with one task is legal **only with a named-blocker justification**: a body line directly under the `## Wave N — <label>` header, `Solo: <reason>`, where the reason names the blocking task/wave, the contended runtime resource, or `lone remaining task` (reserved for the genuinely final unmatched task; doc-only trailing waves qualify). Category-only justifications ("dependency" with no named task) do not satisfy the rule.
|
|
132
|
+
- A pure dependency chain yields one task per wave — no parallelism, which is correct; each such wave carries its `Solo:` line naming the prior-wave dependency.
|
|
132
133
|
- Each wave after the first states its dependency on prior waves.
|
|
133
134
|
|
|
134
135
|
**File-ownership contract.** The per-task `**Files:**` block *is* the ownership declaration — no new syntax. Rule: **within a wave, the union of every task's declared paths must be pairwise disjoint.** Globs are allowed for `Modify` when exact paths are unknown, but must not overlap another same-wave task's paths. A task that must touch another's file belongs in a later wave.
|
|
@@ -152,6 +153,8 @@ Parallel-safe: Tasks 1–3 own disjoint files (see each task's Files block).
|
|
|
152
153
|
|
|
153
154
|
## Wave 2 — Wire-up
|
|
154
155
|
|
|
156
|
+
Solo: Task 4 depends on Wave 1 Task 1's API (named-blocker justification).
|
|
157
|
+
|
|
155
158
|
Depends on Wave 1: Task 4 consumes the API introduced by Task 1.
|
|
156
159
|
|
|
157
160
|
### Task 4: ...
|
|
@@ -269,6 +272,7 @@ After drafting the plan and before announcing it complete, run three checks your
|
|
|
269
272
|
- **Placeholder scan.** Grep the doc for `TODO`, `TBD`, `xxx`, `[fill in]`, `<example>`, `etc.`, "probably", "something like". Resolve or convert each into an explicit Open Question.
|
|
270
273
|
- **Type / API consistency.** Function signatures and field names that appear in multiple tasks must match exactly. The plan is its own contract — internal contradictions surface as bugs during execution.
|
|
271
274
|
- **Wave disjointness.** For every multi-task wave, confirm the tasks' `Files:` sets are pairwise disjoint **and** that no two tasks contend on a shared mutable runtime resource (DB/schema, port, fixture, external service, shared temp path). Either kind of overlap = mis-grouped wave; split or re-order before handoff.
|
|
275
|
+
- **Solo-wave justification.** Every single-task wave carries a `Solo:` line naming its specific blocker. A solo wave without one is mis-grouped or under-justified — merge it or justify it before handoff.
|
|
272
276
|
- **Scoped-test coverage.** Every code-touching wave declares at least one scoped test command; only doc-only waves may have none.
|
|
273
277
|
- **Header-only entrypoint.** The full verification entrypoint appears only in the plan header's `**Verification:**` line. Grep the task body for the header's command string — expect zero hits.
|
|
274
278
|
|
|
@@ -40,9 +40,9 @@ reference/ # optional progressive-disclosure files
|
|
|
40
40
|
<supporting>.md # prompt templates (dispatch payloads)
|
|
41
41
|
```
|
|
42
42
|
|
|
43
|
-
`reference/` is the pi pattern for keeping SKILL.md tight while still shipping deep guidance. See `.pi/skills/test-driven-development/reference/`
|
|
43
|
+
`reference/` is the pi pattern for keeping SKILL.md tight while still shipping deep guidance. See `.pi/skills/test-driven-development/reference/` for a working example.
|
|
44
44
|
|
|
45
|
-
Prompt templates and other dispatch payloads - files filled in and passed wholesale into a subagent `task` - live as siblings of SKILL.md, not under `reference/`. See `requesting-code-review/code-reviewer.md` and the three `subagent-driven-development/*-prompt.md` files. The decision criterion is destination, not format: a file passed wholesale into a subagent's `task` is a sibling; a file read at a decision point for deep guidance, examples, or rationale is `reference/`. Some older skills (`
|
|
45
|
+
Prompt templates and other dispatch payloads - files filled in and passed wholesale into a subagent `task` - live as siblings of SKILL.md, not under `reference/`. See `requesting-code-review/code-reviewer.md` and the three `subagent-driven-development/*-prompt.md` files. The decision criterion is destination, not format: a file passed wholesale into a subagent's `task` is a sibling; a file read at a decision point for deep guidance, examples, or rationale is `reference/`. Some older skills (`test-driven-development`) keep deep-guidance `*.md` files flat as siblings, predating the `reference/` convention (obra/superpowers lineage) - that is descriptive history, not a mandate to move them.
|
|
46
46
|
|
|
47
47
|
### Reference Files Bundled With This Skill
|
|
48
48
|
|
|
@@ -151,7 +151,7 @@ description: Use when implementing any feature or bugfix, before writing impleme
|
|
|
151
151
|
Use skill name with explicit requirement markers. **Never** force-load with `@` syntax — that burns context before the file is needed.
|
|
152
152
|
|
|
153
153
|
- ✅ `**REQUIRED SUB-SKILL:** Use /skill:test-driven-development`
|
|
154
|
-
- ✅ `**REQUIRED BACKGROUND:** You MUST understand /skill:
|
|
154
|
+
- ✅ `**REQUIRED BACKGROUND:** You MUST understand /skill:verification-before-completion`
|
|
155
155
|
- ✅ `> **Related skills:** Pair with /skill:verification-before-completion`
|
|
156
156
|
- ❌ `@.pi/skills/test-driven-development/SKILL.md`
|
|
157
157
|
|
|
@@ -1,151 +0,0 @@
|
|
|
1
|
-
---
|
|
2
|
-
name: systematic-debugging
|
|
3
|
-
description: Use when encountering any bug, test failure, or unexpected behavior, before proposing fixes
|
|
4
|
-
---
|
|
5
|
-
|
|
6
|
-
> **Related skills:** Write a failing test for the bug with `/skill:test-driven-development`. Verify the fix with `/skill:verification-before-completion`.
|
|
7
|
-
|
|
8
|
-
# Systematic Debugging
|
|
9
|
-
|
|
10
|
-
## Overview
|
|
11
|
-
|
|
12
|
-
Random fixes waste time and create new bugs. Quick patches mask underlying issues.
|
|
13
|
-
|
|
14
|
-
**Core principle:** ALWAYS find root cause before attempting fixes. Symptom fixes are failure.
|
|
15
|
-
|
|
16
|
-
**Violating the letter of this process is violating the spirit of debugging.**
|
|
17
|
-
|
|
18
|
-
Debug discipline is enforced by this skill, not by runtime hooks. The pi `verify-before-ship` extension only gates ship commands; it does not track investigation patterns. Hold yourself to the process below.
|
|
19
|
-
|
|
20
|
-
## The Iron Law
|
|
21
|
-
|
|
22
|
-
```
|
|
23
|
-
NO FIXES WITHOUT ROOT CAUSE INVESTIGATION FIRST
|
|
24
|
-
```
|
|
25
|
-
|
|
26
|
-
If you haven't completed Phase 1, you cannot propose fixes.
|
|
27
|
-
|
|
28
|
-
## When to Use
|
|
29
|
-
|
|
30
|
-
Use for ANY technical issue: test failures, bugs, unexpected behavior, performance problems, build failures, integration issues.
|
|
31
|
-
|
|
32
|
-
**Use this ESPECIALLY when:**
|
|
33
|
-
- Under time pressure (emergencies make guessing tempting)
|
|
34
|
-
- "Just one quick fix" seems obvious
|
|
35
|
-
- You've already tried multiple fixes
|
|
36
|
-
- Previous fix didn't work
|
|
37
|
-
- You don't fully understand the issue
|
|
38
|
-
|
|
39
|
-
**Don't skip when:**
|
|
40
|
-
- Issue seems simple (simple bugs have root causes too)
|
|
41
|
-
- You're in a hurry (rushing guarantees rework)
|
|
42
|
-
|
|
43
|
-
## The Four Phases
|
|
44
|
-
|
|
45
|
-
You MUST complete each phase before proceeding to the next.
|
|
46
|
-
|
|
47
|
-
### Phase 1: Root Cause Investigation
|
|
48
|
-
|
|
49
|
-
**BEFORE attempting ANY fix:**
|
|
50
|
-
|
|
51
|
-
1. **Read Error Messages Carefully** — Don't skip past errors or warnings. Read stack traces completely. Note line numbers, file paths, error codes.
|
|
52
|
-
|
|
53
|
-
2. **Reproduce Consistently** — Can you trigger it reliably? What are the exact steps? If not reproducible → gather more data, don't guess.
|
|
54
|
-
|
|
55
|
-
3. **Check Recent Changes** — Git diff, recent commits, new dependencies, config changes, environmental differences.
|
|
56
|
-
|
|
57
|
-
4. **Gather Evidence in Multi-Component Systems** — For each component boundary: log what enters, what exits, verify config propagation. Run once to see WHERE it breaks, then investigate that component.
|
|
58
|
-
|
|
59
|
-
**Example (multi-layer system):**
|
|
60
|
-
```bash
|
|
61
|
-
# Layer 1: Workflow
|
|
62
|
-
echo "=== Secrets available: ==="
|
|
63
|
-
echo "IDENTITY: ${IDENTITY:+SET}${IDENTITY:-UNSET}"
|
|
64
|
-
|
|
65
|
-
# Layer 2: Build script
|
|
66
|
-
echo "=== Env vars in build script: ==="
|
|
67
|
-
env | grep IDENTITY || echo "IDENTITY not in environment"
|
|
68
|
-
|
|
69
|
-
# Layer 3: Signing
|
|
70
|
-
echo "=== Keychain state: ==="
|
|
71
|
-
security list-keychains
|
|
72
|
-
security find-identity -v
|
|
73
|
-
```
|
|
74
|
-
**This reveals:** Which layer fails (e.g., secrets → workflow ✓, workflow → build ✗)
|
|
75
|
-
|
|
76
|
-
5. **Trace Data Flow** — Where does the bad value originate? What called this with the bad value? Keep tracing up until you find the source. Fix at source, not at symptom. See `root-cause-tracing.md` for the complete technique.
|
|
77
|
-
|
|
78
|
-
### Phase 2: Pattern Analysis
|
|
79
|
-
|
|
80
|
-
1. **Find Working Examples** — Locate similar working code in same codebase.
|
|
81
|
-
2. **Compare Against References** — Read reference implementation COMPLETELY. Don't skim.
|
|
82
|
-
3. **Identify Differences** — List every difference, however small. Don't assume "that can't matter."
|
|
83
|
-
4. **Understand Dependencies** — What components, settings, config, environment does this need?
|
|
84
|
-
|
|
85
|
-
### Phase 3: Hypothesis and Testing
|
|
86
|
-
|
|
87
|
-
1. **Form Single Hypothesis** — State clearly: "I think X is the root cause because Y." Be specific, not vague.
|
|
88
|
-
2. **Test Minimally** — Make the SMALLEST possible change. One variable at a time. Don't fix multiple things at once.
|
|
89
|
-
3. **Verify Before Continuing** — Did it work? Yes → Phase 4. No → Form NEW hypothesis. DON'T add more fixes on top.
|
|
90
|
-
4. **When You Don't Know** — Say "I don't understand X." Don't pretend to know. Ask for help. Research more. The escape valve is real: an honest "I'm stuck on X" beats a confident wrong fix every time.
|
|
91
|
-
|
|
92
|
-
### Phase 4: Implementation
|
|
93
|
-
|
|
94
|
-
1. **Create Failing Test Case** — Use `/skill:test-driven-development` for writing proper failing tests. MUST have before fixing.
|
|
95
|
-
|
|
96
|
-
2. **Implement Single Fix** — ONE change at a time. No "while I'm here" improvements. No bundled refactoring.
|
|
97
|
-
|
|
98
|
-
3. **Verify Fix** — Test passes? No other tests broken? Issue actually resolved?
|
|
99
|
-
|
|
100
|
-
4. **If Fix Doesn't Work:**
|
|
101
|
-
- If < 3 attempts: Return to Phase 1, re-analyze with new information
|
|
102
|
-
- **If ≥ 3 attempts: STOP (see below)**
|
|
103
|
-
|
|
104
|
-
### When 3+ Fixes Fail: Question Architecture
|
|
105
|
-
|
|
106
|
-
**This is NOT a failed hypothesis — it's a wrong architecture.**
|
|
107
|
-
|
|
108
|
-
Pattern indicating architectural problem:
|
|
109
|
-
- Each fix reveals new shared state/coupling in different places
|
|
110
|
-
- Fixes require "massive refactoring" to implement
|
|
111
|
-
- Each fix creates new symptoms elsewhere
|
|
112
|
-
|
|
113
|
-
**STOP and question fundamentals:**
|
|
114
|
-
- Is this pattern fundamentally sound?
|
|
115
|
-
- Are we sticking with it through sheer inertia?
|
|
116
|
-
- Should we refactor architecture vs. continue fixing symptoms?
|
|
117
|
-
|
|
118
|
-
**Discuss with your human partner before attempting more fixes.**
|
|
119
|
-
|
|
120
|
-
## Red Flags and Rationalizations
|
|
121
|
-
|
|
122
|
-
Read `reference/rationalizations.md` for the full table of excuses and the partner-signal redirections. Short version:
|
|
123
|
-
|
|
124
|
-
- "Quick fix for now, investigate later" → return to Phase 1.
|
|
125
|
-
- "Just try changing X and see if it works" → return to Phase 1.
|
|
126
|
-
- "It's probably X, let me fix that" → return to Phase 1.
|
|
127
|
-
- "One more fix attempt" after 2+ failures → question architecture, don't fix again.
|
|
128
|
-
- Each fix reveals a new problem in a different place → question architecture.
|
|
129
|
-
|
|
130
|
-
## When Process Reveals "No Root Cause"
|
|
131
|
-
|
|
132
|
-
If investigation reveals issue is truly environmental, timing-dependent, or external:
|
|
133
|
-
1. Document what you investigated
|
|
134
|
-
2. Implement appropriate handling (retry, timeout, error message)
|
|
135
|
-
3. Add monitoring/logging for future investigation
|
|
136
|
-
|
|
137
|
-
**But:** 95% of "no root cause" cases are incomplete investigation.
|
|
138
|
-
|
|
139
|
-
## Supporting Techniques
|
|
140
|
-
|
|
141
|
-
These techniques are part of systematic debugging and available in this directory:
|
|
142
|
-
|
|
143
|
-
- **`root-cause-tracing.md`** — Trace bugs backward through call stack to find original trigger
|
|
144
|
-
- **`defense-in-depth.md`** — Add validation at multiple layers after finding root cause
|
|
145
|
-
- **`condition-based-waiting.md`** — Replace arbitrary timeouts with condition polling
|
|
146
|
-
|
|
147
|
-
Read directly when needed: `reference/rationalizations.md` and the supporting `*.md` files in this directory.
|
|
148
|
-
|
|
149
|
-
## Project overrides
|
|
150
|
-
|
|
151
|
-
If a gauntlet overrides file exists - checked in order: `.pi/gauntlet-overrides.md`, `<repo root>/gauntlet-overrides.md`, `<repo root>/doc/gauntlet-overrides.md`; first found wins - read it. Any sections relevant to this skill — by name match, by topic (routing, verification, worktrees, etc.), or by workflow convention — override or extend the instructions above. Project-local `AGENTS.md` is already in context — check it for project-specific routing tables, service paths, and verification commands.
|
|
@@ -1,158 +0,0 @@
|
|
|
1
|
-
// Complete implementation of condition-based waiting utilities
|
|
2
|
-
// From: Lace test infrastructure improvements (2025-10-03)
|
|
3
|
-
// Context: Fixed 15 flaky tests by replacing arbitrary timeouts
|
|
4
|
-
|
|
5
|
-
import type { ThreadManager } from "~/threads/thread-manager";
|
|
6
|
-
import type { LaceEvent, LaceEventType } from "~/threads/types";
|
|
7
|
-
|
|
8
|
-
/**
|
|
9
|
-
* Wait for a specific event type to appear in thread
|
|
10
|
-
*
|
|
11
|
-
* @param threadManager - The thread manager to query
|
|
12
|
-
* @param threadId - Thread to check for events
|
|
13
|
-
* @param eventType - Type of event to wait for
|
|
14
|
-
* @param timeoutMs - Maximum time to wait (default 5000ms)
|
|
15
|
-
* @returns Promise resolving to the first matching event
|
|
16
|
-
*
|
|
17
|
-
* Example:
|
|
18
|
-
* await waitForEvent(threadManager, agentThreadId, 'TOOL_RESULT');
|
|
19
|
-
*/
|
|
20
|
-
export function waitForEvent(
|
|
21
|
-
threadManager: ThreadManager,
|
|
22
|
-
threadId: string,
|
|
23
|
-
eventType: LaceEventType,
|
|
24
|
-
timeoutMs = 5000,
|
|
25
|
-
): Promise<LaceEvent> {
|
|
26
|
-
return new Promise((resolve, reject) => {
|
|
27
|
-
const startTime = Date.now();
|
|
28
|
-
|
|
29
|
-
const check = () => {
|
|
30
|
-
const events = threadManager.getEvents(threadId);
|
|
31
|
-
const event = events.find((e) => e.type === eventType);
|
|
32
|
-
|
|
33
|
-
if (event) {
|
|
34
|
-
resolve(event);
|
|
35
|
-
} else if (Date.now() - startTime > timeoutMs) {
|
|
36
|
-
reject(new Error(`Timeout waiting for ${eventType} event after ${timeoutMs}ms`));
|
|
37
|
-
} else {
|
|
38
|
-
setTimeout(check, 10); // Poll every 10ms for efficiency
|
|
39
|
-
}
|
|
40
|
-
};
|
|
41
|
-
|
|
42
|
-
check();
|
|
43
|
-
});
|
|
44
|
-
}
|
|
45
|
-
|
|
46
|
-
/**
|
|
47
|
-
* Wait for a specific number of events of a given type
|
|
48
|
-
*
|
|
49
|
-
* @param threadManager - The thread manager to query
|
|
50
|
-
* @param threadId - Thread to check for events
|
|
51
|
-
* @param eventType - Type of event to wait for
|
|
52
|
-
* @param count - Number of events to wait for
|
|
53
|
-
* @param timeoutMs - Maximum time to wait (default 5000ms)
|
|
54
|
-
* @returns Promise resolving to all matching events once count is reached
|
|
55
|
-
*
|
|
56
|
-
* Example:
|
|
57
|
-
* // Wait for 2 AGENT_MESSAGE events (initial response + continuation)
|
|
58
|
-
* await waitForEventCount(threadManager, agentThreadId, 'AGENT_MESSAGE', 2);
|
|
59
|
-
*/
|
|
60
|
-
export function waitForEventCount(
|
|
61
|
-
threadManager: ThreadManager,
|
|
62
|
-
threadId: string,
|
|
63
|
-
eventType: LaceEventType,
|
|
64
|
-
count: number,
|
|
65
|
-
timeoutMs = 5000,
|
|
66
|
-
): Promise<LaceEvent[]> {
|
|
67
|
-
return new Promise((resolve, reject) => {
|
|
68
|
-
const startTime = Date.now();
|
|
69
|
-
|
|
70
|
-
const check = () => {
|
|
71
|
-
const events = threadManager.getEvents(threadId);
|
|
72
|
-
const matchingEvents = events.filter((e) => e.type === eventType);
|
|
73
|
-
|
|
74
|
-
if (matchingEvents.length >= count) {
|
|
75
|
-
resolve(matchingEvents);
|
|
76
|
-
} else if (Date.now() - startTime > timeoutMs) {
|
|
77
|
-
reject(
|
|
78
|
-
new Error(
|
|
79
|
-
`Timeout waiting for ${count} ${eventType} events after ${timeoutMs}ms (got ${matchingEvents.length})`,
|
|
80
|
-
),
|
|
81
|
-
);
|
|
82
|
-
} else {
|
|
83
|
-
setTimeout(check, 10);
|
|
84
|
-
}
|
|
85
|
-
};
|
|
86
|
-
|
|
87
|
-
check();
|
|
88
|
-
});
|
|
89
|
-
}
|
|
90
|
-
|
|
91
|
-
/**
|
|
92
|
-
* Wait for an event matching a custom predicate
|
|
93
|
-
* Useful when you need to check event data, not just type
|
|
94
|
-
*
|
|
95
|
-
* @param threadManager - The thread manager to query
|
|
96
|
-
* @param threadId - Thread to check for events
|
|
97
|
-
* @param predicate - Function that returns true when event matches
|
|
98
|
-
* @param description - Human-readable description for error messages
|
|
99
|
-
* @param timeoutMs - Maximum time to wait (default 5000ms)
|
|
100
|
-
* @returns Promise resolving to the first matching event
|
|
101
|
-
*
|
|
102
|
-
* Example:
|
|
103
|
-
* // Wait for TOOL_RESULT with specific ID
|
|
104
|
-
* await waitForEventMatch(
|
|
105
|
-
* threadManager,
|
|
106
|
-
* agentThreadId,
|
|
107
|
-
* (e) => e.type === 'TOOL_RESULT' && e.data.id === 'call_123',
|
|
108
|
-
* 'TOOL_RESULT with id=call_123'
|
|
109
|
-
* );
|
|
110
|
-
*/
|
|
111
|
-
export function waitForEventMatch(
|
|
112
|
-
threadManager: ThreadManager,
|
|
113
|
-
threadId: string,
|
|
114
|
-
predicate: (event: LaceEvent) => boolean,
|
|
115
|
-
description: string,
|
|
116
|
-
timeoutMs = 5000,
|
|
117
|
-
): Promise<LaceEvent> {
|
|
118
|
-
return new Promise((resolve, reject) => {
|
|
119
|
-
const startTime = Date.now();
|
|
120
|
-
|
|
121
|
-
const check = () => {
|
|
122
|
-
const events = threadManager.getEvents(threadId);
|
|
123
|
-
const event = events.find(predicate);
|
|
124
|
-
|
|
125
|
-
if (event) {
|
|
126
|
-
resolve(event);
|
|
127
|
-
} else if (Date.now() - startTime > timeoutMs) {
|
|
128
|
-
reject(new Error(`Timeout waiting for ${description} after ${timeoutMs}ms`));
|
|
129
|
-
} else {
|
|
130
|
-
setTimeout(check, 10);
|
|
131
|
-
}
|
|
132
|
-
};
|
|
133
|
-
|
|
134
|
-
check();
|
|
135
|
-
});
|
|
136
|
-
}
|
|
137
|
-
|
|
138
|
-
// Usage example from actual debugging session:
|
|
139
|
-
//
|
|
140
|
-
// BEFORE (flaky):
|
|
141
|
-
// ---------------
|
|
142
|
-
// const messagePromise = agent.sendMessage('Execute tools');
|
|
143
|
-
// await new Promise(r => setTimeout(r, 300)); // Hope tools start in 300ms
|
|
144
|
-
// agent.abort();
|
|
145
|
-
// await messagePromise;
|
|
146
|
-
// await new Promise(r => setTimeout(r, 50)); // Hope results arrive in 50ms
|
|
147
|
-
// expect(toolResults.length).toBe(2); // Fails randomly
|
|
148
|
-
//
|
|
149
|
-
// AFTER (reliable):
|
|
150
|
-
// ----------------
|
|
151
|
-
// const messagePromise = agent.sendMessage('Execute tools');
|
|
152
|
-
// await waitForEventCount(threadManager, threadId, 'TOOL_CALL', 2); // Wait for tools to start
|
|
153
|
-
// agent.abort();
|
|
154
|
-
// await messagePromise;
|
|
155
|
-
// await waitForEventCount(threadManager, threadId, 'TOOL_RESULT', 2); // Wait for results
|
|
156
|
-
// expect(toolResults.length).toBe(2); // Always succeeds
|
|
157
|
-
//
|
|
158
|
-
// Result: 60% pass rate → 100%, 40% faster execution
|