pi-crew 0.10.4 → 0.10.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +233 -0
- package/agents/analyst.md +37 -2
- package/agents/cold-verifier.md +10 -1
- package/agents/councillor-critic.md +39 -0
- package/agents/councillor-pragmatist.md +39 -0
- package/agents/councillor-skeptic.md +41 -0
- package/agents/critic.md +40 -2
- package/agents/designer.md +58 -0
- package/agents/executor.md +39 -2
- package/agents/explorer.md +38 -2
- package/agents/librarian.md +49 -0
- package/agents/oracle.md +54 -0
- package/agents/orchestrator.md +48 -0
- package/agents/planner.md +41 -2
- package/agents/reviewer.md +39 -2
- package/agents/security-reviewer.md +43 -2
- package/agents/test-engineer.md +48 -2
- package/agents/verifier.md +14 -1
- package/agents/writer.md +32 -2
- package/dist/index.mjs +1297 -853
- package/package.json +1 -1
- package/skills/async-worker-recovery/SKILL.md +4 -1
- package/skills/child-pi-spawning/SKILL.md +4 -1
- package/skills/context-artifact-hygiene/SKILL.md +4 -1
- package/skills/council/SKILL.md +24 -45
- package/skills/delegation-patterns/SKILL.md +18 -1
- package/skills/distill-persona/SKILL.md +4 -1
- package/skills/distill-software/SKILL.md +4 -1
- package/skills/event-log-tracing/SKILL.md +4 -1
- package/skills/git-master/SKILL.md +4 -1
- package/skills/iterative-audit/SKILL.md +4 -1
- package/skills/live-agent-lifecycle/SKILL.md +4 -1
- package/skills/mailbox-interactive/SKILL.md +4 -1
- package/skills/model-routing-context/SKILL.md +10 -1
- package/skills/multi-perspective-review/SKILL.md +18 -1
- package/skills/observability-reliability/SKILL.md +4 -1
- package/skills/orchestration/SKILL.md +18 -1
- package/skills/ownership-session-security/SKILL.md +4 -1
- package/skills/pi-extension-lifecycle/SKILL.md +4 -1
- package/skills/post-mortem/SKILL.md +4 -1
- package/skills/read-only-explorer/SKILL.md +4 -1
- package/skills/real-test-pi-crew/SKILL.md +165 -12
- package/skills/requirements-to-task-packet/SKILL.md +10 -1
- package/skills/research/SKILL.md +4 -1
- package/skills/resource-discovery-config/SKILL.md +10 -1
- package/skills/runtime-state-reader/SKILL.md +4 -1
- package/skills/safe-bash/SKILL.md +4 -1
- package/skills/scrutinize/SKILL.md +24 -1
- package/skills/secure-agent-orchestration-review/SKILL.md +4 -1
- package/skills/state-mutation-locking/SKILL.md +4 -1
- package/skills/systematic-debugging/SKILL.md +4 -1
- package/skills/verification-before-done/SKILL.md +18 -1
- package/skills/widget-rendering/SKILL.md +4 -1
- package/skills/workspace-isolation/SKILL.md +4 -1
- package/skills/worktree-isolation/SKILL.md +4 -1
- package/src/config/config-validation.ts +1 -0
- package/src/config/types.ts +8 -0
- package/src/errors.ts +1 -1
- package/src/extension/context-status-injection.ts +2 -2
- package/src/extension/knowledge-injection.ts +19 -7
- package/src/extension/post-init-skill-check.ts +32 -0
- package/src/extension/register.ts +9 -1
- package/src/extension/registration/hook-registration.ts +20 -3
- package/src/extension/registration/tool-loop-guard.ts +243 -0
- package/src/extension/team-tool/handle-settings.ts +10 -0
- package/src/extension/team-tool/run.ts +42 -1
- package/src/extension/team-tool-types.ts +6 -0
- package/src/prompt/prompt-runtime.ts +25 -6
- package/src/runtime/async-runner.ts +75 -11
- package/src/runtime/background-runner.ts +73 -7
- package/src/runtime/broker/crew-broker-client.ts +45 -2
- package/src/runtime/broker/crew-broker.ts +22 -27
- package/src/runtime/broker/protocol/request-parsers.ts +10 -2
- package/src/runtime/broker/stdin-handshake.ts +87 -0
- package/src/runtime/broker/wait-push.ts +45 -0
- package/src/runtime/detached-run-results.ts +25 -1
- package/src/runtime/foreground-watchdog.ts +24 -5
- package/src/runtime/live-session/live-session-runtime.ts +1 -1
- package/src/runtime/model/model-scope.ts +2 -2
- package/src/runtime/run-tracker.ts +74 -19
- package/src/runtime/skill-instructions.ts +20 -4
- package/src/runtime/task-runner/child-executor.ts +1 -1
- package/src/runtime/task-runner/prompt-builder.ts +22 -9
- package/src/schema/config-schema.ts +1 -0
- package/src/skills/discover-skills.ts +2 -2
- package/src/ui/settings-overlay.ts +40 -0
- package/src/utils/frontmatter.ts +7 -1
- package/src/utils/ndjson.ts +9 -1
|
@@ -1,6 +1,9 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: real-test-pi-crew
|
|
3
|
-
description:
|
|
3
|
+
description: >
|
|
4
|
+
End-to-end verification for pi-crew changes: fast critical tests, 3-path kill-switch proof, bundle md5 sync, live TUI probing, smoke team runs, a live feature-action battery (team tool + subagent tools), a surface-mode battery (workers in real tmux/herdr panes, degrade-to-headless), and a resource-contract battery (agent .md frontmatter dual-parse, routing render, output contracts).
|
|
5
|
+
When NOT to use: unit tests for isolated modules (use test runner directly); pure test execution.
|
|
6
|
+
|
|
4
7
|
origin: pi-crew
|
|
5
8
|
triggers:
|
|
6
9
|
- "test the change"
|
|
@@ -39,16 +42,30 @@ triggers:
|
|
|
39
42
|
- "wc-gate"
|
|
40
43
|
- "migration validator warning"
|
|
41
44
|
- "slow tier"
|
|
45
|
+
- "agent frontmatter"
|
|
46
|
+
- "folded scalar"
|
|
47
|
+
- "agent body change"
|
|
48
|
+
- "routing metadata"
|
|
49
|
+
- "output contract"
|
|
50
|
+
- "loop guard"
|
|
51
|
+
- "post-init skill check"
|
|
52
|
+
- "resource contract"
|
|
53
|
+
- "sigterm"
|
|
54
|
+
- "silent bash"
|
|
55
|
+
- "worker killed mid command"
|
|
56
|
+
- "tier 12"
|
|
42
57
|
---
|
|
43
58
|
|
|
44
59
|
# real-test-pi-crew
|
|
45
60
|
|
|
46
61
|
End-to-end verification discipline for pi-crew changes. Distilled from the broker Phase-4 rollout (commits `1cb2dca` → `d599578` → `612e18b` → `4186284`, July 2026). The pain this skill prevents: shipping code that compiles + unit-tests-green but breaks in the user's live Pi session, or hangs the verifier worker.
|
|
47
62
|
|
|
48
|
-
**When to use**: after any change to `src/runtime/broker/*.ts` (broker + tokens + issuer), `src/ui/`, `src/config/` (incl. `src/config/migration-validator.ts`), `src/extension/registration/lifecycle-handlers.ts`, `src/runtime/child-pi/*.ts` (worker spawn/kill/steering), `src/runtime/surface/*.ts` (MuxSurface providers, degrade, launch script), `src/prompt/*.ts` (worker-side tools: ask / message / delegate / surface-worker recorder), `src/runtime/goal-workflow/plan-templates.ts`, `src/runtime/team-runner.ts` or `src/runtime/task-runner/**` (scheduler / execution — Tier 7 smoke), `src/state/**` (durable state — Tier 7 + 9a events/status + **Tier 11a read-your-writes**), `src/runtime/live-session/**` + `src/runtime/custom-tools/*` (live-session mode + worker custom tools), `src/schema/team-tool-schema.ts` (or any `Type.Unsafe({...})` schema definition), `src/extension/registration/team-tool.ts`, `workflows/*.workflow.md`, `.github/workflows/*.yml` (CI env — Tier 11e), `scripts/wc-gate.mjs` (Tier 11b), or before any commit touching these paths. Schema changes additionally require Tier 9 (feature battery) because the team tool's TypeBox schema is validated by pi-ai BEFORE the handler runs — a too-strict or malformed schema breaks every action silently. Surface changes additionally require Tier 10 (surface-mode battery) because surface is fail-closed: every failure degrades to headless and the run still goes green — only pane-level evidence proves the panes engaged.
|
|
63
|
+
**When to use**: after any change to `src/runtime/broker/*.ts` (broker + tokens + issuer), `src/ui/`, `src/config/` (incl. `src/config/migration-validator.ts`), `src/extension/registration/lifecycle-handlers.ts`, `src/runtime/child-pi/*.ts` (worker spawn/kill/steering), `src/runtime/surface/*.ts` (MuxSurface providers, degrade, launch script), `src/prompt/*.ts` (worker-side tools: ask / message / delegate / surface-worker recorder), `src/runtime/goal-workflow/plan-templates.ts`, `src/runtime/team-runner.ts` or `src/runtime/task-runner/**` (scheduler / execution — Tier 7 smoke), `src/state/**` (durable state — Tier 7 + 9a events/status + **Tier 11a read-your-writes**), `src/runtime/live-session/**` + `src/runtime/custom-tools/*` (live-session mode + worker custom tools), `src/schema/team-tool-schema.ts` (or any `Type.Unsafe({...})` schema definition), `src/extension/registration/team-tool.ts`, `workflows/*.workflow.md`, `.github/workflows/*.yml` (CI env — Tier 11e), `scripts/wc-gate.mjs` (Tier 11b), or before any commit touching these paths. Schema changes additionally require Tier 9 (feature battery) because the team tool's TypeBox schema is validated by pi-ai BEFORE the handler runs — a too-strict or malformed schema breaks every action silently. Surface changes additionally require Tier 10 (surface-mode battery) because surface is fail-closed: every failure degrades to headless and the run still goes green — only pane-level evidence proves the panes engaged. Resource `.md` changes (agent bodies/frontmatter, skill metadata, discovery, frontmatter parsing) additionally require **Tier 12** (resource-contract battery) because the agent/team/workflow frontmatter parser is line-based, not YAML — a folded scalar parses as `">"` for every consumer while all other tiers stay green.
|
|
49
64
|
|
|
50
65
|
> **Path map (2026-08-26 reorg + A1)**: `src/runtime/crew-broker*.ts` → `src/runtime/broker/`; `src/runtime/child-pi*.ts` → `src/runtime/child-pi/`; `src/runtime/plan-templates.ts` (flat) → `src/runtime/goal-workflow/plan-templates.ts`; NEW dirs `src/runtime/surface/` and `src/prompt/`. Test files moved with them (`test/unit/crew-broker-*.test.ts` → `test/unit/runtime/broker/`, `test/unit/keybinding-map.parity.test.ts` → `test/unit/ui/`, ...).
|
|
51
66
|
|
|
67
|
+
> **2026-09-11 update (Batch-1..10, branch `fix/bundle-skill-resolution-and-skill-meta`, tip `aa899a1e`)**: builtin agents 17 → **18** (librarian, oracle, designer, 3 councillors, orchestrator); every agent carries flat routing metadata; NEW **Tier 12** (resource-contract battery) for `agents/*.md` / `skills/*/SKILL.md` / discovery / frontmatter changes; staleness gate gained a path-leak scan (ARCH-7); `scripts/release-smoke.mjs` gained a tarball import + peer-install gate (ARCH-6); two new operational quirks documented (broker SIGTERM on long silent bash; `wait-request-broker.test.ts` 180s-per-file load flake). Orientation doc: `CONTEXT.md`.
|
|
68
|
+
|
|
52
69
|
## Core principle: disk ≠ live Pi
|
|
53
70
|
|
|
54
71
|
Two locations hold pi-crew state:
|
|
@@ -65,7 +82,7 @@ The 3-way resolution order for `dist/index.mjs` (per `index.ts:1-25`):
|
|
|
65
82
|
|
|
66
83
|
> **Note on version pins**: this skill mentions specific versions (v0.9.17, v0.9.46, v0.9.47) as anchors for *when a behavior was introduced*, not as a constraint on which version the skill applies to. The verification discipline (Tiers 1–10) applies to every pi-crew release. Verify the version pin is still accurate via `git log --oneline -- index.ts` and `git log --oneline -- src/ui/run-dashboard.ts`.
|
|
67
84
|
|
|
68
|
-
**
|
|
85
|
+
**Resource `.md` files are runtime data too** — `agents/*.md` and `skills/*/SKILL.md` load at RUN-CONSTRUCTION time from the package dir (they are NOT embedded in `dist/index.mjs`): edits take effect on the next team run / discovery call (discovery cache TTL ~30s — `invalidateAgentDiscoveryCache()` forces a fresh read), with NO bundle rebuild and NO Pi restart. `workflows/*.workflow.md` and task prompt strings inside `src/runtime/goal-workflow/plan-templates.ts` are the same: loaded per-call, NOT bundled. (Caveat: `src/` TypeScript that CONSUMES these files still follows the bundle rule below.)
|
|
69
86
|
|
|
70
87
|
**The most common silent-failure mode**: edit `src/`, run `npm test` (pass!), rebuild bundle (good md5!), but the session still has the old code because Pi wasn't `/quit`-ed + reopened.
|
|
71
88
|
|
|
@@ -102,8 +119,10 @@ The skill maps to existing CI gates as follows:
|
|
|
102
119
|
| `PI_CREW_BROKER=0 npm run test:critical` | Tier 2 (env kill switch path) | n/a — manual |
|
|
103
120
|
| `npm run typecheck` | Tier 3 | `.github/workflows/*.yml` (every PR) |
|
|
104
121
|
| `npm run check:wc-gate` | Tier 11b | **in BOTH `ci` and `ci:fast` scripts** (`package.json:71-72`) + explicit step in `.github/workflows/ci.yml:66-71` (since `09dda842` — was `ci:fast`-only, i.e. advisory) |
|
|
105
|
-
| Bundle-staleness check | Tier 3 last step | `scripts/check-bundle-staleness.mjs`; `--committed-hash` mode = Tier 11j release gate |
|
|
106
|
-
|
|
|
122
|
+
| Bundle-staleness check (incl. **ARCH-7 path-leak scan** since `7d18508b` — line-scans `dist/index.mjs` + structural sourcemap check for tracked-source leaks) | Tier 3 last step | `scripts/check-bundle-staleness.mjs`; `--committed-hash` mode = Tier 11j release gate |
|
|
123
|
+
| `npm run test:bundle` (bundle import smoke, 2 tests) | Tier 3 post-build sanity | `test/unit/bundle-load.test.ts` |
|
|
124
|
+
| `node scripts/release-smoke.mjs` (manual, release cut) | Tier 3/11j companion | ARCH-6: installs pi-* peer deps, `import()`s the tarball-installed bundle (`:77`), shape-checks exports |
|
|
125
|
+
| Full `npm test` (= unit 823 files + integration 31) | n/a — too slow for in-loop | CI only; slow tier (3 files) is a SEPARATE glob `test:integration:slow` — only `npm run test:full` includes it |
|
|
107
126
|
| `PI_CREW_SMOKE=1` env | Tier 11e | set ONLY in `weekly-smoke.yml` (auth-gated); nightly.yml deliberately does NOT (comment at `:24`) |
|
|
108
127
|
|
|
109
128
|
To add Tier 1 to a pre-commit hook:
|
|
@@ -134,7 +153,7 @@ To add Tier 1 to CI as a fast-feedback gate (under 30s):
|
|
|
134
153
|
|
|
135
154
|
**What**: run the curated 14-file fast subset.
|
|
136
155
|
|
|
137
|
-
**Why this exists**: full `npm run test:unit` runs
|
|
156
|
+
**Why this exists**: full `npm run test:unit` runs 823 files (was 642 at skill-writing time — it keeps growing), several minutes. Verifier worker response timeout would kill the worker mid-run → run = "hang". The fix (introduced in commit `1cb2dca`) splits out a `test:critical` subset covering exactly what changed in the broker/UI work.
|
|
138
157
|
|
|
139
158
|
**How**:
|
|
140
159
|
|
|
@@ -209,6 +228,7 @@ All three must show `# pass 101 # fail 0`. Measured times in this session (2026-
|
|
|
209
228
|
npm run typecheck # ~20s, exits 0 with "strip-types import ok"
|
|
210
229
|
npm run build:bundle # <1s, prints "[build-bundle] dist/index.mjs NNNN KB in NNN ms"
|
|
211
230
|
md5sum dist/index.mjs
|
|
231
|
+
node scripts/check-bundle-staleness.mjs # ARCH-7: staleness + path-leak scan — exit 0
|
|
212
232
|
```
|
|
213
233
|
|
|
214
234
|
Compare the printed md5 against what the user's Pi session loaded. If they differ → the session is running stale bundle.
|
|
@@ -222,7 +242,7 @@ Compare the printed md5 against what the user's Pi session loaded. If they diffe
|
|
|
222
242
|
| Bundle builder | `scripts/build-bundle.mjs` (esbuild-based, bundles `index.bundle.ts` → `dist/index.mjs`) |
|
|
223
243
|
| Bundle resolution rule | `index.ts:1-25` (entrypoint docstring); also `scripts/build-bundle.mjs:14-20` (entrypoint preference); **symlink is live for source files but the bundled `dist/index.mjs` is loaded** |
|
|
224
244
|
| Postinstall hook | `scripts/postinstall.mjs:43` — best-effort bundle rebuild; falls back to strip-types if esbuild missing |
|
|
225
|
-
| Bundle md5 anchors | `1cc4d55e18add7b9a036c569143320b6` (Phase-4 flip, ~2.78 MB) → `16e29d053bd370e24f40df147dadcb79` (v0.9.66, 2026-08-11) → `9b557ac106b82e1ee33d39dd0d6c7dd7` (post-MuxSurface-A1 main, 2026-08-27). **Always check current**: `md5sum dist/index.mjs` |
|
|
245
|
+
| Bundle md5 anchors | `1cc4d55e18add7b9a036c569143320b6` (Phase-4 flip, ~2.78 MB) → `16e29d053bd370e24f40df147dadcb79` (v0.9.66, 2026-08-11) → `9b557ac106b82e1ee33d39dd0d6c7dd7` (post-MuxSurface-A1 main, 2026-08-27) → `945720b1ad25673d86e263cdd834532f` (post-Batch-10 branch tip `aa899a1e`, 2026-09-11, ~3.30 MB). **Always check current**: `md5sum dist/index.mjs` |
|
|
226
246
|
|
|
227
247
|
---
|
|
228
248
|
|
|
@@ -232,7 +252,8 @@ Compare the printed md5 against what the user's Pi session loaded. If they diffe
|
|
|
232
252
|
|
|
233
253
|
**The immediate-vs-rebuild rule** (which edits take effect without a rebuild):
|
|
234
254
|
- `workflows/*.workflow.md` edits → **immediate**, no rebuild, no restart
|
|
235
|
-
- `
|
|
255
|
+
- `agents/*.md` + `skills/*/SKILL.md` edits → **immediate** — runtime data loaded from the package dir per discovery/run (cache TTL ~30s); NOT embedded in `dist/index.mjs`
|
|
256
|
+
- `src/runtime/goal-workflow/plan-templates.ts` → **needs rebuild** (correction of the pre-v0.9.17 claim above): it is `src/` TypeScript imported by the bundle, so its `taskTemplate` strings ship inside `dist/index.mjs` — edit → `build:bundle` → restart
|
|
236
257
|
- Everything else (`src/` edits, `package.json`) → must `npm run build:bundle` THEN user `/quit` + reopen Pi
|
|
237
258
|
|
|
238
259
|
**How to verify in this session**:
|
|
@@ -367,7 +388,7 @@ else:
|
|
|
367
388
|
|
|
368
389
|
**What**: prove the verifier worker completes within `RESPONSE_TIMEOUT_MS` (**600s since the stuck-worker hardening — was 300s when this skill was distilled; `DEFAULT_CHILD_PI.responseTimeoutMs = 10 * 60_000`**).
|
|
369
390
|
|
|
370
|
-
**Why this is its own tier**: `test:critical` covers unit-level invariants, but the verifier LLM is a separate failure mode — it reads the verifier prompt from `src/runtime/goal-workflow/plan-templates.ts:144, 147` (taskTemplate strings) or from `workflows/*.workflow.md` (workflow verifier sections), then decides which bash command to run. If the prompt says "Run tests" without specifying which, the LLM runs `npm test` (
|
|
391
|
+
**Why this is its own tier**: `test:critical` covers unit-level invariants, but the verifier LLM is a separate failure mode — it reads the verifier prompt from `src/runtime/goal-workflow/plan-templates.ts:144, 147` (taskTemplate strings) or from `workflows/*.workflow.md` (workflow verifier sections), then decides which bash command to run. If the prompt says "Run tests" without specifying which, the LLM runs `npm test` (823 files) and the worker gets killed by the response timeout with exit 143.
|
|
371
392
|
|
|
372
393
|
**How** (from parent Pi session — `team` is a tool, not a shell command):
|
|
373
394
|
|
|
@@ -409,6 +430,7 @@ The `team` tool is described in the agent's system prompt. Use `team action='sta
|
|
|
409
430
|
|
|
410
431
|
1. **Verifier LLM runs `npm test`** (full unit + integration suite, >4 min) instead of `npm run test:critical`. Symptom: worker killed with exit 143 at the response timeout (300s historically — the measured runs below predate the bump to 600s). Fix: rewrite the verifier prompt to specify the exact fast command AND include "Do NOT run `npm test` or `npm run test:unit`".
|
|
411
432
|
2. **Verifier LLM improvises** with a clean-cache `npm test` run anyway. The cache directive ("cache to `.crew/cache/`", "do NOT re-run") catches this — the second worker that observes a cached log should not re-run.
|
|
433
|
+
3. **Broker SIGTERMs the worker mid long-silent-bash** (Batch-1 postmortem, `postmortem-batch-1-sigterm.md`): a worker running ONE >5–10 min command (full `npm test` ≈ 10 min) emits no LLM activity; the broker's responsiveness check kills it mid-run — exit 143 WHILE the command is still running, not a 600s response timeout. The transcript shows the command started and never returned. Work is usually intact (manual re-run was green); the kill is the "hang". Fix direction: split long suites into <5 min chunks or emit progress between commands. Tracked as `CONTEXT.md` Flagged #1.
|
|
412
434
|
|
|
413
435
|
---
|
|
414
436
|
|
|
@@ -619,7 +641,7 @@ grep -n "Log the event first" src/runtime/recovery/crash-recovery.ts # design
|
|
|
619
641
|
# 3. buffered-site census — snapshot & audit:
|
|
620
642
|
grep -rln "appendEventBuffered" src/ | wc -l # 16 files / ~70 raw matches (incl. imports+definition) at v0.10.5; audited live conversions = 43; EVERY new site needs the reader-audit
|
|
621
643
|
# 4. the full gate — test:critical has NO stores/dwf/recovery coverage:
|
|
622
|
-
npm run test:unit #
|
|
644
|
+
npm run test:unit # 823 files, ~7500 tests, 15-18 min under load — MANDATORY after any delayed-write conversion program
|
|
623
645
|
```
|
|
624
646
|
|
|
625
647
|
### 11b. wc-gate enforcement (M4 done-gate)
|
|
@@ -706,6 +728,86 @@ node scripts/check-bundle-staleness.mjs --committed-hash # "OK: committed dist
|
|
|
706
728
|
|
|
707
729
|
---
|
|
708
730
|
|
|
731
|
+
## Tier 12 — Resource-contract battery (agent .md + skill metadata)
|
|
732
|
+
|
|
733
|
+
**What**: prove agent frontmatter/bodies and skill metadata still parse and render after edits — through BOTH parsers and into the routing guidance the leader sees.
|
|
734
|
+
|
|
735
|
+
**Why this is its own tier**: `agents/*.md` are contracts — frontmatter grants tools and routing metadata (`useWhen`/`avoidWhen`/`cost`/`category`), the body IS the child's system prompt (`systemPromptMode: replace`), and the `## Output format` section is test-enforced. The agent/team/workflow frontmatter parser (`src/utils/frontmatter.ts`, `parseLines`) is **line-based, not YAML** — a folded scalar (`description: >`) parses as the literal string `">"` for EVERY consumer while typecheck/lint/test:critical all stay green (real regression: Batch 9, all 17 agents; fixed in `aa899a1e` by restoring single-line quoted values + teaching `parseLines` to strip symmetric quotes). Skills are exempt (they go through the real `yaml` package — folded scalars are FINE in `skills/*/SKILL.md`). Only a dual-parse probe catches this class.
|
|
736
|
+
|
|
737
|
+
**When required**: any change to `agents/*.md`, `skills/*/SKILL.md`, `src/agents/discover-agents.ts`, `src/skills/discover-skills.ts`, `src/utils/frontmatter.ts`, `src/runtime/skill-instructions.ts` (skill override resolution), or `src/extension/autonomous-policy.ts` (guidance render).
|
|
738
|
+
|
|
739
|
+
**Frontmatter contract rules** (agents/teams/workflows — the line-based parser):
|
|
740
|
+
- values stay **single-line**; a value containing `": "` MUST be wrapped in symmetric double quotes (`parseLines` strips them, `aa899a1e`)
|
|
741
|
+
- NEVER folded scalars (`key: >` / `key: |`) — they parse as `">"` / `"|"` (`CONTEXT.md` Flagged #4)
|
|
742
|
+
- routing keys are FLAT top-level CSV — `useWhen: "a, b"`, `avoidWhen: "…"`, `cost: cheap`, `category: orchestration` (parsed at `src/agents/discover-agents.ts:388-391`; a nested `routing:` block is silently ignored)
|
|
743
|
+
|
|
744
|
+
**How**:
|
|
745
|
+
|
|
746
|
+
```bash
|
|
747
|
+
# 12a. Output contracts — every builtin agent must have '## Output format' + fenced block:
|
|
748
|
+
node --experimental-strip-types --no-warnings --test --test-force-exit test/unit/agents/agent-output-contracts.test.ts
|
|
749
|
+
# 1 test iterating ALL builtin agents (18 @ 2026-09-11)
|
|
750
|
+
|
|
751
|
+
# 12b. BOTH-parser proof (discovery + strict YAML) — ALWAYS invalidate the discovery cache first:
|
|
752
|
+
node --experimental-strip-types --no-warnings -e '
|
|
753
|
+
import("./src/agents/discover-agents.ts").then(mod => {
|
|
754
|
+
mod.invalidateAgentDiscoveryCache();
|
|
755
|
+
const list = mod.discoverAgents(process.cwd()).builtin;
|
|
756
|
+
const bad = list.filter(a => !a.description?.includes("When NOT to use:") || a.description.startsWith(String.fromCharCode(34)));
|
|
757
|
+
const noRoute = list.filter(a => !a.routing?.useWhen);
|
|
758
|
+
console.log("agents:", list.length, "| bad desc:", bad.length, "| no routing:", noRoute.length);
|
|
759
|
+
});'
|
|
760
|
+
# expect: agents: 18 | bad desc: 0 | no routing: 0
|
|
761
|
+
node -e '
|
|
762
|
+
const yaml=require("yaml"),fs=require("fs");let ok=0,fail=[];
|
|
763
|
+
for (const f of fs.readdirSync("agents")){
|
|
764
|
+
const m=/^---\r?\n([\s\S]*?)\r?\n---/.exec(fs.readFileSync("agents/"+f,"utf-8"));
|
|
765
|
+
if(!m)continue;
|
|
766
|
+
try{const p=yaml.parse(m[1]);if(p.name&&p.description)ok++;}catch{fail.push(f);}
|
|
767
|
+
}
|
|
768
|
+
console.log("strict YAML:",ok,"ok /",fail.length,"fail",fail.length?JSON.stringify(fail):"");'
|
|
769
|
+
# expect: strict YAML: 18 ok / 0 fail
|
|
770
|
+
|
|
771
|
+
# 12c. Routing guidance renders (leader-side):
|
|
772
|
+
node --experimental-strip-types --no-warnings -e '
|
|
773
|
+
import("./src/agents/discover-agents.ts").then(async da=>{
|
|
774
|
+
const pol=await import("./src/extension/autonomous-policy.ts");
|
|
775
|
+
da.invalidateAgentDiscoveryCache();
|
|
776
|
+
const g=pol.buildResourceRoutingGuidance(process.cwd(),40000);
|
|
777
|
+
const agentLines=g.split("\n").filter(l=>l.startsWith("- ")&&!/defaultWorkflow=|roles=|steps=/.test(l)&&/\((builtin|project|user)\):/.test(l));
|
|
778
|
+
const withRoute=agentLines.filter(l=>l.includes("useWhen="));
|
|
779
|
+
console.log("rendered agent lines:",agentLines.length,"| with useWhen:",withRoute.length,
|
|
780
|
+
"| orchestrator:",g.includes("- orchestrator ("),"| verifier:",g.includes("- verifier ("));
|
|
781
|
+
});'
|
|
782
|
+
# expect: every rendered AGENT line carries useWhen= (workflow lines legitimately lack it).
|
|
783
|
+
# The list is BUDGET-TRUNCATED BY DESIGN — at 40000 chars ~16/18 agents render; the tail
|
|
784
|
+
# (alphabetically last: verifier, writer) is cut first. Accept: with-route == agent-lines,
|
|
785
|
+
# newest agent (orchestrator) present, count >= 15. Do NOT assert 18/18 — truncation is correct.
|
|
786
|
+
|
|
787
|
+
# 12d. Fast unit batteries for the resource layer:
|
|
788
|
+
node --experimental-strip-types --no-warnings --test --test-force-exit \
|
|
789
|
+
test/unit/bundle-skill-resolution.test.ts \
|
|
790
|
+
test/unit/extension/registration/tool-loop-guard.test.ts \
|
|
791
|
+
test/unit/runtime/core/skill-instructions.test.ts
|
|
792
|
+
# packageRoot skill resolution + loop guard (12 tests) + skill override wildcard `*` / denylist `!name` (26 tests)
|
|
793
|
+
```
|
|
794
|
+
|
|
795
|
+
**Acceptance**: 12a green; 12b BOTH parsers clean (18/18 descriptions with When-NOT, 0 quote leakage, 18/18 routing, 0 strict-YAML fails); 12c every rendered agent line carries `useWhen=` (budget-truncation is by design — see the note in 12c); 12d all pass. Agent/skill-only changes need NO bundle rebuild (runtime-loaded from the package dir) — but `src/` changes in the same commit still follow the Tier 3 bundle rule.
|
|
796
|
+
|
|
797
|
+
**References**:
|
|
798
|
+
|
|
799
|
+
| What | Where |
|
|
800
|
+
|---|---|
|
|
801
|
+
| Line-based parser + quote-strip | `src/utils/frontmatter.ts` (`parseLines`) — quote-strip added in `aa899a1e` |
|
|
802
|
+
| Flat routing keys parse | `src/agents/discover-agents.ts:388-391, 476` |
|
|
803
|
+
| Output-contract AC | `test/unit/agents/agent-output-contracts.test.ts` (Batch 9, `06c5d7ca`) |
|
|
804
|
+
| Skill override `*`/`!name` | `src/runtime/skill-instructions.ts` (`collectTaskSkillNames`, Batch 1+2 `c97bc578`) |
|
|
805
|
+
| Guidance builder | `src/extension/autonomous-policy.ts` (`buildResourceRoutingGuidance`) |
|
|
806
|
+
| Orchestrator (18th agent) | `agents/orchestrator.md` (Batch 10 `aa899a1e`) — process-only body; discovery guidance is the single routing authority |
|
|
807
|
+
| Quirk registry | `CONTEXT.md` — glossary + Flagged (#1 broker SIGTERM, #4 frontmatter parser) |
|
|
808
|
+
|
|
809
|
+
---
|
|
810
|
+
|
|
709
811
|
## Anti-patterns (the cost is real, observed in this session)
|
|
710
812
|
|
|
711
813
|
| Anti-pattern | Cost | Where fixed | Reference |
|
|
@@ -716,7 +818,7 @@ node scripts/check-bundle-staleness.mjs --committed-hash # "OK: committed dist
|
|
|
716
818
|
| Test using real `loadConfig()` to mock config | Flaky when env / disk config changes | `612e18b` | `test/unit/runtime/broker/crew-broker-server-gate.test.ts:78` (use `brokerEnv: "0"` instead of `flagOn: false`) |
|
|
717
819
|
| Source edit seen immediately | No, requires bundle rebuild + reload | n/a (permanent) | `index.ts:1-25` — bundle resolution rules |
|
|
718
820
|
| Skip disabled-path proof | `effectiveEnabled()` regression slips through | n/a (permanent) | Tier 2 above |
|
|
719
|
-
| `npm run test:unit` against the full suite (
|
|
821
|
+
| `npm run test:unit` against the full suite (823 files now, 642 then) | several minutes; mis-judges verifier runtime | n/a (permanent) | Tier 1 above |
|
|
720
822
|
| Skip typecheck | TS errors slip past `test:critical` (which uses `--test-timeout=30000`) | n/a (permanent) | Tier 3 above |
|
|
721
823
|
| Run `pi` from a stale bundle | Session shows old behavior despite src/ edits | n/a (permanent) | `scripts/check-bundle-staleness.mjs` — CI gate |
|
|
722
824
|
| Test by reading code | Proves nothing about runtime | n/a (permanent) | All tiers above |
|
|
@@ -740,6 +842,9 @@ node scripts/check-bundle-staleness.mjs --committed-hash # "OK: committed dist
|
|
|
740
842
|
| **Fix finding của reviewer mà không tự verify** (deep review 2026-09-10): 1 trong 4 HIGH findings là false positive — "background-runner exit-loss" thực tế được cover bởi EL-2 `flushBufferedQueuesSync()` (sync lock + appendFileSync + fsync) trên `process.on("exit")` tại event-log.ts:1227. Fix theo finding mù quáng sẽ ĐÃ THÊM regression. | n/a (process) | Mọi finding trước khi fix: trace counter-evidence (exit handlers, sync flush paths). Finding = hypothesis, không phải fact. |
|
|
741
843
|
| **Duplicated defaults map drift (G17-class)** (P0 remediation, `b6eba80f`): 2 bản EFFECTIVE_DEFAULTS (`settings-overlay.ts`, `handle-settings.ts`) hardcode `"aboveEditor"` trong khi nguồn chân lý (defaults.ts/install.mjs) nói `"bottom"` — suite không có test so 2 bản với nhau, drift sống sót qua 7500 tests. | `b6eba80f` | Defaults phải có MỘT nguồn chân lý, hoặc test so các bản sao. Live probe: `team-settings get <key>`. Xem Tier 11g. |
|
|
742
844
|
| **Test vacuous — assert trên fixture chứ không trên wiring** (P1 remediation, `09dda842`): migration-validator test 2 từng assert key tự chế không có trong registry → luôn pass dù validator chưa được wire vào registerPiTeams. | `09dda842` | Test phải dùng key THẬT từ registry (`PI_CREW_BROKER_DIAG_UI` severity "removed"), và wiring test phải prove call-site (register.ts:68), không chỉ prove pure function. |
|
|
845
|
+
| **Folded YAML scalar (`key: >`) in agent/team/workflow frontmatter** (Batch-9 regression, fixed `aa899a1e`): `utils/frontmatter.ts` is line-based — folded descriptions parsed as literal `">"` for ALL 17 agents while typecheck/lint/test:critical stayed green (skills unaffected: real `yaml` package). Symptom: guidance renders `name (builtin): >`, When-NOT text missing. | `aa899a1e` | Agent/teams/workflows frontmatter values stay single-line; quote values containing `": "` (parser strips symmetric quotes); run the Tier 12b dual-parse probe after EVERY resource `.md` frontmatter edit. Folded scalars remain fine in `skills/*/SKILL.md` only. |
|
|
846
|
+
| **Worker killed mid long-silent-bash** (Batch-1 postmortem): one >5–10 min command (full `npm test` ≈ 10 min) emits no LLM activity → the broker's responsiveness check SIGTERMs the worker mid-run — exit 143 WHILE the command runs, not a 600s response timeout. Work was intact; manual re-run green — the kill WAS the "hang". | n/a (quirk — `CONTEXT.md` Flagged #1; P2 candidate) | Split long suites into <5 min chunks or emit progress between commands. On exit-143-mid-command: re-run manually BEFORE diagnosing a code bug. Postmortem: `postmortem-batch-1-sigterm.md` (workspace root). |
|
|
847
|
+
| **Treating `wait-request-broker.test.ts` load-timeout as a product bug**: the test runner's per-file 180s timeout is below this file's full-suite runtime under parallel load — fails only with the whole suite, passes in isolation. Pre-existing flake, NOT a regression from your change. | n/a (test infra) | Re-run the single file before fixing anything: `node scripts/test-runner.mjs test/unit/runtime/broker/wait-request-broker.test.ts`. Green in isolation = infra flake; move on. |
|
|
743
848
|
|
|
744
849
|
---
|
|
745
850
|
|
|
@@ -769,6 +874,9 @@ When a tier fails, the recovery is usually quick. Match the symptom to the cause
|
|
|
769
874
|
| `delegate` rejects with a policy message | By design when depth cap hit (`maxDepth: 4`) or `nesting.enabled: false` in USER config (sensitive — project cannot flip) | Check depth in the rejection payload; `delegate.rejected` event in events.jsonl confirms the structured (non-silent) path |
|
|
770
875
|
| herdr provider never engages | pi is not itself running inside a herdr pane (design: no socket guessing) | Run pi inside herdr, then `runtime.surface.mode` auto/`herdr`; verify `~/.config/herdr/herdr.sock` responds |
|
|
771
876
|
| Surface run >5 phút bị stale-reconcile giết oan (worker khỏe, pane sống) | F1 (đã fix f12f4f5d + af2f8eb4): recorder chỉ flush ở turn boundary → lastSeen đóng băng giữa turn; reconciler cũ time-based không pid-gate. **Bẫy đa host**: MỘT pi session chạy bundle cũ cũng đủ giết run của session khác (sweep quét mọi runs) — tát cả host phải cùng version | Kiểm tra mọi pi process cùng bundle (`ps` lstart vs dist mtime); `PI_CREW_DEBUG_STALE=1` sidecar /tmp/pi-crew-f1-debug.log ghi mọi verdict STALE để bắt hung thủ; kỳ vọng sidecar rỗng khi mọi host đã fix |
|
|
877
|
+
| Worker exits 143 WHILE a long bash command is still running (no LLM-activity window before the kill) | Broker responsiveness SIGTERM on silent long commands (`CONTEXT.md` Flagged #1) — distinct from `RESPONSE_TIMEOUT_MS` (600s no-response) | Split the command; emit progress between steps; re-run the suite manually — work is usually intact. See `postmortem-batch-1-sigterm.md` |
|
|
878
|
+
| Full `test:unit` fails ONLY on `wait-request-broker.test.ts` under parallel load | Per-file 180s runner timeout vs the file's real runtime (passes isolated) | `node scripts/test-runner.mjs test/unit/runtime/broker/wait-request-broker.test.ts` — green in isolation = infra flake, not a regression |
|
|
879
|
+
| Guidance / `team action='list'` shows an agent description as `>` or missing When-NOT text | Folded-scalar frontmatter (`description: >`) — the line-based parser reads `>` literally (CONTEXT.md Flagged #4) | Restore the single-line value (double-quote it if it contains `": "`); re-run the Tier 12b dual-parse probe |
|
|
772
880
|
|
|
773
881
|
## Performance budget (per-tier soft limits)
|
|
774
882
|
|
|
@@ -862,6 +970,7 @@ The "skill stack" for a typical pi-crew change:
|
|
|
862
970
|
6. tier 7 (smoke team) ← this skill, if plan/workflow change
|
|
863
971
|
7. tier 9 (feature battery) ← this skill, if schema/tool-surface change
|
|
864
972
|
8. tier 10 (surface battery) ← this skill, if surface/pane change
|
|
973
|
+
8b. tier 12 (resource contracts) ← this skill, if agents/skills .md or discovery change
|
|
865
974
|
9. commit + push
|
|
866
975
|
10. verify-before-complete ← make the "done" claim with evidence
|
|
867
976
|
```
|
|
@@ -905,6 +1014,16 @@ Use this to answer "đủ full tính năng chưa?" without re-deriving. Every us
|
|
|
905
1014
|
| ui.widgetPlacement default | `settings-overlay.ts:351`, `handle-settings.ts:43` | T11g + live team-settings get |
|
|
906
1015
|
| Worktree twins contract | `src/worktree/worktree-manager.ts` | T11h (3× consecutive runs) |
|
|
907
1016
|
| Bundle committed-hash gate | `scripts/check-bundle-staleness.mjs --committed-hash` | T11j |
|
|
1017
|
+
| Agent routing metadata (`useWhen`/`avoidWhen`/`cost`/`category`; 18 agents) | `agents/*.md` frontmatter; parsed `src/agents/discover-agents.ts:388-391` | T12b (dual parse) + T12c (guidance render) |
|
|
1018
|
+
| Agent output contracts (`## Output format` + fenced block, all builtins) | enforced by `test/unit/agents/agent-output-contracts.test.ts` | T12a |
|
|
1019
|
+
| Orchestrator agent (18th builtin, delegated orchestration) | `agents/orchestrator.md` | 9a `team action='list'` shows 18 + T12 |
|
|
1020
|
+
| Skill override wildcard/denylist (`*`, `!name`) | `src/runtime/skill-instructions.ts` (`collectTaskSkillNames`) | T12d (skill-instructions unit tests) |
|
|
1021
|
+
| Tool loop guard (read-only tools warn@3/block@5; ask wait-guard) | `src/extension/registration/tool-loop-guard.ts`; config `runtime.reliability.loopGuard` | `tool-loop-guard.test.ts` (12 tests, T12d) + live: same read-only tool 5× → structured block; exempt tools (team/Agent/…) unaffected |
|
|
1022
|
+
| Post-init skill check (SKILL.md presence/severity) | `src/extension/post-init-skill-check.ts`, wired `register.ts:132` | startup log probe: `[pi-crew] …` warn/error only when skills broken |
|
|
1023
|
+
| Detached-run delivery bound (3 attempts → drop + warn) | `src/runtime/detached-run-results.ts` (`MAX_DELIVERY_ATTEMPTS`) | unit tests |
|
|
1024
|
+
| Byte-stable worker prefix (ARCH-3) | `src/runtime/task-runner/prompt-builder.ts` stablePrefix/dynamicSuffix split | byte-identity unit test (strictEqual) |
|
|
1025
|
+
| Release tarball import gate (ARCH-6) | `scripts/release-smoke.mjs` (installs pi-* peers, `import()`s installed bundle `:77`, shape-checks exports) | release cut: `node scripts/release-smoke.mjs` |
|
|
1026
|
+
| Bundle path-leak scan (ARCH-7) | `scripts/check-bundle-staleness.mjs` (line-scan dist + structural sourcemap check) | T3 staleness run + T11j |
|
|
908
1027
|
|
|
909
1028
|
---
|
|
910
1029
|
|
|
@@ -923,7 +1042,11 @@ The skill mentions specific commits, line numbers, and version pins. As the code
|
|
|
923
1042
|
| Verify herdr wire details | Each herdr release bump | `herdr api schema --json` vs `src/runtime/surface/herdr-provider.ts` (envelope/pane.read source/1-conn-per-request were verified on herdr 0.8.2) |
|
|
924
1043
|
| Verify Tier 11 census numbers | Each `src/state/**` write-path commit | `grep -rln "appendEventBuffered" src/ \| wc -l` — update the 16-file / 43-conversion anchor in 11a when it drifts |
|
|
925
1044
|
| Verify wc-gate still enforced | Each `package.json` / ci.yml edit | `node -e "require('./package.json').scripts.ci.includes('check:wc-gate')"` + grep ci.yml — a gate removed from `ci` reverts to advisory |
|
|
926
|
-
| Verify migration-validator wiring | Each `register.ts` refactor | `grep -n validateEnv src/extension/register.ts` — must stay after `installChildProcessAbortShield`, before `startRuntimeWarmup`, warn-only |
|
|
1045
|
+
| Verify migration-validator wiring | Each `register.ts` refactor | `grep -n validateEnv src/extension/register.ts` — must stay after `installChildProcessAbortShield`, before `startRuntimeWarmup`, warn-only. ALSO `grep -n runPostInitSkillCheck src/extension/register.ts` (:132) — async post-init, warn/error only |
|
|
1046
|
+
| Verify builtin agent count + contracts | Each `agents/*.md` commit | Tier 12a/12b — **18 @ 2026-09-11** (`aa899a1e`); update this skill's count when it changes |
|
|
1047
|
+
| Verify frontmatter stays single-line/quoted | Each `agents/`, `teams/`, `workflows/` `.md` edit | Tier 12b dual-parse probe — BOTH discovery and strict `yaml` must pass |
|
|
1048
|
+
| Verify staleness leak-scan still runs | Each `check-bundle-staleness.mjs` edit | `node scripts/check-bundle-staleness.mjs` after `build:bundle` — exit 0 (staleness + path-leak) |
|
|
1049
|
+
| Verify release-smoke peer pins + import gate | Each `release-smoke.mjs` edit / release cut | `node scripts/release-smoke.mjs` — peer install + import + shape checks green |
|
|
927
1050
|
|
|
928
1051
|
The skill does NOT need to be updated for every commit — only when the cited lines/files move. Consider it a "living reference" not a "live spec".
|
|
929
1052
|
|
|
@@ -981,6 +1104,13 @@ grep -rn '"ui.widgetPlacement"' src/ui/settings-overlay.ts src/extension/team-to
|
|
|
981
1104
|
for i in 1 2 3; do node --experimental-strip-types --no-warnings --test test/unit/worktree/worktree-twins-contract.test.ts 2>&1 | grep -E '^# (pass|fail)'; done # 11h
|
|
982
1105
|
grep -rln 'appendEventBuffered' src/ | wc -l # 11a: census (16 files @ v0.10.5)
|
|
983
1106
|
node scripts/check-bundle-staleness.mjs --committed-hash # 11j: OK
|
|
1107
|
+
# Tier 12 (resource contracts — agents/skills .md + discovery changes)
|
|
1108
|
+
node --experimental-strip-types --no-warnings --test --test-force-exit test/unit/agents/agent-output-contracts.test.ts # 12a
|
|
1109
|
+
node --experimental-strip-types --no-warnings -e 'import("./src/agents/discover-agents.ts").then(m=>{m.invalidateAgentDiscoveryCache();const l=m.discoverAgents(process.cwd()).builtin;console.log("agents:",l.length,"| bad desc:",l.filter(a=>!a.description?.includes("When NOT to use:")).length,"| no routing:",l.filter(a=>!a.routing?.useWhen).length);})' # 12b: 18 | 0 | 0
|
|
1110
|
+
node -e 'const yaml=require("yaml"),fs=require("fs");let ok=0;for(const f of fs.readdirSync("agents")){const m=/^---\r?\n([\s\S]*?)\r?\n---/.exec(fs.readFileSync("agents/"+f,"utf-8"));if(m){try{if(yaml.parse(m[1]).name)ok++;}catch{}}}console.log("strict YAML:",ok)' # 12b: 18
|
|
1111
|
+
node --experimental-strip-types --no-warnings --test --test-force-exit test/unit/bundle-skill-resolution.test.ts test/unit/extension/registration/tool-loop-guard.test.ts test/unit/runtime/core/skill-instructions.test.ts # 12d
|
|
1112
|
+
node scripts/check-bundle-staleness.mjs # staleness + ARCH-7 path-leak scan (also after every build:bundle)
|
|
1113
|
+
node scripts/release-smoke.mjs # release cut: peer install + tarball import + shape check (ARCH-6)
|
|
984
1114
|
# 11a full gate (after ANY delayed-write conversion program): npm run test:unit # ~7500 tests, 15-18 min
|
|
985
1115
|
```
|
|
986
1116
|
|
|
@@ -1001,6 +1131,7 @@ Before claiming "tested":
|
|
|
1001
1131
|
- [ ] **Output report**: save `docs/real-test/reports/real-test-<YYYY-MM-DD>-<slug>.md` from `skills/real-test-pi-crew/REPORT-TEMPLATE.md`, filled DURING the run with per-tier evidence (counts/md5/runId) — not reconstructed from memory afterward. This is what makes past runs verifiable instead of trust-the-summary.
|
|
1002
1132
|
- [ ] Tier 10: surface battery — **required if you touched `src/runtime/surface/**`, `src/prompt/surface-worker.ts`, the surface branch of `src/runtime/child-pi/child-pi.ts`, or the surface config keys**. 10a E2E 3/3 per backend available (tmux trong tmux; herdr ngoài tmux + socket sống — skip vì thiếu mux là correct-by-design nhưng KHÔNG tính pass cho backend đó); 10b live run với session ĐÃ reload bundle mới (xem Anti-patterns "file-md5 only") + `visibleAgents` set + pane-level evidence (pane id/title during run, `worker.surface_spawned`/`worker.surface_closed` events, pane auto-closed after — KHÔNG dùng `manifest.surface.panes` làm evidence engage, xem Anti-patterns "panes == {}"); 10c herdr live chỉ khi pi chạy trong herdr pane (skip kèm lý do nếu không).
|
|
1003
1133
|
- [ ] Tier 11: remediation regression battery — **required if you touched `src/state/**` write paths, `migration-validator.ts`/its wiring, `scripts/wc-gate.mjs` or `ci` scripts, `.github/workflows/*` env, EFFECTIVE_DEFAULTS maps, or you are cutting a release**. Sub-checks a–j per Tier 11; 11a item 4 (full `test:unit`) mandatory after any delayed-write conversion program, skippable for doc-only changes. Record: buffered-site census count, wc-gate max, staleness `--committed-hash` result.
|
|
1134
|
+
- [ ] Tier 12: resource-contract battery — **required if you touched `agents/*.md`, `skills/*/SKILL.md`, `src/agents/discover-agents.ts`, `src/skills/discover-skills.ts`, `src/utils/frontmatter.ts`, `src/runtime/skill-instructions.ts`, or `src/extension/autonomous-policy.ts`**. 12a contracts green; 12b BOTH parsers clean (agent count — **18 @ 2026-09-11** — 0 bad descriptions, 0 missing routing, 0 strict-YAML fails); 12c every rendered agent line carries `useWhen=` (budget-truncated by design; newest agent visible); 12d unit batteries pass. Agent/skill-only changes need NO bundle rebuild (runtime-loaded from the package dir) — `src/` changes in the same commit still follow the Tier 3 bundle rule.
|
|
1004
1135
|
|
|
1005
1136
|
**"All tiers pass" is a claim that needs per-row evidence.** Tier 9 means 9a **and** 9b **and** whichever of 9c–9f applies to the change — not "9a passed, therefore 9 passed". Tier 10 means pane-level evidence exists, not "run went green" (surface fail-closes to headless on every failure, so green proves nothing). If any required item above is unchecked or lacks concrete evidence (a number, an md5, a runId, a pane id), the answer to "is it tested?" is **no** — say so explicitly instead of rounding up to "pass".
|
|
1006
1137
|
|
|
@@ -1068,6 +1199,28 @@ Workflow files:
|
|
|
1068
1199
|
- `workflows/plan-execute.workflow.md:30` — verifier prompt
|
|
1069
1200
|
- `workflows/review.workflow.md:31` — verifier prompt
|
|
1070
1201
|
|
|
1202
|
+
Resource-contract files (Tier 12):
|
|
1203
|
+
- `src/utils/frontmatter.ts` — LINE-BASED parser (`parseLines`): single-line values, symmetric-quote strip (`aa899a1e`); folded scalars unsupported for agents/teams/workflows (skills use the real `yaml` package — folded OK there)
|
|
1204
|
+
- `src/agents/discover-agents.ts:388-391, 476` — flat routing keys (`useWhen`/`avoidWhen`/`cost`/`category` as top-level CSV); discovery cache TTL ~30s (`invalidateAgentDiscoveryCache()`)
|
|
1205
|
+
- `src/extension/autonomous-policy.ts` — `buildResourceRoutingGuidance` renders routing cards into the leader's injected policy (the single canonical routing source)
|
|
1206
|
+
- `src/runtime/skill-instructions.ts` — `collectTaskSkillNames`: `*` wildcard + `!name` denylist skill overrides
|
|
1207
|
+
- `src/extension/registration/tool-loop-guard.ts` — ARCH-1 loop guard: read-only tools warn@3/block@5, ask wait-guard warn@2/block@3rd, FIFO 512; exempts team/crew_agent/Agent/get_subagent_result; config `runtime.reliability.loopGuard`
|
|
1208
|
+
- `src/extension/post-init-skill-check.ts` (32L) — SKILL.md presence check; wired async at `register.ts:132`, warn/error log only
|
|
1209
|
+
- `src/runtime/detached-run-results.ts` — `MAX_DELIVERY_ATTEMPTS = 3`; drop + `detached-run-results.delivery-gave-up` log
|
|
1210
|
+
- `test/unit/agents/agent-output-contracts.test.ts` — output-contract AC across ALL builtin agents
|
|
1211
|
+
- `test/unit/bundle-skill-resolution.test.ts` + `test/unit/extension/registration/tool-loop-guard.test.ts` (12 tests) + `test/unit/runtime/core/skill-instructions.test.ts` (26 tests)
|
|
1212
|
+
- `scripts/release-smoke.mjs` — ARCH-6: installs pi-* peers, `import()`s the tarball-installed bundle (`:77`), shape-checks exports
|
|
1213
|
+
- `CONTEXT.md` — repo orientation: glossary + Flagged quirks (#1 broker SIGTERM, #2 wait-broker flake, #4 frontmatter parser)
|
|
1214
|
+
|
|
1215
|
+
Batch-1..10 wave (branch `fix/bundle-skill-resolution-and-skill-meta`, 2026-09-11, base v0.10.5):
|
|
1216
|
+
- `c97bc578` — BUG-1 packageRoot skill resolution + SKILL-HYGIENE-1 post-init check + SKILL-HYGIENE-2 `*`/`!name` + SKILL-META-1 (34 skills When-NOT — folded OK for skills)
|
|
1217
|
+
- `3de89a2f` / `07c5e014` / `24c63c75` / `b120187f` — skill Budget/Self-restraint + agent body upgrades (all roles; librarian/oracle/designer added)
|
|
1218
|
+
- `60e2cb96` — councillor agents (`inheritProjectContext: false`, deny-all-write toolset)
|
|
1219
|
+
- `d36ad4eb` — ARCH-1 tool loop guard + ARCH-3 byte-stable prefix
|
|
1220
|
+
- `7d18508b` — ARCH-2/5/6/7 (ARCH-4 skipped per ADR 2026-08-15 — live-session frozen)
|
|
1221
|
+
- `06c5d7ca` — PROMPT-1/2/5 (output-contract AC, task-rejection line, agent When-NOT — introduced the folded-scalar regression)
|
|
1222
|
+
- `aa899a1e` — Batch 10: routing metadata (18 agents), orchestrator, delivery bound, CONTEXT.md, folded-scalar fix + quote-strip
|
|
1223
|
+
|
|
1071
1224
|
Commits (chronological, the patterns they introduced):
|
|
1072
1225
|
- `1cb2dca` — `test:critical` script + plan-templates verifier fix
|
|
1073
1226
|
- `d599578` — 4 workflow verifier prompt fixes
|
|
@@ -1,6 +1,9 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: requirements-to-task-packet
|
|
3
|
-
description:
|
|
3
|
+
description: >
|
|
4
|
+
Use when a goal, issue, roadmap item, review finding, or user request must become actionable worker tasks.
|
|
5
|
+
When NOT to use: already-decided work that just needs execution; tasks so simple they fit one sentence.
|
|
6
|
+
|
|
4
7
|
origin: pi-crew
|
|
5
8
|
triggers:
|
|
6
9
|
- "convert requirements"
|
|
@@ -108,3 +111,9 @@ If ANY answer is NO → Stop. Complete task packet before dispatching.
|
|
|
108
111
|
- Buried assumptions.
|
|
109
112
|
- Expanding scope because context remains.
|
|
110
113
|
- Treating tests as proof when the requirement was never asserted.
|
|
114
|
+
|
|
115
|
+
## Self-restraint
|
|
116
|
+
|
|
117
|
+
"Creating nothing is a valid result." If the evidence does not support a meaningful change, say so explicitly rather than inventing one. The next attempt may find stronger evidence; an invented change now damages trust in every future report.
|
|
118
|
+
|
|
119
|
+
"Creating nothing" here means the requirements are already actionable as-is — no packet needed. Inventing scope or splitting trivial work into packets adds overhead without value.
|
package/skills/research/SKILL.md
CHANGED
|
@@ -1,6 +1,9 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: research
|
|
3
|
-
description:
|
|
3
|
+
description: >
|
|
4
|
+
Deep-research skill combining iterative depth, structured+validated output, rigor mechanisms, anti-thrash, and pi-native hooks for general deep research. REQUIRED — read the full skill file first (iterative-depth protocol with rigor scripts); run verify_citations and source_evaluator on your output before claiming done.
|
|
5
|
+
When NOT to use: codebase-specific questions (use read-only-explorer); post-implementation audit (use iterative-audit).
|
|
6
|
+
|
|
4
7
|
origin: local
|
|
5
8
|
language: en
|
|
6
9
|
distilled_against: 4-source-field-snapshot
|
|
@@ -1,6 +1,9 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: resource-discovery-config
|
|
3
|
-
description:
|
|
3
|
+
description: >
|
|
4
|
+
pi-crew resource and configuration discovery workflow.
|
|
5
|
+
When NOT to use: runtime state questions (use runtime-state-reader); model-specific config (use model-routing-context).
|
|
6
|
+
|
|
4
7
|
origin: pi-crew
|
|
5
8
|
triggers:
|
|
6
9
|
- "discover agents"
|
|
@@ -58,3 +61,9 @@ node --experimental-strip-types --test test/unit/config-schema-validation.test.t
|
|
|
58
61
|
npm test
|
|
59
62
|
npm pack --dry-run
|
|
60
63
|
```
|
|
64
|
+
|
|
65
|
+
## Self-restraint
|
|
66
|
+
|
|
67
|
+
"Creating nothing is a valid result." If the evidence does not support a meaningful change, say so explicitly rather than inventing one. The next attempt may find stronger evidence; an invented change now damages trust in every future report.
|
|
68
|
+
|
|
69
|
+
"Creating nothing" here means the current discovery already finds what consumers need. Registering resources nobody uses pollutes the discovery surface and slows every lookup.
|
|
@@ -1,6 +1,9 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: runtime-state-reader
|
|
3
|
-
description:
|
|
3
|
+
description: >
|
|
4
|
+
Safe read-only navigation of pi-crew run state.
|
|
5
|
+
When NOT to use: write actions or mutations (use state-mutation-locking); configuration changes (use resource-discovery-config).
|
|
6
|
+
|
|
4
7
|
origin: pi-crew
|
|
5
8
|
triggers:
|
|
6
9
|
- "inspect manifest"
|
|
@@ -1,6 +1,9 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: safe-bash
|
|
3
|
-
description:
|
|
3
|
+
description: >
|
|
4
|
+
Safe shell-command workflow.
|
|
5
|
+
When NOT to use: subprocess orchestration (use child-pi-spawning); destructive operations without sandbox (use ownership-session-security).
|
|
6
|
+
|
|
4
7
|
origin: pi-crew
|
|
5
8
|
triggers:
|
|
6
9
|
- "run this command"
|
|
@@ -1,6 +1,9 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: scrutinize
|
|
3
|
-
description:
|
|
3
|
+
description: >
|
|
4
|
+
Outsider-perspective review questioning intent before tracing code.
|
|
5
|
+
When NOT to use: implementation work (use systematic-debugging); trivial one-line fixes that don't warrant outsider review.
|
|
6
|
+
|
|
4
7
|
origin: pi-crew
|
|
5
8
|
triggers:
|
|
6
9
|
- "scrutinize this"
|
|
@@ -84,3 +87,23 @@ If ANY answer is NO → Stop. Complete scrutiny requirements before reporting.
|
|
|
84
87
|
- **One simpler-alternative pass is MANDATORY.** Skip only if user says "don't question scope."
|
|
85
88
|
- **Distinguish claim from verification.** "The PR says X" and "I traced X and confirmed" are different.
|
|
86
89
|
- **No flattery, no hedging.** State the finding.
|
|
90
|
+
|
|
91
|
+
## Budget
|
|
92
|
+
|
|
93
|
+
This skill applies a 3-attempt budget: 1 initial + max 2 re-attempts.
|
|
94
|
+
|
|
95
|
+
Stamp every invocation:
|
|
96
|
+
|
|
97
|
+
```
|
|
98
|
+
attempt X of 3 (Y attempts remaining)
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
An attempt is one full outsider-perspective review pass. Re-attempt when the review uncovers intent-level questions that change the approach.
|
|
102
|
+
|
|
103
|
+
Re-attempts only when the previous attempt materially changes the decision or risk. Do NOT spend a re-attempt on mechanical changes or already-resolved findings. When exhausted, escalate to the user with options (accept risk / change scope / exceptional budget).
|
|
104
|
+
|
|
105
|
+
## Self-restraint
|
|
106
|
+
|
|
107
|
+
"Creating nothing is a valid result." If the evidence does not support a meaningful change, say so explicitly rather than inventing one. The next attempt may find stronger evidence; an invented change now damages trust in every future report.
|
|
108
|
+
|
|
109
|
+
"Creating nothing" here means concluding the intent was sound and the approach justified — no findings needed. An invented finding to justify the review round damages trust in every future review.
|
|
@@ -1,6 +1,9 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: secure-agent-orchestration-review
|
|
3
|
-
description:
|
|
3
|
+
description: >
|
|
4
|
+
Use when reviewing delegation, skill loading, tool access, worker prompts, artifacts, runtime config, state, ownership, or subprocess execution.
|
|
5
|
+
When NOT to use: pure code quality issues without security implications (use multi-perspective-review); single-file fixes (use scrutinize).
|
|
6
|
+
|
|
4
7
|
origin: pi-crew
|
|
5
8
|
triggers:
|
|
6
9
|
- "review delegation"
|
|
@@ -1,6 +1,9 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: state-mutation-locking
|
|
3
|
-
description:
|
|
3
|
+
description: >
|
|
4
|
+
Durable state mutation and locking workflow.
|
|
5
|
+
When NOT to use: read-only inspection (use runtime-state-reader); single-process non-concurrent edits.
|
|
6
|
+
|
|
4
7
|
origin: pi-crew
|
|
5
8
|
triggers:
|
|
6
9
|
- "modify manifest"
|
|
@@ -1,6 +1,9 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: systematic-debugging
|
|
3
|
-
description:
|
|
3
|
+
description: >
|
|
4
|
+
Four-phase debugging discipline with refuse gates.
|
|
5
|
+
When NOT to use: trivial fixes where root cause is obvious; production incidents needing immediate rollback (use post-mortem after).
|
|
6
|
+
|
|
4
7
|
origin: pi-crew
|
|
5
8
|
triggers:
|
|
6
9
|
- "debug this"
|
|
@@ -1,6 +1,9 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: verification-before-done
|
|
3
|
-
description:
|
|
3
|
+
description: >
|
|
4
|
+
Evidence before claims.
|
|
5
|
+
When NOT to use: mid-task self-correction (use scrutinize); pure execution tasks where output IS evidence.
|
|
6
|
+
|
|
4
7
|
origin: pi-crew
|
|
5
8
|
triggers:
|
|
6
9
|
- "done"
|
|
@@ -80,3 +83,17 @@ Stop before saying done if you are using words like "should", "probably", "looks
|
|
|
80
83
|
- **Don't** use fuzzy language like "seems", "probably", "looks like"
|
|
81
84
|
- **Don't** skip providing verification commands for claims
|
|
82
85
|
- **Don't** claim done if you're still using hypotheses instead of evidence
|
|
86
|
+
|
|
87
|
+
## Budget
|
|
88
|
+
|
|
89
|
+
This skill applies a 3-attempt budget: 1 initial + max 2 re-attempts.
|
|
90
|
+
|
|
91
|
+
Stamp every invocation:
|
|
92
|
+
|
|
93
|
+
```
|
|
94
|
+
attempt X of 3 (Y attempts remaining)
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
An attempt is one fresh verification run (identify command → run → read output → compare to claim). Re-attempt when output contradicts the claim or reveals a new failure mode.
|
|
98
|
+
|
|
99
|
+
Re-attempts only when the previous attempt materially changes the decision or risk. Do NOT spend a re-attempt on mechanical changes or already-resolved findings. When exhausted, escalate to the user with options (accept risk / change scope / exceptional budget).
|
|
@@ -1,6 +1,9 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: widget-rendering
|
|
3
|
-
description:
|
|
3
|
+
description: >
|
|
4
|
+
Pi TUI crew widget data sources, display priority, and rendering performance.
|
|
5
|
+
When NOT to use: non-TUI displays; backend data sources without widget context (use runtime-state-reader).
|
|
6
|
+
|
|
4
7
|
origin: pi-crew
|
|
5
8
|
triggers:
|
|
6
9
|
- "empty agent"
|
|
@@ -1,6 +1,9 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: workspace-isolation
|
|
3
|
-
description:
|
|
3
|
+
description: >
|
|
4
|
+
Workspace isolation boundaries.
|
|
5
|
+
When NOT to use: concurrent edits within same repo (use worktree-isolation); non-git projects.
|
|
6
|
+
|
|
4
7
|
origin: pi-crew
|
|
5
8
|
triggers:
|
|
6
9
|
- "workspace isolation"
|
|
@@ -1,6 +1,9 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: worktree-isolation
|
|
3
|
-
description:
|
|
3
|
+
description: >
|
|
4
|
+
Conflict-safe git worktree workflow.
|
|
5
|
+
When NOT to use: non-git changes (use direct edit); single-commit patches (use git-master).
|
|
6
|
+
|
|
4
7
|
origin: pi-crew
|
|
5
8
|
triggers:
|
|
6
9
|
- "create worktree"
|
|
@@ -640,6 +640,7 @@ function parseReliabilityConfig(value: unknown): CrewReliabilityConfig | undefin
|
|
|
640
640
|
forcePreflight: parseWithSchema(Type.Boolean(), obj.forcePreflight),
|
|
641
641
|
ambientStatusInjection: parseWithSchema(Type.Boolean(), obj.ambientStatusInjection),
|
|
642
642
|
perWriteValidation: parseWithSchema(Type.Boolean(), obj.perWriteValidation),
|
|
643
|
+
loopGuard: parseWithSchema(Type.Boolean(), obj.loopGuard),
|
|
643
644
|
scopeModels: parseWithSchema(Type.Boolean(), obj.scopeModels),
|
|
644
645
|
};
|
|
645
646
|
return Object.values(reliability).some((entry) => entry !== undefined) ? reliability : undefined;
|
package/src/config/types.ts
CHANGED
|
@@ -274,6 +274,14 @@ export interface CrewReliabilityConfig {
|
|
|
274
274
|
* Set to `false` to disable.
|
|
275
275
|
*/
|
|
276
276
|
perWriteValidation?: boolean;
|
|
277
|
+
/**
|
|
278
|
+
* Tool loop guard (ARCH-1). Warns at 3 consecutive identical tool results
|
|
279
|
+
* (identical args + byte-identical output) and hard-blocks read-only file
|
|
280
|
+
* tools (read/grep/glob/find/ls) at 5 — the model-side infinite-loop
|
|
281
|
+
* failure mode. `ask` repeats within a turn are warned at 2 and the 3rd
|
|
282
|
+
* call refused. Default: true (opt-out). Set to `false` to disable.
|
|
283
|
+
*/
|
|
284
|
+
loopGuard?: boolean;
|
|
277
285
|
/**
|
|
278
286
|
* Opt-in model scope enforcement (F7). When true, subagent model choices
|
|
279
287
|
* that fall outside the user's pi `enabledModels` allowlist are flagged:
|
package/src/errors.ts
CHANGED
|
@@ -64,7 +64,7 @@ const DEFAULT_HELP: Record<ErrorCode, string | undefined> = {
|
|
|
64
64
|
[ErrorCode.RunStale]:
|
|
65
65
|
"The worker stopped heartbeating and was treated as a zombie. Re-run the team (resume or fresh); if it recurs, check `runtime.executeWorkers` / system load.",
|
|
66
66
|
[ErrorCode.ModelOutOfScope]:
|
|
67
|
-
"The requested model is not in your pi `enabledModels` allowlist. Either pick a model listed in `enabledModels` (settings.json) or extend the allowlist. The scope gate is opt-in — disable `
|
|
67
|
+
"The requested model is not in your pi `enabledModels` allowlist. Either pick a model listed in `enabledModels` (settings.json) or extend the allowlist. The scope gate is opt-in — disable `reliability.scopeModels` to allow any model.",
|
|
68
68
|
};
|
|
69
69
|
|
|
70
70
|
/**
|