pi-crew 0.10.4 → 0.10.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (88) hide show
  1. package/CHANGELOG.md +233 -0
  2. package/agents/analyst.md +37 -2
  3. package/agents/cold-verifier.md +10 -1
  4. package/agents/councillor-critic.md +39 -0
  5. package/agents/councillor-pragmatist.md +39 -0
  6. package/agents/councillor-skeptic.md +41 -0
  7. package/agents/critic.md +40 -2
  8. package/agents/designer.md +58 -0
  9. package/agents/executor.md +39 -2
  10. package/agents/explorer.md +38 -2
  11. package/agents/librarian.md +49 -0
  12. package/agents/oracle.md +54 -0
  13. package/agents/orchestrator.md +48 -0
  14. package/agents/planner.md +41 -2
  15. package/agents/reviewer.md +39 -2
  16. package/agents/security-reviewer.md +43 -2
  17. package/agents/test-engineer.md +48 -2
  18. package/agents/verifier.md +14 -1
  19. package/agents/writer.md +32 -2
  20. package/dist/index.mjs +1297 -853
  21. package/package.json +1 -1
  22. package/skills/async-worker-recovery/SKILL.md +4 -1
  23. package/skills/child-pi-spawning/SKILL.md +4 -1
  24. package/skills/context-artifact-hygiene/SKILL.md +4 -1
  25. package/skills/council/SKILL.md +24 -45
  26. package/skills/delegation-patterns/SKILL.md +18 -1
  27. package/skills/distill-persona/SKILL.md +4 -1
  28. package/skills/distill-software/SKILL.md +4 -1
  29. package/skills/event-log-tracing/SKILL.md +4 -1
  30. package/skills/git-master/SKILL.md +4 -1
  31. package/skills/iterative-audit/SKILL.md +4 -1
  32. package/skills/live-agent-lifecycle/SKILL.md +4 -1
  33. package/skills/mailbox-interactive/SKILL.md +4 -1
  34. package/skills/model-routing-context/SKILL.md +10 -1
  35. package/skills/multi-perspective-review/SKILL.md +18 -1
  36. package/skills/observability-reliability/SKILL.md +4 -1
  37. package/skills/orchestration/SKILL.md +18 -1
  38. package/skills/ownership-session-security/SKILL.md +4 -1
  39. package/skills/pi-extension-lifecycle/SKILL.md +4 -1
  40. package/skills/post-mortem/SKILL.md +4 -1
  41. package/skills/read-only-explorer/SKILL.md +4 -1
  42. package/skills/real-test-pi-crew/SKILL.md +165 -12
  43. package/skills/requirements-to-task-packet/SKILL.md +10 -1
  44. package/skills/research/SKILL.md +4 -1
  45. package/skills/resource-discovery-config/SKILL.md +10 -1
  46. package/skills/runtime-state-reader/SKILL.md +4 -1
  47. package/skills/safe-bash/SKILL.md +4 -1
  48. package/skills/scrutinize/SKILL.md +24 -1
  49. package/skills/secure-agent-orchestration-review/SKILL.md +4 -1
  50. package/skills/state-mutation-locking/SKILL.md +4 -1
  51. package/skills/systematic-debugging/SKILL.md +4 -1
  52. package/skills/verification-before-done/SKILL.md +18 -1
  53. package/skills/widget-rendering/SKILL.md +4 -1
  54. package/skills/workspace-isolation/SKILL.md +4 -1
  55. package/skills/worktree-isolation/SKILL.md +4 -1
  56. package/src/config/config-validation.ts +1 -0
  57. package/src/config/types.ts +8 -0
  58. package/src/errors.ts +1 -1
  59. package/src/extension/context-status-injection.ts +2 -2
  60. package/src/extension/knowledge-injection.ts +19 -7
  61. package/src/extension/post-init-skill-check.ts +32 -0
  62. package/src/extension/register.ts +9 -1
  63. package/src/extension/registration/hook-registration.ts +20 -3
  64. package/src/extension/registration/tool-loop-guard.ts +243 -0
  65. package/src/extension/team-tool/handle-settings.ts +10 -0
  66. package/src/extension/team-tool/run.ts +42 -1
  67. package/src/extension/team-tool-types.ts +6 -0
  68. package/src/prompt/prompt-runtime.ts +25 -6
  69. package/src/runtime/async-runner.ts +75 -11
  70. package/src/runtime/background-runner.ts +73 -7
  71. package/src/runtime/broker/crew-broker-client.ts +45 -2
  72. package/src/runtime/broker/crew-broker.ts +22 -27
  73. package/src/runtime/broker/protocol/request-parsers.ts +10 -2
  74. package/src/runtime/broker/stdin-handshake.ts +87 -0
  75. package/src/runtime/broker/wait-push.ts +45 -0
  76. package/src/runtime/detached-run-results.ts +25 -1
  77. package/src/runtime/foreground-watchdog.ts +24 -5
  78. package/src/runtime/live-session/live-session-runtime.ts +1 -1
  79. package/src/runtime/model/model-scope.ts +2 -2
  80. package/src/runtime/run-tracker.ts +74 -19
  81. package/src/runtime/skill-instructions.ts +20 -4
  82. package/src/runtime/task-runner/child-executor.ts +1 -1
  83. package/src/runtime/task-runner/prompt-builder.ts +22 -9
  84. package/src/schema/config-schema.ts +1 -0
  85. package/src/skills/discover-skills.ts +2 -2
  86. package/src/ui/settings-overlay.ts +40 -0
  87. package/src/utils/frontmatter.ts +7 -1
  88. package/src/utils/ndjson.ts +9 -1
@@ -1,6 +1,9 @@
1
1
  ---
2
2
  name: real-test-pi-crew
3
- description: "End-to-end verification for pi-crew changes: fast critical tests, 3-path kill-switch proof, bundle md5 sync, live TUI probing, smoke team runs, a live feature-action battery (team tool + subagent tools), and a surface-mode battery (workers in real tmux/herdr panes, degrade-to-headless)."
3
+ description: >
4
+ End-to-end verification for pi-crew changes: fast critical tests, 3-path kill-switch proof, bundle md5 sync, live TUI probing, smoke team runs, a live feature-action battery (team tool + subagent tools), a surface-mode battery (workers in real tmux/herdr panes, degrade-to-headless), and a resource-contract battery (agent .md frontmatter dual-parse, routing render, output contracts).
5
+ When NOT to use: unit tests for isolated modules (use test runner directly); pure test execution.
6
+
4
7
  origin: pi-crew
5
8
  triggers:
6
9
  - "test the change"
@@ -39,16 +42,30 @@ triggers:
39
42
  - "wc-gate"
40
43
  - "migration validator warning"
41
44
  - "slow tier"
45
+ - "agent frontmatter"
46
+ - "folded scalar"
47
+ - "agent body change"
48
+ - "routing metadata"
49
+ - "output contract"
50
+ - "loop guard"
51
+ - "post-init skill check"
52
+ - "resource contract"
53
+ - "sigterm"
54
+ - "silent bash"
55
+ - "worker killed mid command"
56
+ - "tier 12"
42
57
  ---
43
58
 
44
59
  # real-test-pi-crew
45
60
 
46
61
  End-to-end verification discipline for pi-crew changes. Distilled from the broker Phase-4 rollout (commits `1cb2dca` → `d599578` → `612e18b` → `4186284`, July 2026). The pain this skill prevents: shipping code that compiles + unit-tests-green but breaks in the user's live Pi session, or hangs the verifier worker.
47
62
 
48
- **When to use**: after any change to `src/runtime/broker/*.ts` (broker + tokens + issuer), `src/ui/`, `src/config/` (incl. `src/config/migration-validator.ts`), `src/extension/registration/lifecycle-handlers.ts`, `src/runtime/child-pi/*.ts` (worker spawn/kill/steering), `src/runtime/surface/*.ts` (MuxSurface providers, degrade, launch script), `src/prompt/*.ts` (worker-side tools: ask / message / delegate / surface-worker recorder), `src/runtime/goal-workflow/plan-templates.ts`, `src/runtime/team-runner.ts` or `src/runtime/task-runner/**` (scheduler / execution — Tier 7 smoke), `src/state/**` (durable state — Tier 7 + 9a events/status + **Tier 11a read-your-writes**), `src/runtime/live-session/**` + `src/runtime/custom-tools/*` (live-session mode + worker custom tools), `src/schema/team-tool-schema.ts` (or any `Type.Unsafe({...})` schema definition), `src/extension/registration/team-tool.ts`, `workflows/*.workflow.md`, `.github/workflows/*.yml` (CI env — Tier 11e), `scripts/wc-gate.mjs` (Tier 11b), or before any commit touching these paths. Schema changes additionally require Tier 9 (feature battery) because the team tool's TypeBox schema is validated by pi-ai BEFORE the handler runs — a too-strict or malformed schema breaks every action silently. Surface changes additionally require Tier 10 (surface-mode battery) because surface is fail-closed: every failure degrades to headless and the run still goes green — only pane-level evidence proves the panes engaged.
63
+ **When to use**: after any change to `src/runtime/broker/*.ts` (broker + tokens + issuer), `src/ui/`, `src/config/` (incl. `src/config/migration-validator.ts`), `src/extension/registration/lifecycle-handlers.ts`, `src/runtime/child-pi/*.ts` (worker spawn/kill/steering), `src/runtime/surface/*.ts` (MuxSurface providers, degrade, launch script), `src/prompt/*.ts` (worker-side tools: ask / message / delegate / surface-worker recorder), `src/runtime/goal-workflow/plan-templates.ts`, `src/runtime/team-runner.ts` or `src/runtime/task-runner/**` (scheduler / execution — Tier 7 smoke), `src/state/**` (durable state — Tier 7 + 9a events/status + **Tier 11a read-your-writes**), `src/runtime/live-session/**` + `src/runtime/custom-tools/*` (live-session mode + worker custom tools), `src/schema/team-tool-schema.ts` (or any `Type.Unsafe({...})` schema definition), `src/extension/registration/team-tool.ts`, `workflows/*.workflow.md`, `.github/workflows/*.yml` (CI env — Tier 11e), `scripts/wc-gate.mjs` (Tier 11b), or before any commit touching these paths. Schema changes additionally require Tier 9 (feature battery) because the team tool's TypeBox schema is validated by pi-ai BEFORE the handler runs — a too-strict or malformed schema breaks every action silently. Surface changes additionally require Tier 10 (surface-mode battery) because surface is fail-closed: every failure degrades to headless and the run still goes green — only pane-level evidence proves the panes engaged. Resource `.md` changes (agent bodies/frontmatter, skill metadata, discovery, frontmatter parsing) additionally require **Tier 12** (resource-contract battery) because the agent/team/workflow frontmatter parser is line-based, not YAML — a folded scalar parses as `">"` for every consumer while all other tiers stay green.
49
64
 
50
65
  > **Path map (2026-08-26 reorg + A1)**: `src/runtime/crew-broker*.ts` → `src/runtime/broker/`; `src/runtime/child-pi*.ts` → `src/runtime/child-pi/`; `src/runtime/plan-templates.ts` (flat) → `src/runtime/goal-workflow/plan-templates.ts`; NEW dirs `src/runtime/surface/` and `src/prompt/`. Test files moved with them (`test/unit/crew-broker-*.test.ts` → `test/unit/runtime/broker/`, `test/unit/keybinding-map.parity.test.ts` → `test/unit/ui/`, ...).
51
66
 
67
+ > **2026-09-11 update (Batch-1..10, branch `fix/bundle-skill-resolution-and-skill-meta`, tip `aa899a1e`)**: builtin agents 17 → **18** (librarian, oracle, designer, 3 councillors, orchestrator); every agent carries flat routing metadata; NEW **Tier 12** (resource-contract battery) for `agents/*.md` / `skills/*/SKILL.md` / discovery / frontmatter changes; staleness gate gained a path-leak scan (ARCH-7); `scripts/release-smoke.mjs` gained a tarball import + peer-install gate (ARCH-6); two new operational quirks documented (broker SIGTERM on long silent bash; `wait-request-broker.test.ts` 180s-per-file load flake). Orientation doc: `CONTEXT.md`.
68
+
52
69
  ## Core principle: disk ≠ live Pi
53
70
 
54
71
  Two locations hold pi-crew state:
@@ -65,7 +82,7 @@ The 3-way resolution order for `dist/index.mjs` (per `index.ts:1-25`):
65
82
 
66
83
  > **Note on version pins**: this skill mentions specific versions (v0.9.17, v0.9.46, v0.9.47) as anchors for *when a behavior was introduced*, not as a constraint on which version the skill applies to. The verification discipline (Tiers 1–10) applies to every pi-crew release. Verify the version pin is still accurate via `git log --oneline -- index.ts` and `git log --oneline -- src/ui/run-dashboard.ts`.
67
84
 
68
- **Workflow files are runtime data** — `workflows/*.workflow.md` and task prompt strings inside `src/runtime/goal-workflow/plan-templates.ts` are loaded per-call, NOT bundled. Edits take effect immediately, no rebuild needed.
85
+ **Resource `.md` files are runtime data too** — `agents/*.md` and `skills/*/SKILL.md` load at RUN-CONSTRUCTION time from the package dir (they are NOT embedded in `dist/index.mjs`): edits take effect on the next team run / discovery call (discovery cache TTL ~30s — `invalidateAgentDiscoveryCache()` forces a fresh read), with NO bundle rebuild and NO Pi restart. `workflows/*.workflow.md` and task prompt strings inside `src/runtime/goal-workflow/plan-templates.ts` are the same: loaded per-call, NOT bundled. (Caveat: `src/` TypeScript that CONSUMES these files still follows the bundle rule below.)
69
86
 
70
87
  **The most common silent-failure mode**: edit `src/`, run `npm test` (pass!), rebuild bundle (good md5!), but the session still has the old code because Pi wasn't `/quit`-ed + reopened.
71
88
 
@@ -102,8 +119,10 @@ The skill maps to existing CI gates as follows:
102
119
  | `PI_CREW_BROKER=0 npm run test:critical` | Tier 2 (env kill switch path) | n/a — manual |
103
120
  | `npm run typecheck` | Tier 3 | `.github/workflows/*.yml` (every PR) |
104
121
  | `npm run check:wc-gate` | Tier 11b | **in BOTH `ci` and `ci:fast` scripts** (`package.json:71-72`) + explicit step in `.github/workflows/ci.yml:66-71` (since `09dda842` — was `ci:fast`-only, i.e. advisory) |
105
- | Bundle-staleness check | Tier 3 last step | `scripts/check-bundle-staleness.mjs`; `--committed-hash` mode = Tier 11j release gate |
106
- | Full `npm test` (= unit 819 files + integration 31) | n/a — too slow for in-loop | CI only; slow tier (3 files) is a SEPARATE glob `test:integration:slow` — only `npm run test:full` includes it |
122
+ | Bundle-staleness check (incl. **ARCH-7 path-leak scan** since `7d18508b` — line-scans `dist/index.mjs` + structural sourcemap check for tracked-source leaks) | Tier 3 last step | `scripts/check-bundle-staleness.mjs`; `--committed-hash` mode = Tier 11j release gate |
123
+ | `npm run test:bundle` (bundle import smoke, 2 tests) | Tier 3 post-build sanity | `test/unit/bundle-load.test.ts` |
124
+ | `node scripts/release-smoke.mjs` (manual, release cut) | Tier 3/11j companion | ARCH-6: installs pi-* peer deps, `import()`s the tarball-installed bundle (`:77`), shape-checks exports |
125
+ | Full `npm test` (= unit 823 files + integration 31) | n/a — too slow for in-loop | CI only; slow tier (3 files) is a SEPARATE glob `test:integration:slow` — only `npm run test:full` includes it |
107
126
  | `PI_CREW_SMOKE=1` env | Tier 11e | set ONLY in `weekly-smoke.yml` (auth-gated); nightly.yml deliberately does NOT (comment at `:24`) |
108
127
 
109
128
  To add Tier 1 to a pre-commit hook:
@@ -134,7 +153,7 @@ To add Tier 1 to CI as a fast-feedback gate (under 30s):
134
153
 
135
154
  **What**: run the curated 14-file fast subset.
136
155
 
137
- **Why this exists**: full `npm run test:unit` runs 819 files (was 642 at skill-writing time — it keeps growing), several minutes. Verifier worker response timeout would kill the worker mid-run → run = "hang". The fix (introduced in commit `1cb2dca`) splits out a `test:critical` subset covering exactly what changed in the broker/UI work.
156
+ **Why this exists**: full `npm run test:unit` runs 823 files (was 642 at skill-writing time — it keeps growing), several minutes. Verifier worker response timeout would kill the worker mid-run → run = "hang". The fix (introduced in commit `1cb2dca`) splits out a `test:critical` subset covering exactly what changed in the broker/UI work.
138
157
 
139
158
  **How**:
140
159
 
@@ -209,6 +228,7 @@ All three must show `# pass 101 # fail 0`. Measured times in this session (2026-
209
228
  npm run typecheck # ~20s, exits 0 with "strip-types import ok"
210
229
  npm run build:bundle # <1s, prints "[build-bundle] dist/index.mjs NNNN KB in NNN ms"
211
230
  md5sum dist/index.mjs
231
+ node scripts/check-bundle-staleness.mjs # ARCH-7: staleness + path-leak scan — exit 0
212
232
  ```
213
233
 
214
234
  Compare the printed md5 against what the user's Pi session loaded. If they differ → the session is running stale bundle.
@@ -222,7 +242,7 @@ Compare the printed md5 against what the user's Pi session loaded. If they diffe
222
242
  | Bundle builder | `scripts/build-bundle.mjs` (esbuild-based, bundles `index.bundle.ts` → `dist/index.mjs`) |
223
243
  | Bundle resolution rule | `index.ts:1-25` (entrypoint docstring); also `scripts/build-bundle.mjs:14-20` (entrypoint preference); **symlink is live for source files but the bundled `dist/index.mjs` is loaded** |
224
244
  | Postinstall hook | `scripts/postinstall.mjs:43` — best-effort bundle rebuild; falls back to strip-types if esbuild missing |
225
- | Bundle md5 anchors | `1cc4d55e18add7b9a036c569143320b6` (Phase-4 flip, ~2.78 MB) → `16e29d053bd370e24f40df147dadcb79` (v0.9.66, 2026-08-11) → `9b557ac106b82e1ee33d39dd0d6c7dd7` (post-MuxSurface-A1 main, 2026-08-27). **Always check current**: `md5sum dist/index.mjs` |
245
+ | Bundle md5 anchors | `1cc4d55e18add7b9a036c569143320b6` (Phase-4 flip, ~2.78 MB) → `16e29d053bd370e24f40df147dadcb79` (v0.9.66, 2026-08-11) → `9b557ac106b82e1ee33d39dd0d6c7dd7` (post-MuxSurface-A1 main, 2026-08-27) → `945720b1ad25673d86e263cdd834532f` (post-Batch-10 branch tip `aa899a1e`, 2026-09-11, ~3.30 MB). **Always check current**: `md5sum dist/index.mjs` |
226
246
 
227
247
  ---
228
248
 
@@ -232,7 +252,8 @@ Compare the printed md5 against what the user's Pi session loaded. If they diffe
232
252
 
233
253
  **The immediate-vs-rebuild rule** (which edits take effect without a rebuild):
234
254
  - `workflows/*.workflow.md` edits → **immediate**, no rebuild, no restart
235
- - `src/runtime/goal-workflow/plan-templates.ts` `taskTemplate` strings → **immediate**, runtime data
255
+ - `agents/*.md` + `skills/*/SKILL.md` edits → **immediate** — runtime data loaded from the package dir per discovery/run (cache TTL ~30s); NOT embedded in `dist/index.mjs`
256
+ - `src/runtime/goal-workflow/plan-templates.ts` → **needs rebuild** (correction of the pre-v0.9.17 claim above): it is `src/` TypeScript imported by the bundle, so its `taskTemplate` strings ship inside `dist/index.mjs` — edit → `build:bundle` → restart
236
257
  - Everything else (`src/` edits, `package.json`) → must `npm run build:bundle` THEN user `/quit` + reopen Pi
237
258
 
238
259
  **How to verify in this session**:
@@ -367,7 +388,7 @@ else:
367
388
 
368
389
  **What**: prove the verifier worker completes within `RESPONSE_TIMEOUT_MS` (**600s since the stuck-worker hardening — was 300s when this skill was distilled; `DEFAULT_CHILD_PI.responseTimeoutMs = 10 * 60_000`**).
369
390
 
370
- **Why this is its own tier**: `test:critical` covers unit-level invariants, but the verifier LLM is a separate failure mode — it reads the verifier prompt from `src/runtime/goal-workflow/plan-templates.ts:144, 147` (taskTemplate strings) or from `workflows/*.workflow.md` (workflow verifier sections), then decides which bash command to run. If the prompt says "Run tests" without specifying which, the LLM runs `npm test` (810+ files) and the worker gets killed by the response timeout with exit 143.
391
+ **Why this is its own tier**: `test:critical` covers unit-level invariants, but the verifier LLM is a separate failure mode — it reads the verifier prompt from `src/runtime/goal-workflow/plan-templates.ts:144, 147` (taskTemplate strings) or from `workflows/*.workflow.md` (workflow verifier sections), then decides which bash command to run. If the prompt says "Run tests" without specifying which, the LLM runs `npm test` (823 files) and the worker gets killed by the response timeout with exit 143.
371
392
 
372
393
  **How** (from parent Pi session — `team` is a tool, not a shell command):
373
394
 
@@ -409,6 +430,7 @@ The `team` tool is described in the agent's system prompt. Use `team action='sta
409
430
 
410
431
  1. **Verifier LLM runs `npm test`** (full unit + integration suite, >4 min) instead of `npm run test:critical`. Symptom: worker killed with exit 143 at the response timeout (300s historically — the measured runs below predate the bump to 600s). Fix: rewrite the verifier prompt to specify the exact fast command AND include "Do NOT run `npm test` or `npm run test:unit`".
411
432
  2. **Verifier LLM improvises** with a clean-cache `npm test` run anyway. The cache directive ("cache to `.crew/cache/`", "do NOT re-run") catches this — the second worker that observes a cached log should not re-run.
433
+ 3. **Broker SIGTERMs the worker mid long-silent-bash** (Batch-1 postmortem, `postmortem-batch-1-sigterm.md`): a worker running ONE >5–10 min command (full `npm test` ≈ 10 min) emits no LLM activity; the broker's responsiveness check kills it mid-run — exit 143 WHILE the command is still running, not a 600s response timeout. The transcript shows the command started and never returned. Work is usually intact (manual re-run was green); the kill is the "hang". Fix direction: split long suites into <5 min chunks or emit progress between commands. Tracked as `CONTEXT.md` Flagged #1.
412
434
 
413
435
  ---
414
436
 
@@ -619,7 +641,7 @@ grep -n "Log the event first" src/runtime/recovery/crash-recovery.ts # design
619
641
  # 3. buffered-site census — snapshot & audit:
620
642
  grep -rln "appendEventBuffered" src/ | wc -l # 16 files / ~70 raw matches (incl. imports+definition) at v0.10.5; audited live conversions = 43; EVERY new site needs the reader-audit
621
643
  # 4. the full gate — test:critical has NO stores/dwf/recovery coverage:
622
- npm run test:unit # 819 files, ~7500 tests, 15-18 min under load — MANDATORY after any delayed-write conversion program
644
+ npm run test:unit # 823 files, ~7500 tests, 15-18 min under load — MANDATORY after any delayed-write conversion program
623
645
  ```
624
646
 
625
647
  ### 11b. wc-gate enforcement (M4 done-gate)
@@ -706,6 +728,86 @@ node scripts/check-bundle-staleness.mjs --committed-hash # "OK: committed dist
706
728
 
707
729
  ---
708
730
 
731
+ ## Tier 12 — Resource-contract battery (agent .md + skill metadata)
732
+
733
+ **What**: prove agent frontmatter/bodies and skill metadata still parse and render after edits — through BOTH parsers and into the routing guidance the leader sees.
734
+
735
+ **Why this is its own tier**: `agents/*.md` are contracts — frontmatter grants tools and routing metadata (`useWhen`/`avoidWhen`/`cost`/`category`), the body IS the child's system prompt (`systemPromptMode: replace`), and the `## Output format` section is test-enforced. The agent/team/workflow frontmatter parser (`src/utils/frontmatter.ts`, `parseLines`) is **line-based, not YAML** — a folded scalar (`description: >`) parses as the literal string `">"` for EVERY consumer while typecheck/lint/test:critical all stay green (real regression: Batch 9, all 17 agents; fixed in `aa899a1e` by restoring single-line quoted values + teaching `parseLines` to strip symmetric quotes). Skills are exempt (they go through the real `yaml` package — folded scalars are FINE in `skills/*/SKILL.md`). Only a dual-parse probe catches this class.
736
+
737
+ **When required**: any change to `agents/*.md`, `skills/*/SKILL.md`, `src/agents/discover-agents.ts`, `src/skills/discover-skills.ts`, `src/utils/frontmatter.ts`, `src/runtime/skill-instructions.ts` (skill override resolution), or `src/extension/autonomous-policy.ts` (guidance render).
738
+
739
+ **Frontmatter contract rules** (agents/teams/workflows — the line-based parser):
740
+ - values stay **single-line**; a value containing `": "` MUST be wrapped in symmetric double quotes (`parseLines` strips them, `aa899a1e`)
741
+ - NEVER folded scalars (`key: >` / `key: |`) — they parse as `">"` / `"|"` (`CONTEXT.md` Flagged #4)
742
+ - routing keys are FLAT top-level CSV — `useWhen: "a, b"`, `avoidWhen: "…"`, `cost: cheap`, `category: orchestration` (parsed at `src/agents/discover-agents.ts:388-391`; a nested `routing:` block is silently ignored)
743
+
744
+ **How**:
745
+
746
+ ```bash
747
+ # 12a. Output contracts — every builtin agent must have '## Output format' + fenced block:
748
+ node --experimental-strip-types --no-warnings --test --test-force-exit test/unit/agents/agent-output-contracts.test.ts
749
+ # 1 test iterating ALL builtin agents (18 @ 2026-09-11)
750
+
751
+ # 12b. BOTH-parser proof (discovery + strict YAML) — ALWAYS invalidate the discovery cache first:
752
+ node --experimental-strip-types --no-warnings -e '
753
+ import("./src/agents/discover-agents.ts").then(mod => {
754
+ mod.invalidateAgentDiscoveryCache();
755
+ const list = mod.discoverAgents(process.cwd()).builtin;
756
+ const bad = list.filter(a => !a.description?.includes("When NOT to use:") || a.description.startsWith(String.fromCharCode(34)));
757
+ const noRoute = list.filter(a => !a.routing?.useWhen);
758
+ console.log("agents:", list.length, "| bad desc:", bad.length, "| no routing:", noRoute.length);
759
+ });'
760
+ # expect: agents: 18 | bad desc: 0 | no routing: 0
761
+ node -e '
762
+ const yaml=require("yaml"),fs=require("fs");let ok=0,fail=[];
763
+ for (const f of fs.readdirSync("agents")){
764
+ const m=/^---\r?\n([\s\S]*?)\r?\n---/.exec(fs.readFileSync("agents/"+f,"utf-8"));
765
+ if(!m)continue;
766
+ try{const p=yaml.parse(m[1]);if(p.name&&p.description)ok++;}catch{fail.push(f);}
767
+ }
768
+ console.log("strict YAML:",ok,"ok /",fail.length,"fail",fail.length?JSON.stringify(fail):"");'
769
+ # expect: strict YAML: 18 ok / 0 fail
770
+
771
+ # 12c. Routing guidance renders (leader-side):
772
+ node --experimental-strip-types --no-warnings -e '
773
+ import("./src/agents/discover-agents.ts").then(async da=>{
774
+ const pol=await import("./src/extension/autonomous-policy.ts");
775
+ da.invalidateAgentDiscoveryCache();
776
+ const g=pol.buildResourceRoutingGuidance(process.cwd(),40000);
777
+ const agentLines=g.split("\n").filter(l=>l.startsWith("- ")&&!/defaultWorkflow=|roles=|steps=/.test(l)&&/\((builtin|project|user)\):/.test(l));
778
+ const withRoute=agentLines.filter(l=>l.includes("useWhen="));
779
+ console.log("rendered agent lines:",agentLines.length,"| with useWhen:",withRoute.length,
780
+ "| orchestrator:",g.includes("- orchestrator ("),"| verifier:",g.includes("- verifier ("));
781
+ });'
782
+ # expect: every rendered AGENT line carries useWhen= (workflow lines legitimately lack it).
783
+ # The list is BUDGET-TRUNCATED BY DESIGN — at 40000 chars ~16/18 agents render; the tail
784
+ # (alphabetically last: verifier, writer) is cut first. Accept: with-route == agent-lines,
785
+ # newest agent (orchestrator) present, count >= 15. Do NOT assert 18/18 — truncation is correct.
786
+
787
+ # 12d. Fast unit batteries for the resource layer:
788
+ node --experimental-strip-types --no-warnings --test --test-force-exit \
789
+ test/unit/bundle-skill-resolution.test.ts \
790
+ test/unit/extension/registration/tool-loop-guard.test.ts \
791
+ test/unit/runtime/core/skill-instructions.test.ts
792
+ # packageRoot skill resolution + loop guard (12 tests) + skill override wildcard `*` / denylist `!name` (26 tests)
793
+ ```
794
+
795
+ **Acceptance**: 12a green; 12b BOTH parsers clean (18/18 descriptions with When-NOT, 0 quote leakage, 18/18 routing, 0 strict-YAML fails); 12c every rendered agent line carries `useWhen=` (budget-truncation is by design — see the note in 12c); 12d all pass. Agent/skill-only changes need NO bundle rebuild (runtime-loaded from the package dir) — but `src/` changes in the same commit still follow the Tier 3 bundle rule.
796
+
797
+ **References**:
798
+
799
+ | What | Where |
800
+ |---|---|
801
+ | Line-based parser + quote-strip | `src/utils/frontmatter.ts` (`parseLines`) — quote-strip added in `aa899a1e` |
802
+ | Flat routing keys parse | `src/agents/discover-agents.ts:388-391, 476` |
803
+ | Output-contract AC | `test/unit/agents/agent-output-contracts.test.ts` (Batch 9, `06c5d7ca`) |
804
+ | Skill override `*`/`!name` | `src/runtime/skill-instructions.ts` (`collectTaskSkillNames`, Batch 1+2 `c97bc578`) |
805
+ | Guidance builder | `src/extension/autonomous-policy.ts` (`buildResourceRoutingGuidance`) |
806
+ | Orchestrator (18th agent) | `agents/orchestrator.md` (Batch 10 `aa899a1e`) — process-only body; discovery guidance is the single routing authority |
807
+ | Quirk registry | `CONTEXT.md` — glossary + Flagged (#1 broker SIGTERM, #4 frontmatter parser) |
808
+
809
+ ---
810
+
709
811
  ## Anti-patterns (the cost is real, observed in this session)
710
812
 
711
813
  | Anti-pattern | Cost | Where fixed | Reference |
@@ -716,7 +818,7 @@ node scripts/check-bundle-staleness.mjs --committed-hash # "OK: committed dist
716
818
  | Test using real `loadConfig()` to mock config | Flaky when env / disk config changes | `612e18b` | `test/unit/runtime/broker/crew-broker-server-gate.test.ts:78` (use `brokerEnv: "0"` instead of `flagOn: false`) |
717
819
  | Source edit seen immediately | No, requires bundle rebuild + reload | n/a (permanent) | `index.ts:1-25` — bundle resolution rules |
718
820
  | Skip disabled-path proof | `effectiveEnabled()` regression slips through | n/a (permanent) | Tier 2 above |
719
- | `npm run test:unit` against the full suite (810 files now, 642 then) | several minutes; mis-judges verifier runtime | n/a (permanent) | Tier 1 above |
821
+ | `npm run test:unit` against the full suite (823 files now, 642 then) | several minutes; mis-judges verifier runtime | n/a (permanent) | Tier 1 above |
720
822
  | Skip typecheck | TS errors slip past `test:critical` (which uses `--test-timeout=30000`) | n/a (permanent) | Tier 3 above |
721
823
  | Run `pi` from a stale bundle | Session shows old behavior despite src/ edits | n/a (permanent) | `scripts/check-bundle-staleness.mjs` — CI gate |
722
824
  | Test by reading code | Proves nothing about runtime | n/a (permanent) | All tiers above |
@@ -740,6 +842,9 @@ node scripts/check-bundle-staleness.mjs --committed-hash # "OK: committed dist
740
842
  | **Fix finding của reviewer mà không tự verify** (deep review 2026-09-10): 1 trong 4 HIGH findings là false positive — "background-runner exit-loss" thực tế được cover bởi EL-2 `flushBufferedQueuesSync()` (sync lock + appendFileSync + fsync) trên `process.on("exit")` tại event-log.ts:1227. Fix theo finding mù quáng sẽ ĐÃ THÊM regression. | n/a (process) | Mọi finding trước khi fix: trace counter-evidence (exit handlers, sync flush paths). Finding = hypothesis, không phải fact. |
741
843
  | **Duplicated defaults map drift (G17-class)** (P0 remediation, `b6eba80f`): 2 bản EFFECTIVE_DEFAULTS (`settings-overlay.ts`, `handle-settings.ts`) hardcode `"aboveEditor"` trong khi nguồn chân lý (defaults.ts/install.mjs) nói `"bottom"` — suite không có test so 2 bản với nhau, drift sống sót qua 7500 tests. | `b6eba80f` | Defaults phải có MỘT nguồn chân lý, hoặc test so các bản sao. Live probe: `team-settings get <key>`. Xem Tier 11g. |
742
844
  | **Test vacuous — assert trên fixture chứ không trên wiring** (P1 remediation, `09dda842`): migration-validator test 2 từng assert key tự chế không có trong registry → luôn pass dù validator chưa được wire vào registerPiTeams. | `09dda842` | Test phải dùng key THẬT từ registry (`PI_CREW_BROKER_DIAG_UI` severity "removed"), và wiring test phải prove call-site (register.ts:68), không chỉ prove pure function. |
845
+ | **Folded YAML scalar (`key: >`) in agent/team/workflow frontmatter** (Batch-9 regression, fixed `aa899a1e`): `utils/frontmatter.ts` is line-based — folded descriptions parsed as literal `">"` for ALL 17 agents while typecheck/lint/test:critical stayed green (skills unaffected: real `yaml` package). Symptom: guidance renders `name (builtin): >`, When-NOT text missing. | `aa899a1e` | Agent/teams/workflows frontmatter values stay single-line; quote values containing `": "` (parser strips symmetric quotes); run the Tier 12b dual-parse probe after EVERY resource `.md` frontmatter edit. Folded scalars remain fine in `skills/*/SKILL.md` only. |
846
+ | **Worker killed mid long-silent-bash** (Batch-1 postmortem): one >5–10 min command (full `npm test` ≈ 10 min) emits no LLM activity → the broker's responsiveness check SIGTERMs the worker mid-run — exit 143 WHILE the command runs, not a 600s response timeout. Work was intact; manual re-run green — the kill WAS the "hang". | n/a (quirk — `CONTEXT.md` Flagged #1; P2 candidate) | Split long suites into <5 min chunks or emit progress between commands. On exit-143-mid-command: re-run manually BEFORE diagnosing a code bug. Postmortem: `postmortem-batch-1-sigterm.md` (workspace root). |
847
+ | **Treating `wait-request-broker.test.ts` load-timeout as a product bug**: the test runner's per-file 180s timeout is below this file's full-suite runtime under parallel load — fails only with the whole suite, passes in isolation. Pre-existing flake, NOT a regression from your change. | n/a (test infra) | Re-run the single file before fixing anything: `node scripts/test-runner.mjs test/unit/runtime/broker/wait-request-broker.test.ts`. Green in isolation = infra flake; move on. |
743
848
 
744
849
  ---
745
850
 
@@ -769,6 +874,9 @@ When a tier fails, the recovery is usually quick. Match the symptom to the cause
769
874
  | `delegate` rejects with a policy message | By design when depth cap hit (`maxDepth: 4`) or `nesting.enabled: false` in USER config (sensitive — project cannot flip) | Check depth in the rejection payload; `delegate.rejected` event in events.jsonl confirms the structured (non-silent) path |
770
875
  | herdr provider never engages | pi is not itself running inside a herdr pane (design: no socket guessing) | Run pi inside herdr, then `runtime.surface.mode` auto/`herdr`; verify `~/.config/herdr/herdr.sock` responds |
771
876
  | Surface run >5 phút bị stale-reconcile giết oan (worker khỏe, pane sống) | F1 (đã fix f12f4f5d + af2f8eb4): recorder chỉ flush ở turn boundary → lastSeen đóng băng giữa turn; reconciler cũ time-based không pid-gate. **Bẫy đa host**: MỘT pi session chạy bundle cũ cũng đủ giết run của session khác (sweep quét mọi runs) — tát cả host phải cùng version | Kiểm tra mọi pi process cùng bundle (`ps` lstart vs dist mtime); `PI_CREW_DEBUG_STALE=1` sidecar /tmp/pi-crew-f1-debug.log ghi mọi verdict STALE để bắt hung thủ; kỳ vọng sidecar rỗng khi mọi host đã fix |
877
+ | Worker exits 143 WHILE a long bash command is still running (no LLM-activity window before the kill) | Broker responsiveness SIGTERM on silent long commands (`CONTEXT.md` Flagged #1) — distinct from `RESPONSE_TIMEOUT_MS` (600s no-response) | Split the command; emit progress between steps; re-run the suite manually — work is usually intact. See `postmortem-batch-1-sigterm.md` |
878
+ | Full `test:unit` fails ONLY on `wait-request-broker.test.ts` under parallel load | Per-file 180s runner timeout vs the file's real runtime (passes isolated) | `node scripts/test-runner.mjs test/unit/runtime/broker/wait-request-broker.test.ts` — green in isolation = infra flake, not a regression |
879
+ | Guidance / `team action='list'` shows an agent description as `>` or missing When-NOT text | Folded-scalar frontmatter (`description: >`) — the line-based parser reads `>` literally (CONTEXT.md Flagged #4) | Restore the single-line value (double-quote it if it contains `": "`); re-run the Tier 12b dual-parse probe |
772
880
 
773
881
  ## Performance budget (per-tier soft limits)
774
882
 
@@ -862,6 +970,7 @@ The "skill stack" for a typical pi-crew change:
862
970
  6. tier 7 (smoke team) ← this skill, if plan/workflow change
863
971
  7. tier 9 (feature battery) ← this skill, if schema/tool-surface change
864
972
  8. tier 10 (surface battery) ← this skill, if surface/pane change
973
+ 8b. tier 12 (resource contracts) ← this skill, if agents/skills .md or discovery change
865
974
  9. commit + push
866
975
  10. verify-before-complete ← make the "done" claim with evidence
867
976
  ```
@@ -905,6 +1014,16 @@ Use this to answer "đủ full tính năng chưa?" without re-deriving. Every us
905
1014
  | ui.widgetPlacement default | `settings-overlay.ts:351`, `handle-settings.ts:43` | T11g + live team-settings get |
906
1015
  | Worktree twins contract | `src/worktree/worktree-manager.ts` | T11h (3× consecutive runs) |
907
1016
  | Bundle committed-hash gate | `scripts/check-bundle-staleness.mjs --committed-hash` | T11j |
1017
+ | Agent routing metadata (`useWhen`/`avoidWhen`/`cost`/`category`; 18 agents) | `agents/*.md` frontmatter; parsed `src/agents/discover-agents.ts:388-391` | T12b (dual parse) + T12c (guidance render) |
1018
+ | Agent output contracts (`## Output format` + fenced block, all builtins) | enforced by `test/unit/agents/agent-output-contracts.test.ts` | T12a |
1019
+ | Orchestrator agent (18th builtin, delegated orchestration) | `agents/orchestrator.md` | 9a `team action='list'` shows 18 + T12 |
1020
+ | Skill override wildcard/denylist (`*`, `!name`) | `src/runtime/skill-instructions.ts` (`collectTaskSkillNames`) | T12d (skill-instructions unit tests) |
1021
+ | Tool loop guard (read-only tools warn@3/block@5; ask wait-guard) | `src/extension/registration/tool-loop-guard.ts`; config `runtime.reliability.loopGuard` | `tool-loop-guard.test.ts` (12 tests, T12d) + live: same read-only tool 5× → structured block; exempt tools (team/Agent/…) unaffected |
1022
+ | Post-init skill check (SKILL.md presence/severity) | `src/extension/post-init-skill-check.ts`, wired `register.ts:132` | startup log probe: `[pi-crew] …` warn/error only when skills broken |
1023
+ | Detached-run delivery bound (3 attempts → drop + warn) | `src/runtime/detached-run-results.ts` (`MAX_DELIVERY_ATTEMPTS`) | unit tests |
1024
+ | Byte-stable worker prefix (ARCH-3) | `src/runtime/task-runner/prompt-builder.ts` stablePrefix/dynamicSuffix split | byte-identity unit test (strictEqual) |
1025
+ | Release tarball import gate (ARCH-6) | `scripts/release-smoke.mjs` (installs pi-* peers, `import()`s installed bundle `:77`, shape-checks exports) | release cut: `node scripts/release-smoke.mjs` |
1026
+ | Bundle path-leak scan (ARCH-7) | `scripts/check-bundle-staleness.mjs` (line-scan dist + structural sourcemap check) | T3 staleness run + T11j |
908
1027
 
909
1028
  ---
910
1029
 
@@ -923,7 +1042,11 @@ The skill mentions specific commits, line numbers, and version pins. As the code
923
1042
  | Verify herdr wire details | Each herdr release bump | `herdr api schema --json` vs `src/runtime/surface/herdr-provider.ts` (envelope/pane.read source/1-conn-per-request were verified on herdr 0.8.2) |
924
1043
  | Verify Tier 11 census numbers | Each `src/state/**` write-path commit | `grep -rln "appendEventBuffered" src/ \| wc -l` — update the 16-file / 43-conversion anchor in 11a when it drifts |
925
1044
  | Verify wc-gate still enforced | Each `package.json` / ci.yml edit | `node -e "require('./package.json').scripts.ci.includes('check:wc-gate')"` + grep ci.yml — a gate removed from `ci` reverts to advisory |
926
- | Verify migration-validator wiring | Each `register.ts` refactor | `grep -n validateEnv src/extension/register.ts` — must stay after `installChildProcessAbortShield`, before `startRuntimeWarmup`, warn-only |
1045
+ | Verify migration-validator wiring | Each `register.ts` refactor | `grep -n validateEnv src/extension/register.ts` — must stay after `installChildProcessAbortShield`, before `startRuntimeWarmup`, warn-only. ALSO `grep -n runPostInitSkillCheck src/extension/register.ts` (:132) — async post-init, warn/error only |
1046
+ | Verify builtin agent count + contracts | Each `agents/*.md` commit | Tier 12a/12b — **18 @ 2026-09-11** (`aa899a1e`); update this skill's count when it changes |
1047
+ | Verify frontmatter stays single-line/quoted | Each `agents/`, `teams/`, `workflows/` `.md` edit | Tier 12b dual-parse probe — BOTH discovery and strict `yaml` must pass |
1048
+ | Verify staleness leak-scan still runs | Each `check-bundle-staleness.mjs` edit | `node scripts/check-bundle-staleness.mjs` after `build:bundle` — exit 0 (staleness + path-leak) |
1049
+ | Verify release-smoke peer pins + import gate | Each `release-smoke.mjs` edit / release cut | `node scripts/release-smoke.mjs` — peer install + import + shape checks green |
927
1050
 
928
1051
  The skill does NOT need to be updated for every commit — only when the cited lines/files move. Consider it a "living reference" not a "live spec".
929
1052
 
@@ -981,6 +1104,13 @@ grep -rn '"ui.widgetPlacement"' src/ui/settings-overlay.ts src/extension/team-to
981
1104
  for i in 1 2 3; do node --experimental-strip-types --no-warnings --test test/unit/worktree/worktree-twins-contract.test.ts 2>&1 | grep -E '^# (pass|fail)'; done # 11h
982
1105
  grep -rln 'appendEventBuffered' src/ | wc -l # 11a: census (16 files @ v0.10.5)
983
1106
  node scripts/check-bundle-staleness.mjs --committed-hash # 11j: OK
1107
+ # Tier 12 (resource contracts — agents/skills .md + discovery changes)
1108
+ node --experimental-strip-types --no-warnings --test --test-force-exit test/unit/agents/agent-output-contracts.test.ts # 12a
1109
+ node --experimental-strip-types --no-warnings -e 'import("./src/agents/discover-agents.ts").then(m=>{m.invalidateAgentDiscoveryCache();const l=m.discoverAgents(process.cwd()).builtin;console.log("agents:",l.length,"| bad desc:",l.filter(a=>!a.description?.includes("When NOT to use:")).length,"| no routing:",l.filter(a=>!a.routing?.useWhen).length);})' # 12b: 18 | 0 | 0
1110
+ node -e 'const yaml=require("yaml"),fs=require("fs");let ok=0;for(const f of fs.readdirSync("agents")){const m=/^---\r?\n([\s\S]*?)\r?\n---/.exec(fs.readFileSync("agents/"+f,"utf-8"));if(m){try{if(yaml.parse(m[1]).name)ok++;}catch{}}}console.log("strict YAML:",ok)' # 12b: 18
1111
+ node --experimental-strip-types --no-warnings --test --test-force-exit test/unit/bundle-skill-resolution.test.ts test/unit/extension/registration/tool-loop-guard.test.ts test/unit/runtime/core/skill-instructions.test.ts # 12d
1112
+ node scripts/check-bundle-staleness.mjs # staleness + ARCH-7 path-leak scan (also after every build:bundle)
1113
+ node scripts/release-smoke.mjs # release cut: peer install + tarball import + shape check (ARCH-6)
984
1114
  # 11a full gate (after ANY delayed-write conversion program): npm run test:unit # ~7500 tests, 15-18 min
985
1115
  ```
986
1116
 
@@ -1001,6 +1131,7 @@ Before claiming "tested":
1001
1131
  - [ ] **Output report**: save `docs/real-test/reports/real-test-<YYYY-MM-DD>-<slug>.md` from `skills/real-test-pi-crew/REPORT-TEMPLATE.md`, filled DURING the run with per-tier evidence (counts/md5/runId) — not reconstructed from memory afterward. This is what makes past runs verifiable instead of trust-the-summary.
1002
1132
  - [ ] Tier 10: surface battery — **required if you touched `src/runtime/surface/**`, `src/prompt/surface-worker.ts`, the surface branch of `src/runtime/child-pi/child-pi.ts`, or the surface config keys**. 10a E2E 3/3 per backend available (tmux trong tmux; herdr ngoài tmux + socket sống — skip vì thiếu mux là correct-by-design nhưng KHÔNG tính pass cho backend đó); 10b live run với session ĐÃ reload bundle mới (xem Anti-patterns "file-md5 only") + `visibleAgents` set + pane-level evidence (pane id/title during run, `worker.surface_spawned`/`worker.surface_closed` events, pane auto-closed after — KHÔNG dùng `manifest.surface.panes` làm evidence engage, xem Anti-patterns "panes == {}"); 10c herdr live chỉ khi pi chạy trong herdr pane (skip kèm lý do nếu không).
1003
1133
  - [ ] Tier 11: remediation regression battery — **required if you touched `src/state/**` write paths, `migration-validator.ts`/its wiring, `scripts/wc-gate.mjs` or `ci` scripts, `.github/workflows/*` env, EFFECTIVE_DEFAULTS maps, or you are cutting a release**. Sub-checks a–j per Tier 11; 11a item 4 (full `test:unit`) mandatory after any delayed-write conversion program, skippable for doc-only changes. Record: buffered-site census count, wc-gate max, staleness `--committed-hash` result.
1134
+ - [ ] Tier 12: resource-contract battery — **required if you touched `agents/*.md`, `skills/*/SKILL.md`, `src/agents/discover-agents.ts`, `src/skills/discover-skills.ts`, `src/utils/frontmatter.ts`, `src/runtime/skill-instructions.ts`, or `src/extension/autonomous-policy.ts`**. 12a contracts green; 12b BOTH parsers clean (agent count — **18 @ 2026-09-11** — 0 bad descriptions, 0 missing routing, 0 strict-YAML fails); 12c every rendered agent line carries `useWhen=` (budget-truncated by design; newest agent visible); 12d unit batteries pass. Agent/skill-only changes need NO bundle rebuild (runtime-loaded from the package dir) — `src/` changes in the same commit still follow the Tier 3 bundle rule.
1004
1135
 
1005
1136
  **"All tiers pass" is a claim that needs per-row evidence.** Tier 9 means 9a **and** 9b **and** whichever of 9c–9f applies to the change — not "9a passed, therefore 9 passed". Tier 10 means pane-level evidence exists, not "run went green" (surface fail-closes to headless on every failure, so green proves nothing). If any required item above is unchecked or lacks concrete evidence (a number, an md5, a runId, a pane id), the answer to "is it tested?" is **no** — say so explicitly instead of rounding up to "pass".
1006
1137
 
@@ -1068,6 +1199,28 @@ Workflow files:
1068
1199
  - `workflows/plan-execute.workflow.md:30` — verifier prompt
1069
1200
  - `workflows/review.workflow.md:31` — verifier prompt
1070
1201
 
1202
+ Resource-contract files (Tier 12):
1203
+ - `src/utils/frontmatter.ts` — LINE-BASED parser (`parseLines`): single-line values, symmetric-quote strip (`aa899a1e`); folded scalars unsupported for agents/teams/workflows (skills use the real `yaml` package — folded OK there)
1204
+ - `src/agents/discover-agents.ts:388-391, 476` — flat routing keys (`useWhen`/`avoidWhen`/`cost`/`category` as top-level CSV); discovery cache TTL ~30s (`invalidateAgentDiscoveryCache()`)
1205
+ - `src/extension/autonomous-policy.ts` — `buildResourceRoutingGuidance` renders routing cards into the leader's injected policy (the single canonical routing source)
1206
+ - `src/runtime/skill-instructions.ts` — `collectTaskSkillNames`: `*` wildcard + `!name` denylist skill overrides
1207
+ - `src/extension/registration/tool-loop-guard.ts` — ARCH-1 loop guard: read-only tools warn@3/block@5, ask wait-guard warn@2/block@3rd, FIFO 512; exempts team/crew_agent/Agent/get_subagent_result; config `runtime.reliability.loopGuard`
1208
+ - `src/extension/post-init-skill-check.ts` (32L) — SKILL.md presence check; wired async at `register.ts:132`, warn/error log only
1209
+ - `src/runtime/detached-run-results.ts` — `MAX_DELIVERY_ATTEMPTS = 3`; drop + `detached-run-results.delivery-gave-up` log
1210
+ - `test/unit/agents/agent-output-contracts.test.ts` — output-contract AC across ALL builtin agents
1211
+ - `test/unit/bundle-skill-resolution.test.ts` + `test/unit/extension/registration/tool-loop-guard.test.ts` (12 tests) + `test/unit/runtime/core/skill-instructions.test.ts` (26 tests)
1212
+ - `scripts/release-smoke.mjs` — ARCH-6: installs pi-* peers, `import()`s the tarball-installed bundle (`:77`), shape-checks exports
1213
+ - `CONTEXT.md` — repo orientation: glossary + Flagged quirks (#1 broker SIGTERM, #2 wait-broker flake, #4 frontmatter parser)
1214
+
1215
+ Batch-1..10 wave (branch `fix/bundle-skill-resolution-and-skill-meta`, 2026-09-11, base v0.10.5):
1216
+ - `c97bc578` — BUG-1 packageRoot skill resolution + SKILL-HYGIENE-1 post-init check + SKILL-HYGIENE-2 `*`/`!name` + SKILL-META-1 (34 skills When-NOT — folded OK for skills)
1217
+ - `3de89a2f` / `07c5e014` / `24c63c75` / `b120187f` — skill Budget/Self-restraint + agent body upgrades (all roles; librarian/oracle/designer added)
1218
+ - `60e2cb96` — councillor agents (`inheritProjectContext: false`, deny-all-write toolset)
1219
+ - `d36ad4eb` — ARCH-1 tool loop guard + ARCH-3 byte-stable prefix
1220
+ - `7d18508b` — ARCH-2/5/6/7 (ARCH-4 skipped per ADR 2026-08-15 — live-session frozen)
1221
+ - `06c5d7ca` — PROMPT-1/2/5 (output-contract AC, task-rejection line, agent When-NOT — introduced the folded-scalar regression)
1222
+ - `aa899a1e` — Batch 10: routing metadata (18 agents), orchestrator, delivery bound, CONTEXT.md, folded-scalar fix + quote-strip
1223
+
1071
1224
  Commits (chronological, the patterns they introduced):
1072
1225
  - `1cb2dca` — `test:critical` script + plan-templates verifier fix
1073
1226
  - `d599578` — 4 workflow verifier prompt fixes
@@ -1,6 +1,9 @@
1
1
  ---
2
2
  name: requirements-to-task-packet
3
- description: "Use when a goal, issue, roadmap item, review finding, or user request must become actionable worker tasks."
3
+ description: >
4
+ Use when a goal, issue, roadmap item, review finding, or user request must become actionable worker tasks.
5
+ When NOT to use: already-decided work that just needs execution; tasks so simple they fit one sentence.
6
+
4
7
  origin: pi-crew
5
8
  triggers:
6
9
  - "convert requirements"
@@ -108,3 +111,9 @@ If ANY answer is NO → Stop. Complete task packet before dispatching.
108
111
  - Buried assumptions.
109
112
  - Expanding scope because context remains.
110
113
  - Treating tests as proof when the requirement was never asserted.
114
+
115
+ ## Self-restraint
116
+
117
+ "Creating nothing is a valid result." If the evidence does not support a meaningful change, say so explicitly rather than inventing one. The next attempt may find stronger evidence; an invented change now damages trust in every future report.
118
+
119
+ "Creating nothing" here means the requirements are already actionable as-is — no packet needed. Inventing scope or splitting trivial work into packets adds overhead without value.
@@ -1,6 +1,9 @@
1
1
  ---
2
2
  name: research
3
- description: Deep-research skill combining iterative depth, structured+validated output, rigor mechanisms, anti-thrash, and pi-native hooks for general deep research. REQUIRED — read the full skill file first (iterative-depth protocol with rigor scripts); run verify_citations and source_evaluator on your output before claiming done.
3
+ description: >
4
+ Deep-research skill combining iterative depth, structured+validated output, rigor mechanisms, anti-thrash, and pi-native hooks for general deep research. REQUIRED — read the full skill file first (iterative-depth protocol with rigor scripts); run verify_citations and source_evaluator on your output before claiming done.
5
+ When NOT to use: codebase-specific questions (use read-only-explorer); post-implementation audit (use iterative-audit).
6
+
4
7
  origin: local
5
8
  language: en
6
9
  distilled_against: 4-source-field-snapshot
@@ -1,6 +1,9 @@
1
1
  ---
2
2
  name: resource-discovery-config
3
- description: "pi-crew resource and configuration discovery workflow."
3
+ description: >
4
+ pi-crew resource and configuration discovery workflow.
5
+ When NOT to use: runtime state questions (use runtime-state-reader); model-specific config (use model-routing-context).
6
+
4
7
  origin: pi-crew
5
8
  triggers:
6
9
  - "discover agents"
@@ -58,3 +61,9 @@ node --experimental-strip-types --test test/unit/config-schema-validation.test.t
58
61
  npm test
59
62
  npm pack --dry-run
60
63
  ```
64
+
65
+ ## Self-restraint
66
+
67
+ "Creating nothing is a valid result." If the evidence does not support a meaningful change, say so explicitly rather than inventing one. The next attempt may find stronger evidence; an invented change now damages trust in every future report.
68
+
69
+ "Creating nothing" here means the current discovery already finds what consumers need. Registering resources nobody uses pollutes the discovery surface and slows every lookup.
@@ -1,6 +1,9 @@
1
1
  ---
2
2
  name: runtime-state-reader
3
- description: Safe read-only navigation of pi-crew run state.
3
+ description: >
4
+ Safe read-only navigation of pi-crew run state.
5
+ When NOT to use: write actions or mutations (use state-mutation-locking); configuration changes (use resource-discovery-config).
6
+
4
7
  origin: pi-crew
5
8
  triggers:
6
9
  - "inspect manifest"
@@ -1,6 +1,9 @@
1
1
  ---
2
2
  name: safe-bash
3
- description: "Safe shell-command workflow."
3
+ description: >
4
+ Safe shell-command workflow.
5
+ When NOT to use: subprocess orchestration (use child-pi-spawning); destructive operations without sandbox (use ownership-session-security).
6
+
4
7
  origin: pi-crew
5
8
  triggers:
6
9
  - "run this command"
@@ -1,6 +1,9 @@
1
1
  ---
2
2
  name: scrutinize
3
- description: "Outsider-perspective review questioning intent before tracing code."
3
+ description: >
4
+ Outsider-perspective review questioning intent before tracing code.
5
+ When NOT to use: implementation work (use systematic-debugging); trivial one-line fixes that don't warrant outsider review.
6
+
4
7
  origin: pi-crew
5
8
  triggers:
6
9
  - "scrutinize this"
@@ -84,3 +87,23 @@ If ANY answer is NO → Stop. Complete scrutiny requirements before reporting.
84
87
  - **One simpler-alternative pass is MANDATORY.** Skip only if user says "don't question scope."
85
88
  - **Distinguish claim from verification.** "The PR says X" and "I traced X and confirmed" are different.
86
89
  - **No flattery, no hedging.** State the finding.
90
+
91
+ ## Budget
92
+
93
+ This skill applies a 3-attempt budget: 1 initial + max 2 re-attempts.
94
+
95
+ Stamp every invocation:
96
+
97
+ ```
98
+ attempt X of 3 (Y attempts remaining)
99
+ ```
100
+
101
+ An attempt is one full outsider-perspective review pass. Re-attempt when the review uncovers intent-level questions that change the approach.
102
+
103
+ Re-attempts only when the previous attempt materially changes the decision or risk. Do NOT spend a re-attempt on mechanical changes or already-resolved findings. When exhausted, escalate to the user with options (accept risk / change scope / exceptional budget).
104
+
105
+ ## Self-restraint
106
+
107
+ "Creating nothing is a valid result." If the evidence does not support a meaningful change, say so explicitly rather than inventing one. The next attempt may find stronger evidence; an invented change now damages trust in every future report.
108
+
109
+ "Creating nothing" here means concluding the intent was sound and the approach justified — no findings needed. An invented finding to justify the review round damages trust in every future review.
@@ -1,6 +1,9 @@
1
1
  ---
2
2
  name: secure-agent-orchestration-review
3
- description: "Use when reviewing delegation, skill loading, tool access, worker prompts, artifacts, runtime config, state, ownership, or subprocess execution."
3
+ description: >
4
+ Use when reviewing delegation, skill loading, tool access, worker prompts, artifacts, runtime config, state, ownership, or subprocess execution.
5
+ When NOT to use: pure code quality issues without security implications (use multi-perspective-review); single-file fixes (use scrutinize).
6
+
4
7
  origin: pi-crew
5
8
  triggers:
6
9
  - "review delegation"
@@ -1,6 +1,9 @@
1
1
  ---
2
2
  name: state-mutation-locking
3
- description: "Durable state mutation and locking workflow."
3
+ description: >
4
+ Durable state mutation and locking workflow.
5
+ When NOT to use: read-only inspection (use runtime-state-reader); single-process non-concurrent edits.
6
+
4
7
  origin: pi-crew
5
8
  triggers:
6
9
  - "modify manifest"
@@ -1,6 +1,9 @@
1
1
  ---
2
2
  name: systematic-debugging
3
- description: "Four-phase debugging discipline with refuse gates."
3
+ description: >
4
+ Four-phase debugging discipline with refuse gates.
5
+ When NOT to use: trivial fixes where root cause is obvious; production incidents needing immediate rollback (use post-mortem after).
6
+
4
7
  origin: pi-crew
5
8
  triggers:
6
9
  - "debug this"
@@ -1,6 +1,9 @@
1
1
  ---
2
2
  name: verification-before-done
3
- description: "Evidence before claims."
3
+ description: >
4
+ Evidence before claims.
5
+ When NOT to use: mid-task self-correction (use scrutinize); pure execution tasks where output IS evidence.
6
+
4
7
  origin: pi-crew
5
8
  triggers:
6
9
  - "done"
@@ -80,3 +83,17 @@ Stop before saying done if you are using words like "should", "probably", "looks
80
83
  - **Don't** use fuzzy language like "seems", "probably", "looks like"
81
84
  - **Don't** skip providing verification commands for claims
82
85
  - **Don't** claim done if you're still using hypotheses instead of evidence
86
+
87
+ ## Budget
88
+
89
+ This skill applies a 3-attempt budget: 1 initial + max 2 re-attempts.
90
+
91
+ Stamp every invocation:
92
+
93
+ ```
94
+ attempt X of 3 (Y attempts remaining)
95
+ ```
96
+
97
+ An attempt is one fresh verification run (identify command → run → read output → compare to claim). Re-attempt when output contradicts the claim or reveals a new failure mode.
98
+
99
+ Re-attempts only when the previous attempt materially changes the decision or risk. Do NOT spend a re-attempt on mechanical changes or already-resolved findings. When exhausted, escalate to the user with options (accept risk / change scope / exceptional budget).
@@ -1,6 +1,9 @@
1
1
  ---
2
2
  name: widget-rendering
3
- description: "Pi TUI crew widget data sources, display priority, and rendering performance."
3
+ description: >
4
+ Pi TUI crew widget data sources, display priority, and rendering performance.
5
+ When NOT to use: non-TUI displays; backend data sources without widget context (use runtime-state-reader).
6
+
4
7
  origin: pi-crew
5
8
  triggers:
6
9
  - "empty agent"
@@ -1,6 +1,9 @@
1
1
  ---
2
2
  name: workspace-isolation
3
- description: "Workspace isolation boundaries."
3
+ description: >
4
+ Workspace isolation boundaries.
5
+ When NOT to use: concurrent edits within same repo (use worktree-isolation); non-git projects.
6
+
4
7
  origin: pi-crew
5
8
  triggers:
6
9
  - "workspace isolation"
@@ -1,6 +1,9 @@
1
1
  ---
2
2
  name: worktree-isolation
3
- description: "Conflict-safe git worktree workflow."
3
+ description: >
4
+ Conflict-safe git worktree workflow.
5
+ When NOT to use: non-git changes (use direct edit); single-commit patches (use git-master).
6
+
4
7
  origin: pi-crew
5
8
  triggers:
6
9
  - "create worktree"
@@ -640,6 +640,7 @@ function parseReliabilityConfig(value: unknown): CrewReliabilityConfig | undefin
640
640
  forcePreflight: parseWithSchema(Type.Boolean(), obj.forcePreflight),
641
641
  ambientStatusInjection: parseWithSchema(Type.Boolean(), obj.ambientStatusInjection),
642
642
  perWriteValidation: parseWithSchema(Type.Boolean(), obj.perWriteValidation),
643
+ loopGuard: parseWithSchema(Type.Boolean(), obj.loopGuard),
643
644
  scopeModels: parseWithSchema(Type.Boolean(), obj.scopeModels),
644
645
  };
645
646
  return Object.values(reliability).some((entry) => entry !== undefined) ? reliability : undefined;
@@ -274,6 +274,14 @@ export interface CrewReliabilityConfig {
274
274
  * Set to `false` to disable.
275
275
  */
276
276
  perWriteValidation?: boolean;
277
+ /**
278
+ * Tool loop guard (ARCH-1). Warns at 3 consecutive identical tool results
279
+ * (identical args + byte-identical output) and hard-blocks read-only file
280
+ * tools (read/grep/glob/find/ls) at 5 — the model-side infinite-loop
281
+ * failure mode. `ask` repeats within a turn are warned at 2 and the 3rd
282
+ * call refused. Default: true (opt-out). Set to `false` to disable.
283
+ */
284
+ loopGuard?: boolean;
277
285
  /**
278
286
  * Opt-in model scope enforcement (F7). When true, subagent model choices
279
287
  * that fall outside the user's pi `enabledModels` allowlist are flagged:
package/src/errors.ts CHANGED
@@ -64,7 +64,7 @@ const DEFAULT_HELP: Record<ErrorCode, string | undefined> = {
64
64
  [ErrorCode.RunStale]:
65
65
  "The worker stopped heartbeating and was treated as a zombie. Re-run the team (resume or fresh); if it recurs, check `runtime.executeWorkers` / system load.",
66
66
  [ErrorCode.ModelOutOfScope]:
67
- "The requested model is not in your pi `enabledModels` allowlist. Either pick a model listed in `enabledModels` (settings.json) or extend the allowlist. The scope gate is opt-in — disable `runtime.reliability.scopeModels` to allow any model.",
67
+ "The requested model is not in your pi `enabledModels` allowlist. Either pick a model listed in `enabledModels` (settings.json) or extend the allowlist. The scope gate is opt-in — disable `reliability.scopeModels` to allow any model.",
68
68
  };
69
69
 
70
70
  /**