@azure-id/orc 2.0.2 → 2.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (129) hide show
  1. package/CHANGELOG.md +455 -0
  2. package/README-id.md +52 -23
  3. package/README.md +36 -20
  4. package/bin/build-agents.js +205 -206
  5. package/bin/clear-logs.js +620 -0
  6. package/bin/cli.js +47417 -46420
  7. package/bin/fix.js +459 -0
  8. package/bin/gotcha.js +288 -8
  9. package/bin/habit.js +1538 -1453
  10. package/bin/mockrun-catalog.js +3 -0
  11. package/bin/onboarding-content.js +2 -1
  12. package/bin/pricing.json +28 -21
  13. package/bin/trace-write.js +869 -657
  14. package/bin/verify-contracts.js +5552 -5471
  15. package/bin/verify-package.js +716 -705
  16. package/bin/webui/api.js +125 -6
  17. package/bin/webui/css/panels/hookui.css +79 -0
  18. package/bin/webui/fixtures/behaviour.js +30 -0
  19. package/bin/webui/fixtures/extra.js +2036 -2036
  20. package/bin/webui/fixtures/flow.js +81 -81
  21. package/bin/webui/fixtures/hookui.js +506 -431
  22. package/bin/webui/fixtures/index.js +32 -3
  23. package/bin/webui/fixtures/knowledge.js +2 -2
  24. package/bin/webui/fixtures/maintenance.js +123 -42
  25. package/bin/webui/fixtures/rules.js +1007 -1002
  26. package/bin/webui/fixtures/settings.js +6 -7
  27. package/bin/webui/fixtures/stats.js +9 -6
  28. package/bin/webui/fixtures/wait.js +98 -97
  29. package/bin/webui/i18n/en/behaviour.json +13 -0
  30. package/bin/webui/i18n/en/hookui.json +290 -194
  31. package/bin/webui/i18n/en/maintenance.json +71 -52
  32. package/bin/webui/i18n/en/settings.json +60 -60
  33. package/bin/webui/i18n/id/behaviour.json +13 -0
  34. package/bin/webui/i18n/id/hookui.json +290 -194
  35. package/bin/webui/i18n/id/maintenance.json +71 -52
  36. package/bin/webui/i18n/id/settings.json +60 -60
  37. package/bin/webui/js/05-banners.js +181 -172
  38. package/bin/webui/js/06-edit.js +193 -183
  39. package/bin/webui/js/panels/behaviour.js +77 -12
  40. package/bin/webui/js/panels/hookui.js +2196 -1691
  41. package/bin/webui/js/panels/maintenance.js +336 -235
  42. package/bin/webui/js/panels/settings.js +726 -719
  43. package/mock-run/INDEX.md +1 -0
  44. package/mock-run/a-normal-day.md +2 -2
  45. package/mock-run/extra-slots.md +177 -177
  46. package/mock-run/habits.md +1 -1
  47. package/mock-run/orc-aftermath.md +392 -392
  48. package/mock-run/orc-budget.md +29 -33
  49. package/mock-run/orc-cli.md +200 -200
  50. package/mock-run/orc-explain.md +86 -86
  51. package/mock-run/orc-extra.md +389 -392
  52. package/mock-run/orc-fast.md +2 -2
  53. package/mock-run/orc-fix.md +87 -0
  54. package/mock-run/orc-pattern.md +112 -112
  55. package/mock-run/orc-quick.md +146 -146
  56. package/mock-run/orc.md +4 -4
  57. package/package.json +1 -1
  58. package/templates/agents/MODEL-MAPPING.md +176 -173
  59. package/templates/agents/orc-analyze-mini-sonnet-5-high.md +2 -2
  60. package/templates/agents/{orc-claude-writer-opus-5-med.md → orc-claude-writer-opus-5-low.md} +4 -4
  61. package/templates/agents/orc-executor-haiku-4-5.md +1 -1
  62. package/templates/agents/orc-executor-opus-5-low.md +1 -1
  63. package/templates/agents/orc-executor-sonnet-4-6-high.md +1 -1
  64. package/templates/agents/orc-executor-sonnet-4-6-med.md +1 -1
  65. package/templates/agents/orc-executor-sonnet-5-high.md +2 -2
  66. package/templates/agents/orc-executor-sonnet-5-low.md +155 -0
  67. package/templates/agents/orc-executor-sonnet-5-med.md +155 -0
  68. package/templates/agents/orc-pattern-codifier-sonnet-5-high.md +2 -2
  69. package/templates/agents/orc-planner-mini-sonnet-5-high.md +2 -2
  70. package/templates/agents/{orc-recon-sonnet-4-6-med.md → orc-recon-sonnet-5-med.md} +4 -4
  71. package/templates/agents/orc-retro-sonnet-5-high.md +2 -2
  72. package/templates/agents/{orc-reviewer-opus-5-med.md → orc-reviewer-opus-5-low.md} +4 -4
  73. package/templates/agents/{orc-wiki-scanner-opus-5-med.md → orc-wiki-scanner-opus-5-low.md} +3 -3
  74. package/templates/agents/orc-wiki-scanner-sonnet-5-high.md +2 -2
  75. package/templates/commands/orc-fast.md +10 -10
  76. package/templates/commands/orc-fix.md +18 -0
  77. package/templates/commands/orc-wiki.md +64 -64
  78. package/templates/hooks/README.md +75 -7
  79. package/templates/hooks/orc-session-hook.js +78 -3
  80. package/templates/hooks/orc-statusline-render.js +317 -23
  81. package/templates/hooks/orc-statusline.js +162 -49
  82. package/templates/hooks/orc-subagent-line.js +23 -5
  83. package/templates/hooks/orc-weather-fetch.js +123 -0
  84. package/templates/skills/_shared/extra-dispatch.md +1331 -1331
  85. package/templates/skills/_shared/gotchas.md +27 -7
  86. package/templates/skills/_shared/habits.md +2 -2
  87. package/templates/skills/_shared/opus5-only.md +137 -137
  88. package/templates/skills/_shared/phases/review.md +2 -2
  89. package/templates/skills/_shared/phases/trace-verbs.md +17 -3
  90. package/templates/skills/_shared/phases/trace.md +136 -136
  91. package/templates/skills/_shared/review-slice.md +28 -1
  92. package/templates/skills/_shared/wait.md +1 -0
  93. package/templates/skills/orc/README.md +2 -2
  94. package/templates/skills/orc/SKILL.md +227 -227
  95. package/templates/skills/orc/config.md +25 -27
  96. package/templates/skills/orc/examples/full-run-mock.md +74 -73
  97. package/templates/skills/orc/references/effort-and-mode.md +222 -222
  98. package/templates/skills/orc/references/preflight-report.md +3 -3
  99. package/templates/skills/orc/references/ultra-mode.md +1 -1
  100. package/templates/skills/orc/schemas/checkpoint.md +122 -122
  101. package/templates/skills/orc/subskills/orc-review-verify/SKILL.md +3 -3
  102. package/templates/skills/orc/subskills/orc-review-verify/core.md +1 -1
  103. package/templates/skills/orc-analyze/SKILL.md +4 -2
  104. package/templates/skills/orc-claude/SKILL.md +14 -14
  105. package/templates/skills/orc-claude/examples/claude-run-mock.md +65 -65
  106. package/templates/skills/orc-diy/SKILL.md +5 -5
  107. package/templates/skills/orc-fast/SKILL.md +210 -208
  108. package/templates/skills/orc-fix/SKILL.md +110 -0
  109. package/templates/skills/orc-grill/SKILL.md +232 -232
  110. package/templates/skills/orc-grill/references/grill-doc.md +1 -1
  111. package/templates/skills/orc-mini/SKILL.md +4 -4
  112. package/templates/skills/orc-mini/examples/mini-run-mock.md +4 -2
  113. package/templates/skills/orc-quick/README.md +536 -536
  114. package/templates/skills/orc-quick/SKILL.md +2 -2
  115. package/templates/skills/orc-quick/references/context-doc.md +145 -145
  116. package/templates/skills/orc-quick/references/dispatch-gate.md +220 -219
  117. package/templates/skills/orc-quick/references/gh-mode.md +4 -2
  118. package/templates/skills/orc-retro/SKILL.md +1 -1
  119. package/templates/skills/orc-retro/examples/retro-mock.md +16 -16
  120. package/templates/skills/orc-wiki/README.md +325 -325
  121. package/templates/skills/orc-wiki/SKILL.md +195 -195
  122. package/templates/skills/orc-wiki/examples/wiki-run-mock.md +37 -37
  123. package/templates/skills/orc-wiki/references/extra.md +2 -2
  124. package/templates/skills/orc-wiki/references/partial-refresh.md +276 -271
  125. package/templates/skills/orc-wiki/references/phases/phase-0.md +1 -1
  126. package/templates/skills/orc-wiki/references/phases/phase-2.md +3 -3
  127. package/templates/skills/orc-wiki/references/phases/phase-3c.md +3 -2
  128. package/templates/agents/orc-claude-writer-opus-4-8-high.md +0 -48
  129. package/templates/agents/orc-wiki-scanner-opus-4-8-high.md +0 -74
@@ -1,73 +1,74 @@
1
- # Example — Annotated Full Run (v2.2, dry run)
2
-
3
- O: = orchestrator (Opus 4.8 high, always) · U: = user. Nothing real. The
4
- orchestrator NEVER implements — it always spawns scored subagents.
5
-
6
- U: Orchestrate: merchant notification system — in-app, email digest, prefs page.
7
-
8
- ## Phase 0 — Intake
9
- O: Creates run folder run/merchant-notifications/ FIRST.
10
- [reads repo: TS/yarn/jest — asks nothing about stack] Rough size: medium+ → 4-question tier (one batched round):
11
- 1) What should exist when done? 2) Out of scope? 3) How do we know it's done? 4) Patterns/files to avoid?
12
- U: [answers] no push, no admin; bell count + both digest schedules + prefs persist; follow services/ pattern, don't touch auth_service.
13
- O: [drafts run/merchant-notifications/intent-spec.md] Sign-off: gate or soft? (default gate)
14
- U: gate → approved
15
-
16
- ## Phase 1 — Planning
17
- O: Planner? U: OpenSpec
18
- O: [6 tasks with declared_files incl. tests + per-file grounding attestations + sliced acceptance[], depends_on, spec_ref] ⚠ T1/T2 share models/notification.ts → scheduling separates them.
19
- Grounding spot-check (Phase 1 exit): globs all `exists` paths — 13/14 present, models/notifcation.ts MISSING (typo) → plan bounced to planner with the miss list → corrected to models/notification.ts ✓.
20
-
21
- ## Phase 2 — Effort, dispatch style, scoring
22
- O: Refined: medium, partially dependent. Recommend PARALLEL waves. Confirm? U: yes. Batch pause? U: every 2 waves.
23
- O: Facet scoring (planner facets → formula → band; facet gate: breadth + fan recomputed, risk cites checked):
24
- T1 3·new-surface·stateful·new-tests·fan0/3·low = 6+18+16+8+9 = 57 → sonnet-5-high (keystone — every later task imports its enum)
25
- T2 3·imitate·branching·new-tests·fan1/0·low = 6+8+8+8+5 = 35 → sonnet-4-6-med
26
- T3 3·imitate·branching·update-existing·fan1/0·low = 6+8+8+4+5 = 31 → sonnet-4-6-med
27
- T4 2·imitate·branching·update-existing·fan1/0·low = 6+8+8+4+5 = 31 → sonnet-4-6-med
28
- T5 1·mechanical·none·update-existing·fan0/0·low = 2+0+0+4 = 6 → haiku-4-5
29
- T6 1·mechanical·none·none·fan0/0·low = 2+0+0+0 = 2 → haiku-4-5
30
- Plan: 6 tasks, 3 waves W1{T1} W2{T2,T3,T4} W3{T5,T6}, all spawned. Anticipated escalations? U: none, go.
31
-
32
- ## Phase 3 — Execution (ALL spawned; orchestrator coordinates only)
33
- O: [checkpoint + state-of-play written into run/merchant-notifications/ BEFORE dispatch]
34
- ▶ W1: spawn Agent-A(T1, sonnet-5-high) — milestones 40%→100% — return validated ✓ (evidence: `yarn jest models/` exit 0, tail quoted; unmet[] empty) — log: DECISION type enum.
35
- ▶ W2: spawn 3 agents (sonnet-4-6-med each).
36
- Agent-C needs_context (1/2): "needs T1's enum interface" → O adjudicates in-scope → re-slices → resumes. (User never bothered.)
37
- Agent-C user escalation: "digest send time?" → U: 07:00 fixed → ANSWER broadcast to log.
38
- Returns validated ✓ · collision audit clean.
39
- ⏸ BATCH PAUSE (stop sequence):
40
- checkpoint ✓ → state-of-play ✓ → dispatch report shown (models/scores) → "Run /usage to see your remaining limits."
41
- → resume block:
42
- | Continue ORC run `merchant-notifications`.
43
- | Read run/merchant-notifications/state-of-play.md, then run/merchant-notifications/checkpoint.json.
44
- | Resume from checkpoint phase/wave. Intent-spec approved — do not re-plan. Do not redo done tasks.
45
- "Reply continue here, or paste in a fresh session (recommended if long)."
46
-
47
- U: [next day, FRESH session, pastes block]
48
- O: [state-of-play → checkpoint → re-anchored] "Resuming: merchant notifications (in-app+digest+prefs). Still correct?" U: yes.
49
- ▶ W3: spawn T5+T6 (haiku-4-5) → done. All 6 complete.
50
-
51
- ## Phase 5 — Review (spawned worker, Opus 5.5 medium)
52
- O: Code pattern? U: none. Findings (P0–P3 ladder, each P0–P2 anchored file:line + verbatim quote):
53
- P1 T2 no pagination — services/notification_list.ts:88 — O spot-checks: Reads line 88, quote matches ✓ → asks user, U: fix it
54
- · P3 email_digest naming ×2 · 12 tests added. (An unanchored finding would be auto-P3 — never fixed unasked.)
55
-
56
- ## Phase 6 — Verify (spawned worker, Opus 5.5 medium; criteria = definition-of-done)
57
- criteria[] with per-line evidence: bell ✓ (test name) · both schedules ✗ (weekly cron missing = P0, unmet criterion, evidence: no cron entry in scheduler.ts) · prefs ✓ → FAILED → quote spot-check ✓ → P0 auto-fix (single attempt, no ask) → re-verify → PASSED. 14/14.
58
-
59
- ## Phase 7 — Summary
60
- O: 6 tasks/3 waves/6 spawns/1 pause/1 fresh-session resume · 1 needs_context · 1 escalation · verify fixed_then_passed · branch green. P2 fix-batch: none · Apply the 2 P3 cosmetics? U: no.
61
-
62
- ## Phase 8 — Ship (orc-pr, reads pr.md)
63
- O: Branch: feat/notifications. Commit? Push? PR? U: yes all.
64
- O: Ticket/title/target? U: DRP-482 / Add merchant notification system / develop
65
- O: ✓ committed → pushed → run/merchant-notifications/DRP-482-add-merchant-notification-system.md from pr.md (base develop).
66
- Log deleted · checkpoint + dispatch log KEPT · completion dispatch report shown → "Run /usage to see your remaining limits."
67
-
68
- ## What this exercises
69
- always-spawn (even cheap tasks get a subagent) · facet-scored arithmetic · 6-band ladder ·
70
- per-run folder run/merchant-notifications/ · usage reminder (no programmatic /usage) ·
71
- planner grounding spot-check catching a hallucinated path · executor evidence (verbatim
72
- build/test proof) · reviewer quote spot-check before any P0/P1 action ·
73
- definition-of-done catching the weekly-cron gap · needs_context resolved internally · fresh-session resume.
1
+ # Example — Annotated Full Run (v2.2, dry run)
2
+
3
+ O: = orchestrator (Opus 4.8 high, always) · U: = user. Nothing real. The
4
+ orchestrator NEVER implements — it always spawns scored subagents.
5
+
6
+ U: Orchestrate: merchant notification system — in-app, email digest, prefs page.
7
+
8
+ ## Phase 0 — Intake
9
+ O: Creates run folder run/merchant-notifications/ FIRST.
10
+ [reads repo: TS/yarn/jest — asks nothing about stack] Rough size: medium+ → 4-question tier (one batched round):
11
+ 1) What should exist when done? 2) Out of scope? 3) How do we know it's done? 4) Patterns/files to avoid?
12
+ U: [answers] no push, no admin; bell count + both digest schedules + prefs persist; follow services/ pattern, don't touch auth_service.
13
+ O: [drafts run/merchant-notifications/intent-spec.md] Sign-off: gate or soft? (default gate)
14
+ U: gate → approved
15
+
16
+ ## Phase 1 — Planning
17
+ O: Planner? U: OpenSpec
18
+ O: [6 tasks with declared_files incl. tests + per-file grounding attestations + sliced acceptance[], depends_on, spec_ref] ⚠ T1/T2 share models/notification.ts → scheduling separates them.
19
+ Grounding spot-check (Phase 1 exit): globs all `exists` paths — 13/14 present, models/notifcation.ts MISSING (typo) → plan bounced to planner with the miss list → corrected to models/notification.ts ✓.
20
+
21
+ ## Phase 2 — Effort, dispatch style, scoring
22
+ O: Refined: medium, partially dependent. Recommend PARALLEL waves. Confirm? U: yes. Batch pause? U: every 2 waves.
23
+ O: Facet scoring (planner facets → formula → band; facet gate: breadth + fan recomputed, risk cites checked):
24
+ T1 3·new-surface·stateful·new-tests·fan0/3·low = 6+18+16+8+9 = 57 → sonnet-5-high (keystone — every later task imports its enum)
25
+ T2 3·imitate·branching·new-tests·fan1/0·low = 6+8+8+8+5 = 35 → sonnet-5-high
26
+ T3 3·imitate·branching·update-existing·fan1/0·low = 6+8+8+4+5 = 31 → sonnet-5-high
27
+ T4 2·imitate·branching·update-existing·fan1/0·low = 6+8+8+4+5 = 31 → sonnet-5-high
28
+ T5 1·mechanical·none·update-existing·fan0/0·low = 2+0+0+4 = 6 → haiku-4-5
29
+ T6 1·mechanical·none·none·fan0/0·low = 2+0+0+0 = 2 → haiku-4-5
30
+ Plan: 6 tasks, 3 waves W1{T1} W2{T2,T3,T4} W3{T5,T6}, all spawned. Anticipated escalations? U: none, go.
31
+
32
+ ## Phase 3 — Execution (ALL spawned; orchestrator coordinates only)
33
+ O: [checkpoint + state-of-play written into run/merchant-notifications/ BEFORE dispatch]
34
+ ▶ W1: spawn Agent-A(T1, sonnet-5-high) — milestones 40%→100% — return validated ✓ (evidence: `yarn jest models/` exit 0, tail quoted; unmet[] empty) — log: DECISION type enum.
35
+ ▶ W2: spawn 3 agents (sonnet-5-high each).
36
+ Agent-C needs_context (1/2): "needs T1's enum interface" → O adjudicates in-scope → re-slices → resumes. (User never bothered.)
37
+ Agent-C user escalation: "digest send time?" → U: 07:00 fixed → ANSWER broadcast to log.
38
+ Returns validated ✓ · collision audit clean.
39
+ ⏸ BATCH PAUSE (stop sequence):
40
+ checkpoint ✓ → state-of-play ✓ → dispatch report shown (models/scores) → "Run /usage to see your remaining limits."
41
+ → resume block:
42
+ | Continue ORC run `merchant-notifications`.
43
+ | Read run/merchant-notifications/state-of-play.md, then run/merchant-notifications/checkpoint.json.
44
+ | Resume from checkpoint phase/wave. Intent-spec approved — do not re-plan. Do not redo done tasks.
45
+ "Reply continue here, or paste in a fresh session (recommended if long)."
46
+
47
+ U: [next day, FRESH session, pastes block]
48
+ O: [state-of-play → checkpoint → re-anchored] "Resuming: merchant notifications (in-app+digest+prefs). Still correct?" U: yes.
49
+ ▶ W3: spawn T5+T6 (haiku-4-5) → done. All 6 complete.
50
+
51
+ ## Phase 5 — Review (spawned worker, Opus 5.5 medium)
52
+ O: Code pattern? U: none. Findings (P0–P3 ladder, each P0–P2 anchored file:line + verbatim quote):
53
+ P1 T2 no pagination — services/notification_list.ts:88 — O spot-checks: Reads line 88, quote matches ✓ → asks user, U: fix it
54
+ · P3 email_digest naming ×2 · 12 tests added. (An unanchored finding would be auto-P3 — never fixed unasked.)
55
+
56
+ ## Phase 6 — Verify (spawned worker, Opus 5.5 medium; criteria = definition-of-done)
57
+ criteria[] with per-line evidence: bell ✓ (test name) · both schedules ✗ (weekly cron missing = P0, unmet criterion, evidence: no cron entry in scheduler.ts) · prefs ✓ → FAILED → quote spot-check ✓ → P0 auto-fix (single attempt, no ask) → re-verify → PASSED. 14/14.
58
+
59
+ ## Phase 7 — Summary
60
+ O: 6 tasks/3 waves/6 spawns/1 pause/1 fresh-session resume · 1 needs_context · 1 escalation · verify fixed_then_passed · branch green. P2 fix-batch: none · Apply the 2 P3 cosmetics? U: no.
61
+ O: review close → `orc gotcha observe` ×3 (P1 addressed · 2× P3 wontfix, each with `run`) → `FINDING-OUTCOME addressed=1 disputed=0 wontfix=2 open=0 pre=0 suppressed=0 :: functional.check:1/1,evolvability.documentation:0/2`
62
+
63
+ ## Phase 8 — Ship (orc-pr, reads pr.md)
64
+ O: Branch: feat/notifications. Commit? Push? PR? U: yes all.
65
+ O: Ticket/title/target? U: DRP-482 / Add merchant notification system / develop
66
+ O: ✓ committed → pushed → run/merchant-notifications/DRP-482-add-merchant-notification-system.md from pr.md (base develop).
67
+ Log deleted · checkpoint + dispatch log KEPT · completion dispatch report shown → "Run /usage to see your remaining limits."
68
+
69
+ ## What this exercises
70
+ always-spawn (even cheap tasks get a subagent) · facet-scored arithmetic · 5-band ladder ·
71
+ per-run folder run/merchant-notifications/ · usage reminder (no programmatic /usage) ·
72
+ planner grounding spot-check catching a hallucinated path · executor evidence (verbatim
73
+ build/test proof) · reviewer quote spot-check before any P0/P1 action ·
74
+ definition-of-done catching the weekly-cron gap · needs_context resolved internally · fresh-session resume.
@@ -1,222 +1,222 @@
1
- # Reference — Effort, Dispatch Style, and Task Scoring
2
-
3
- **Subagents ALWAYS do the work — the orchestrator never implements** (hard
4
- rule 1). The orchestrator is Opus 4.8 high: it doing implementation is the most
5
- costly way possible, and it burns the context that must survive the whole run.
6
- Spawning keeps orchestrator context lean so runs last longer before any pause.
7
-
8
- Two INDEPENDENT axes. Never conflate them:
9
- - **Run-level effort** (low/medium/high) → picks the **dispatch style**
10
- (sequential vs parallel), which controls only **intra-wave concurrency** —
11
- never WHO does the work, and never WHETHER waves exist.
12
- - **Per-task score** (0–100) → picks each **worker's model**. Always applies.
13
-
14
- A medium run can contain a 30-score task and a 95-score task in the same wave.
15
-
16
- ## Waves always exist; dispatch style is intra-wave only
17
-
18
- **Wave computation runs for every run with ≥2 tasks, sequential included**
19
- (dependency layers + conflict graph + `max_wave_tasks` cap — see
20
- `wave-grouping.md`). Dispatch style does NOT decide whether there are waves; it
21
- decides how a wave's tasks fire:
22
-
23
- - **sequential** → a wave's tasks dispatch one at a time, in order; the wave
24
- closes when all its tasks close;
25
- - **parallel** → a wave's non-conflicting tasks dispatch at once (up to
26
- `max_wave_tasks`).
27
-
28
- The **wave-boundary batch pause binds to wave numbers, identically in both
29
- styles** — a sequential run is NOT "no waves / per-task pauses". Show the wave
30
- plan (wave → tasks → pause marks) to the user BEFORE wave 1 in both styles.
31
-
32
- ## Run-level effort → dispatch style (recommend, user confirms)
33
-
34
- - **Low**, or heavy shared data/code → **sequential**: one scored subagent at a
35
- time, in dependency order (still grouped into waves). Even a single trivial
36
- task = one cheap subagent (typically Sonnet 4.6 medium), never the
37
- orchestrator itself.
38
- - **Medium** → sequential by default; **parallel** intra-wave if 3+ genuinely
39
- independent areas.
40
- - **High** with independent areas → **parallel** waves; consider worktrees for
41
- isolation (merge at end).
42
-
43
- Always RECOMMEND with a one-line why, and let the user pick. Tell them the
44
- agent/task/wave counts, each task's scored model, and where batch pauses fall
45
- BEFORE dispatching.
46
-
47
- ## Per-task scoring — facet-scored, arithmetic, two-writer
48
-
49
- The score is **not judged** from a task title. The **planner** — the one party
50
- that globbed and read every declared file — emits per-task `facets` (facts). The
51
- **orchestrator** — who never read the code — computes the number arithmetically
52
- and audits the facts. No vibes anywhere. Every task is scored, including tiny ones.
53
-
54
- **1. Planner-emitted facets** (`planning-output.md` `facets` block, filled during
55
- grounding — zero extra passes):
56
-
57
- | Facet | Values |
58
- |---|---|
59
- | `breadth` | `len(declared_files)` — computed, not judged |
60
- | `novelty` | mechanical · imitate · new-surface · novel-algorithm |
61
- | `logic` | none · branching · stateful · algorithmic |
62
- | `test_surface` | none · update-existing · new-tests |
63
- | `risk` | `[]` or `[{class, cite}]` — class ∈ auth·money·migration·security·concurrency·data-integrity; **each entry MUST cite the file/requirement** |
64
- | `uncertainty` | low · medium · high (+ one-line reason if not low) |
65
-
66
- `fan_in`/`fan_out` are NOT emitted — the orchestrator computes them from
67
- `depends_on` (forward = fan_in, reverse = fan_out).
68
-
69
- **2. Orchestrator validation gate (Phase 2, deterministic — same bounce
70
- mechanics as grounding):** emit `GATE facet pass|bounce`. Three checks:
71
-
72
- - **Recompute** `breadth` (= `len(declared_files)`) and `fan_in`/`fan_out` from
73
- the plan itself — a mismatch bounces.
74
- - **MEMBERSHIP (v0.34.4):** `novelty`, `logic`, `test_surface` and `uncertainty`
75
- must each be a member of their CLOSED set above, and every `risk[].class` must
76
- be one of the six risk classes. A set lookup, as deterministic as the grounding
77
- Glob. This is the check that was missing: the gate validated only the two
78
- facets it could recompute and TRUSTED the four it could not, so a planner that
79
- invented a `low|medium|high` scale produced an **arithmetically unscorable**
80
- plan that passed the stated gate. There is no `N("low")` — and an orchestrator
81
- applying `risk != [] -> floor 70` to prose risk strings floors EVERY task, a
82
- two-band overshoot caused by a field's shape rather than by the work.
83
- - **Cite check:** a `risk` entry with no `cite` bounces (unchanged).
84
-
85
- A facet-vocabulary miss is a **FIELD-SHAPE bounce**: hand back the miss list and
86
- say *do not re-plan* — the planner corrects only the offending `facets` blocks.
87
- (It may also legitimately DROP a risk entry on the correction pass: implementation
88
- hazards like test seeding or dependency discipline are not risk CLASSES. That
89
- judgment is the planner's to make and is worth accepting.) Without this clause a
90
- shape bounce risks costing a correct plan. A plan with no `facets` at all
91
- (pre-v0.31.0) resumes on the legacy path — never bounced for the missing block.
92
-
93
- **3. The fixed formula (the ONLY scoring text — this replaces base+adjusters):**
94
-
95
- ```
96
- score = B(breadth) + N(novelty) + L(logic) + T(test_surface)
97
- + 5*min(fan_in,3) + 3*min(fan_out,3) + U(uncertainty)
98
-
99
- B: 1f=2 2-3f=6 4-5f=10 6+f=15
100
- N: mechanical=0 imitate=8 new-surface=18 novel-algorithm=30
101
- L: none=0 branching=8 stateful=16 algorithmic=24
102
- T: none=0 update-existing=4 new-tests=8
103
- U: low=0 medium=6 high=12
104
-
105
- risk ≠ [] → floor 70 (DERIVED from a cited risk facet, never remembered).
106
- clamp 0..100 → the RESOLVED table (below).
107
- ```
108
-
109
- **Which table (highest wins):** `opus5_only: true` (the 2-band Opus-5-only
110
- preset: `[0,90)` low · `[90,100]` medium) → `rubric_bands_override`
111
- (hand-written rows) → the default 6-band table. All three are in `config.md`.
112
- The formula, the facets and the risk floor are IDENTICAL in every case — only
113
- the score→agent mapping changes. Three consequences worth stating so nobody
114
- re-derives them per run:
115
-
116
- - **The risk floor still applies** — it raises the SCORE, then the resolved table
117
- maps it. A floored task (≥70) lands `opus-5-low` in the default table and
118
- `opus-5-low` under the Opus-5-only preset too: since v1.0.0 both tables answer
119
- the floor with the same agent, and only a score of 90+ moves it to `opus-5-med`.
120
- - **`opus5_only` FORCES** — while on, a hand-written `rubric_bands_override` is
121
- ignored. It is the one selector that can shadow another; that is deliberate.
122
-
123
- Show the user the full table (task, the facet vector, the arithmetic
124
- `B+N+L+T+fan+U = raw`, any risk floor, final, override+reason if any, dispatched
125
- model) BEFORE dispatching — an un-shown number is not a scored number. **Head it
126
- with the RESOLVED table's name** (`6-band default` / `Opus-5-only ladder
127
- (opus5_only)` / `custom (rubric_bands_override)`): the same logic
128
- applies to the mapping as to the number.
129
-
130
- **Extra (v0.50.0, `extra_enabled`) adds a `via` column and can make the head a
131
- PAIR.** An Extra route row is an OVERLAY that outranks the tables above **only
132
- for the scores it covers**, so a run can genuinely be running two tables at once
133
- and the head names both (`6-band default + extra rows [0,30) [30,70)`). The
134
- column reads `claude` or `extra:<profile> (<engine>)` and comes from
135
- `orc extra resolve --json` — the formula, the facets and the risk floor are
136
- untouched, and **the score is computed before the routing, never after it**. Two
137
- consequences the table has to show rather than imply: a cited `risk[]` HOLDS THE
138
- TASK BACK to its Claude band whatever the row says (`extra_risk_tasks: off`), and
139
- the `model` for a foreign task is the FOREIGN model id, not a Claude tier.
140
- Canonical: `../../_shared/extra-dispatch.md`.
141
-
142
- **4. Consistency check:** two tasks whose facet vectors differ in **≤1 facet must
143
- land in the same band** — or the SCORE line for the outlier **cites the one
144
- differing facet**. This is the specific fix for sibling tasks in the same domain
145
- drifting into three different bands.
146
-
147
- **5. Override protocol** (unchanged): you may override the computed score. Record
148
- `{computed_score, override_score, reason}` in the dispatch log. An override
149
- without a reason is invalid.
150
-
151
- ## EVERY dispatch is scored — the fix-cycle rule
152
-
153
- Fix-cycle dispatches — review-fix, verify-fix, the Phase-7 P2-batch, and any
154
- requeue — are scored through the **same formula**, with two floors that stop a
155
- fix in a risk area from silently dropping to a cheap model:
156
-
157
- - **Inherit the ORIGINAL task's `risk` facets** when the fix touches its files —
158
- a fix in a risk-floor area keeps the ≥70 floor (a userID/context-key fix can
159
- never dispatch below `opus-5-low` again);
160
- - a **P0/P1 fix never dispatches below the band of the task that produced the
161
- finding.**
162
-
163
- Emit a `SCORE task=fix-<n> …` line for each (they appear in the dispatch log and
164
- the completion table like any task).
165
-
166
- ## Worked scoring examples (facet vectors — compare facet-by-facet, not by vibes)
167
-
168
- Each row is the facet vector run through the formula. The number is arithmetic;
169
- the band is what matters. Compare a new task to these facet-by-facet.
170
-
171
- | Task | breadth·novelty·logic·test · fan_in/out · unc | Arithmetic | Band |
172
- |---|---|---|---|
173
- | Rename a config key across 4 files + its test | 5·mechanical·none·update-existing · 0/0 · low | 10+0+0+4 = **14** | haiku [0,30) |
174
- | New CRUD endpoint following a sibling route | 3·imitate·branching·new-tests · 1/0 · low | 6+8+8+8+5 = **35** | sonnet-4-6-med [30,40) |
175
- | Bug fix across 2 files with a repro test | 3·imitate·branching·new-tests · 0/0 · medium | 6+8+8+8+6 = **36** | sonnet-4-6-med [30,40) |
176
- | Isolated component from the design system | 3·new-surface·branching·new-tests · 0/0 · low | 6+18+8+8 = **40** | sonnet-4-6-high [40,55) |
177
- | Notification model + enum other tasks consume | 3·new-surface·stateful·new-tests · 0/3 · low | 6+18+16+8+9 = **57** | sonnet-5-high [55,65) |
178
- | Service-layer refactor behind a stable interface | 5·imitate·stateful·update-existing · 0/3 · medium | 10+8+16+4+9+6 = **53** | sonnet-4-6-high [40,55) |
179
- | Add role check to payment-refund endpoint | 3·imitate·branching·new-tests · 0/0 · low · **risk=[auth,money]** | 30 raw → **floor 70** | opus-5-low [65,90) |
180
- | Migrate orders table to split-name + backfill | 6·new-surface·stateful·new-tests · 1/3 · high · **risk=[migration,data-integrity]** | 15+18+16+8+5+9+12 = 83 (floor 70) → **83** | opus-5-low [65,90) |
181
-
182
- Two disciplines the vectors encode: (1) a small diff is NOT a low score when a
183
- cited `risk` facet forces the floor (the refund row — 30 raw, floored to 70); (2)
184
- a big-looking task IS low when its facets are mechanical (the rename). The risk
185
- floor is always DERIVED from a cited facet — the SCORE line names it, never
186
- applies it silently.
187
-
188
- ## Model ladder → the single score→model table
189
-
190
- The score→model mapping is NOT hardcoded here — it lives in `config.md` as ONE
191
- canonical 6-band table (there is no longer a narrow/wide preset). Read config at
192
- run start and map each task's final score through that table (or
193
- `rubric_bands_override`). The orchestrator dispatches the executor agent BY NAME;
194
- it does not request a raw model. `rubric_bands` sets only how many bands the
195
- rubric REPORTS (score granularity), never which table is used.
196
-
197
- The 6 bands (see config.md for the exact edges): `haiku-4-5` [0,30) ·
198
- `sonnet-4-6-med` [30,40) · `sonnet-4-6-high` [40,55) · `sonnet-5-high` [55,65) ·
199
- `opus-5-low` [65,90) · `opus-5-med` [90,100]. Effort tiers rank
200
- `low < medium < high < xhigh < max`.
201
-
202
- ## Fixed model assignments (not scored)
203
-
204
- - **Orchestrator (you):** Opus 4.8 high — or Opus 5.5 / Fable 5 at medium+, the
205
- two models that clear the guard from medium up. Never downgrade yourself.
206
- - **Review** — Superpowers path: Sonnet 4.6, medium. OpenSpec/self path:
207
- Opus 5.5, medium (`orc-reviewer-opus-5-med`).
208
- - **Verify:** Opus 5.5, medium (`orc-verifier-opus-5-med`).
209
- - **Analyst:** Opus 5.5, high. **Planner:** Opus 5.5, medium. **Test author:**
210
- Opus 5.5, medium. **Combiner:** Opus 5.5, high. **Ultra advisor/judge:** Opus 5.5,
211
- xhigh — every core fixed role is pinned to Opus 5.5 as of v0.34.0.
212
- - **Merge-conflict resolver:** Opus 4.8, medium.
213
-
214
- Note: every band in the table above is a real dispatch target (haiku through
215
- opus-5). If a model tier is unavailable in the environment, fall back UP to the
216
- next capable tier (never silently substitute a different family/effort).
217
-
218
- ## Dispatch log (lives in the checkpoint)
219
-
220
- Every spawn records: `{task_id, computed_score, override_score|null,
221
- override_reason|null, model, effort, spawned_at}`. Feeds the usage report at
222
- every stop.
1
+ # Reference — Effort, Dispatch Style, and Task Scoring
2
+
3
+ **Subagents ALWAYS do the work — the orchestrator never implements** (hard
4
+ rule 1). The orchestrator is Opus 4.8 high: it doing implementation is the most
5
+ costly way possible, and it burns the context that must survive the whole run.
6
+ Spawning keeps orchestrator context lean so runs last longer before any pause.
7
+
8
+ Two INDEPENDENT axes. Never conflate them:
9
+ - **Run-level effort** (low/medium/high) → picks the **dispatch style**
10
+ (sequential vs parallel), which controls only **intra-wave concurrency** —
11
+ never WHO does the work, and never WHETHER waves exist.
12
+ - **Per-task score** (0–100) → picks each **worker's model**. Always applies.
13
+
14
+ A medium run can contain a 30-score task and a 95-score task in the same wave.
15
+
16
+ ## Waves always exist; dispatch style is intra-wave only
17
+
18
+ **Wave computation runs for every run with ≥2 tasks, sequential included**
19
+ (dependency layers + conflict graph + `max_wave_tasks` cap — see
20
+ `wave-grouping.md`). Dispatch style does NOT decide whether there are waves; it
21
+ decides how a wave's tasks fire:
22
+
23
+ - **sequential** → a wave's tasks dispatch one at a time, in order; the wave
24
+ closes when all its tasks close;
25
+ - **parallel** → a wave's non-conflicting tasks dispatch at once (up to
26
+ `max_wave_tasks`).
27
+
28
+ The **wave-boundary batch pause binds to wave numbers, identically in both
29
+ styles** — a sequential run is NOT "no waves / per-task pauses". Show the wave
30
+ plan (wave → tasks → pause marks) to the user BEFORE wave 1 in both styles.
31
+
32
+ ## Run-level effort → dispatch style (recommend, user confirms)
33
+
34
+ - **Low**, or heavy shared data/code → **sequential**: one scored subagent at a
35
+ time, in dependency order (still grouped into waves). Even a single trivial
36
+ task = one cheap subagent (typically Sonnet 5 low), never the
37
+ orchestrator itself.
38
+ - **Medium** → sequential by default; **parallel** intra-wave if 3+ genuinely
39
+ independent areas.
40
+ - **High** with independent areas → **parallel** waves; consider worktrees for
41
+ isolation (merge at end).
42
+
43
+ Always RECOMMEND with a one-line why, and let the user pick. Tell them the
44
+ agent/task/wave counts, each task's scored model, and where batch pauses fall
45
+ BEFORE dispatching.
46
+
47
+ ## Per-task scoring — facet-scored, arithmetic, two-writer
48
+
49
+ The score is **not judged** from a task title. The **planner** — the one party
50
+ that globbed and read every declared file — emits per-task `facets` (facts). The
51
+ **orchestrator** — who never read the code — computes the number arithmetically
52
+ and audits the facts. No vibes anywhere. Every task is scored, including tiny ones.
53
+
54
+ **1. Planner-emitted facets** (`planning-output.md` `facets` block, filled during
55
+ grounding — zero extra passes):
56
+
57
+ | Facet | Values |
58
+ |---|---|
59
+ | `breadth` | `len(declared_files)` — computed, not judged |
60
+ | `novelty` | mechanical · imitate · new-surface · novel-algorithm |
61
+ | `logic` | none · branching · stateful · algorithmic |
62
+ | `test_surface` | none · update-existing · new-tests |
63
+ | `risk` | `[]` or `[{class, cite}]` — class ∈ auth·money·migration·security·concurrency·data-integrity; **each entry MUST cite the file/requirement** |
64
+ | `uncertainty` | low · medium · high (+ one-line reason if not low) |
65
+
66
+ `fan_in`/`fan_out` are NOT emitted — the orchestrator computes them from
67
+ `depends_on` (forward = fan_in, reverse = fan_out).
68
+
69
+ **2. Orchestrator validation gate (Phase 2, deterministic — same bounce
70
+ mechanics as grounding):** emit `GATE facet pass|bounce`. Three checks:
71
+
72
+ - **Recompute** `breadth` (= `len(declared_files)`) and `fan_in`/`fan_out` from
73
+ the plan itself — a mismatch bounces.
74
+ - **MEMBERSHIP (v0.34.4):** `novelty`, `logic`, `test_surface` and `uncertainty`
75
+ must each be a member of their CLOSED set above, and every `risk[].class` must
76
+ be one of the six risk classes. A set lookup, as deterministic as the grounding
77
+ Glob. This is the check that was missing: the gate validated only the two
78
+ facets it could recompute and TRUSTED the four it could not, so a planner that
79
+ invented a `low|medium|high` scale produced an **arithmetically unscorable**
80
+ plan that passed the stated gate. There is no `N("low")` — and an orchestrator
81
+ applying `risk != [] -> floor 70` to prose risk strings floors EVERY task, a
82
+ two-band overshoot caused by a field's shape rather than by the work.
83
+ - **Cite check:** a `risk` entry with no `cite` bounces (unchanged).
84
+
85
+ A facet-vocabulary miss is a **FIELD-SHAPE bounce**: hand back the miss list and
86
+ say *do not re-plan* — the planner corrects only the offending `facets` blocks.
87
+ (It may also legitimately DROP a risk entry on the correction pass: implementation
88
+ hazards like test seeding or dependency discipline are not risk CLASSES. That
89
+ judgment is the planner's to make and is worth accepting.) Without this clause a
90
+ shape bounce risks costing a correct plan. A plan with no `facets` at all
91
+ (pre-v0.31.0) resumes on the legacy path — never bounced for the missing block.
92
+
93
+ **3. The fixed formula (the ONLY scoring text — this replaces base+adjusters):**
94
+
95
+ ```
96
+ score = B(breadth) + N(novelty) + L(logic) + T(test_surface)
97
+ + 5*min(fan_in,3) + 3*min(fan_out,3) + U(uncertainty)
98
+
99
+ B: 1f=2 2-3f=6 4-5f=10 6+f=15
100
+ N: mechanical=0 imitate=8 new-surface=18 novel-algorithm=30
101
+ L: none=0 branching=8 stateful=16 algorithmic=24
102
+ T: none=0 update-existing=4 new-tests=8
103
+ U: low=0 medium=6 high=12
104
+
105
+ risk ≠ [] → floor 70 (DERIVED from a cited risk facet, never remembered).
106
+ clamp 0..100 → the RESOLVED table (below).
107
+ ```
108
+
109
+ **Which table (highest wins):** `opus5_only: true` (the 2-band Opus-5-only
110
+ preset: `[0,90)` low · `[90,100]` medium) → `rubric_bands_override`
111
+ (hand-written rows) → the default 5-band table. All three are in `config.md`.
112
+ The formula, the facets and the risk floor are IDENTICAL in every case — only
113
+ the score→agent mapping changes. Three consequences worth stating so nobody
114
+ re-derives them per run:
115
+
116
+ - **The risk floor still applies** — it raises the SCORE, then the resolved table
117
+ maps it. A floored task (≥70) lands `opus-5-low` in the default table and
118
+ `opus-5-low` under the Opus-5-only preset too: since v1.0.0 both tables answer
119
+ the floor with the same agent, and only a score of 90+ moves it to `opus-5-med`.
120
+ - **`opus5_only` FORCES** — while on, a hand-written `rubric_bands_override` is
121
+ ignored. It is the one selector that can shadow another; that is deliberate.
122
+
123
+ Show the user the full table (task, the facet vector, the arithmetic
124
+ `B+N+L+T+fan+U = raw`, any risk floor, final, override+reason if any, dispatched
125
+ model) BEFORE dispatching — an un-shown number is not a scored number. **Head it
126
+ with the RESOLVED table's name** (`5-band default` / `Opus-5-only ladder
127
+ (opus5_only)` / `custom (rubric_bands_override)`): the same logic
128
+ applies to the mapping as to the number.
129
+
130
+ **Extra (v0.50.0, `extra_enabled`) adds a `via` column and can make the head a
131
+ PAIR.** An Extra route row is an OVERLAY that outranks the tables above **only
132
+ for the scores it covers**, so a run can genuinely be running two tables at once
133
+ and the head names both (`5-band default + extra rows [0,30) [30,70)`). The
134
+ column reads `claude` or `extra:<profile> (<engine>)` and comes from
135
+ `orc extra resolve --json` — the formula, the facets and the risk floor are
136
+ untouched, and **the score is computed before the routing, never after it**. Two
137
+ consequences the table has to show rather than imply: a cited `risk[]` HOLDS THE
138
+ TASK BACK to its Claude band whatever the row says (`extra_risk_tasks: off`), and
139
+ the `model` for a foreign task is the FOREIGN model id, not a Claude tier.
140
+ Canonical: `../../_shared/extra-dispatch.md`.
141
+
142
+ **4. Consistency check:** two tasks whose facet vectors differ in **≤1 facet must
143
+ land in the same band** — or the SCORE line for the outlier **cites the one
144
+ differing facet**. This is the specific fix for sibling tasks in the same domain
145
+ drifting into three different bands.
146
+
147
+ **5. Override protocol** (unchanged): you may override the computed score. Record
148
+ `{computed_score, override_score, reason}` in the dispatch log. An override
149
+ without a reason is invalid.
150
+
151
+ ## EVERY dispatch is scored — the fix-cycle rule
152
+
153
+ Fix-cycle dispatches — review-fix, verify-fix, the Phase-7 P2-batch, and any
154
+ requeue — are scored through the **same formula**, with two floors that stop a
155
+ fix in a risk area from silently dropping to a cheap model:
156
+
157
+ - **Inherit the ORIGINAL task's `risk` facets** when the fix touches its files —
158
+ a fix in a risk-floor area keeps the ≥70 floor (a userID/context-key fix can
159
+ never dispatch below `opus-5-low` again);
160
+ - a **P0/P1 fix never dispatches below the band of the task that produced the
161
+ finding.**
162
+
163
+ Emit a `SCORE task=fix-<n> …` line for each (they appear in the dispatch log and
164
+ the completion table like any task).
165
+
166
+ ## Worked scoring examples (facet vectors — compare facet-by-facet, not by vibes)
167
+
168
+ Each row is the facet vector run through the formula. The number is arithmetic;
169
+ the band is what matters. Compare a new task to these facet-by-facet.
170
+
171
+ | Task | breadth·novelty·logic·test · fan_in/out · unc | Arithmetic | Band |
172
+ |---|---|---|---|
173
+ | Rename a config key across 4 files + its test | 5·mechanical·none·update-existing · 0/0 · low | 10+0+0+4 = **14** | sonnet-5-low [0,21) |
174
+ | New CRUD endpoint following a sibling route | 3·imitate·branching·new-tests · 1/0 · low | 6+8+8+8+5 = **35** | sonnet-5-high [31,41) |
175
+ | Bug fix across 2 files with a repro test | 3·imitate·branching·new-tests · 0/0 · medium | 6+8+8+8+6 = **36** | sonnet-5-high [31,41) |
176
+ | Isolated component from the design system | 3·new-surface·branching·new-tests · 0/0 · low | 6+18+8+8 = **40** | sonnet-5-high [31,41) |
177
+ | Notification model + enum other tasks consume | 3·new-surface·stateful·new-tests · 0/3 · low | 6+18+16+8+9 = **57** | opus-5-low [41,90) |
178
+ | Service-layer refactor behind a stable interface | 5·imitate·stateful·update-existing · 0/3 · medium | 10+8+16+4+9+6 = **53** | opus-5-low [41,90) |
179
+ | Add role check to payment-refund endpoint | 3·imitate·branching·new-tests · 0/0 · low · **risk=[auth,money]** | 30 raw → **floor 70** | opus-5-low [41,90) |
180
+ | Migrate orders table to split-name + backfill | 6·new-surface·stateful·new-tests · 1/3 · high · **risk=[migration,data-integrity]** | 15+18+16+8+5+9+12 = 83 (floor 70) → **83** | opus-5-low [41,90) |
181
+
182
+ Two disciplines the vectors encode: (1) a small diff is NOT a low score when a
183
+ cited `risk` facet forces the floor (the refund row — 30 raw, floored to 70); (2)
184
+ a big-looking task IS low when its facets are mechanical (the rename). The risk
185
+ floor is always DERIVED from a cited facet — the SCORE line names it, never
186
+ applies it silently.
187
+
188
+ ## Model ladder → the single score→model table
189
+
190
+ The score→model mapping is NOT hardcoded here — it lives in `config.md` as ONE
191
+ canonical 5-band table (there is no longer a narrow/wide preset). Read config at
192
+ run start and map each task's final score through that table (or
193
+ `rubric_bands_override`). The orchestrator dispatches the executor agent BY NAME;
194
+ it does not request a raw model. `rubric_bands` sets only how many bands the
195
+ rubric REPORTS (score granularity), never which table is used.
196
+
197
+ The 5 bands (see config.md for the exact edges): `sonnet-5-low` [0,21) ·
198
+ `sonnet-5-med` [21,31) · `sonnet-5-high` [31,41) · `opus-5-low` [41,90) ·
199
+ `opus-5-med` [90,100]. Effort tiers rank
200
+ `low < medium < high < xhigh < max`.
201
+
202
+ ## Fixed model assignments (not scored)
203
+
204
+ - **Orchestrator (you):** Opus 4.8 high — or Opus 5.5 / Fable 5 at medium+, the
205
+ two models that clear the guard from medium up. Never downgrade yourself.
206
+ - **Review** — Superpowers path: Sonnet 4.6, medium. OpenSpec/self path:
207
+ Opus 5.5, low (`orc-reviewer-opus-5-low`; medium until v2.1.0).
208
+ - **Verify:** Opus 5.5, medium (`orc-verifier-opus-5-med`).
209
+ - **Analyst:** Opus 5.5, high. **Planner:** Opus 5.5, medium. **Test author:**
210
+ Opus 5.5, medium. **Combiner:** Opus 5.5, high. **Ultra advisor/judge:** Opus 5.5,
211
+ xhigh — every core fixed role is pinned to Opus 5.5 as of v0.34.0.
212
+ - **Merge-conflict resolver:** Opus 4.8, medium.
213
+
214
+ Note: every band in the table above is a real dispatch target (sonnet-5 through
215
+ opus-5). If a model tier is unavailable in the environment, fall back UP to the
216
+ next capable tier (never silently substitute a different family/effort).
217
+
218
+ ## Dispatch log (lives in the checkpoint)
219
+
220
+ Every spawn records: `{task_id, computed_score, override_score|null,
221
+ override_reason|null, model, effort, spawned_at}`. Feeds the usage report at
222
+ every stop.
@@ -25,7 +25,7 @@ after: src/payments — 2 shipped files rewritten within 30 days of run stor
25
25
  crosslink: 2 boundaries (payments-api) — advisory
26
26
  extra: ON — 4 of 9 tasks foreign · deepseek/deepseek-v4-flash via api [0,30)
27
27
  · glm/glm-4.7 via cli [30,70) · 2 held back (risk: auth, money)
28
- scoring: 6-band default table
28
+ scoring: 5-band default table
29
29
  tdd: 3 tasks with tests (T3, T6, T9) · 2 covered-by-existing · 2 no-behavior
30
30
  skipped: R4 translation strings (no-behavior) · R7 file split
31
31
  (covered-by-existing → test/api/health.test.js:41)
@@ -100,7 +100,7 @@ waves: 3 planned — will pause after wave 2 (batch_pause_every=2)
100
100
  `EXTRA orphan` line from `trace_extras[]` once you have reported it.
101
101
 
102
102
  Canonical: `../../_shared/extra-dispatch.md`.
103
- - **scoring:** which executor table RESOLVED for this run — `6-band default
103
+ - **scoring:** which executor table RESOLVED for this run — `5-band default
104
104
  table` · `Opus-5-only ladder (opus5_only)` · `custom
105
105
  (rubric_bands_override, <n> rows)`. An un-shown table is as unaccountable as
106
106
  an un-shown number, and the Opus-5-only ladder in particular means EVERY
@@ -109,7 +109,7 @@ waves: 3 planned — will pause after wave 2 (batch_pause_every=2)
109
109
  append ` · all fixed roles forced to Opus 5.5` and name any selector it
110
110
  shadowed (a `rubric_bands_override` present but INERT) — a setting
111
111
  the user tuned and the run then ignored has to be said out loud. With Extra in
112
- play the table is a COMPOSITE and reads as one (`6-band default table + extra
112
+ play the table is a COMPOSITE and reads as one (`5-band default table + extra
113
113
  rows [0,30) [30,70)`) — Extra is an overlay, so naming only one of the two would
114
114
  be naming the wrong half for every covered score.
115
115
  - **tdd:** ALWAYS printed on a lane whose TDD policy is on — BOTH branches, not