codex-orchestrator 0.1.36 → 0.1.39

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (113) hide show
  1. package/CHANGELOG.md +34 -0
  2. package/dist/src/codex/command-adapter.d.ts +1 -0
  3. package/dist/src/codex/command-adapter.d.ts.map +1 -1
  4. package/dist/src/codex/command-adapter.js +61 -1
  5. package/dist/src/codex/command-adapter.js.map +1 -1
  6. package/dist/src/config/schema.d.ts +24 -1
  7. package/dist/src/config/schema.d.ts.map +1 -1
  8. package/dist/src/config/schema.js +55 -0
  9. package/dist/src/config/schema.js.map +1 -1
  10. package/dist/src/git/worktree.d.ts.map +1 -1
  11. package/dist/src/git/worktree.js +0 -1
  12. package/dist/src/git/worktree.js.map +1 -1
  13. package/dist/src/index.d.ts +1 -1
  14. package/dist/src/index.d.ts.map +1 -1
  15. package/dist/src/index.js.map +1 -1
  16. package/dist/src/review-handoff.d.ts +3 -0
  17. package/dist/src/review-handoff.d.ts.map +1 -0
  18. package/dist/src/review-handoff.js +8 -0
  19. package/dist/src/review-handoff.js.map +1 -0
  20. package/dist/src/runner/acceptance-proof-runner.d.ts +2 -15
  21. package/dist/src/runner/acceptance-proof-runner.d.ts.map +1 -1
  22. package/dist/src/runner/acceptance-proof-runner.js +40 -72
  23. package/dist/src/runner/acceptance-proof-runner.js.map +1 -1
  24. package/dist/src/runner/acceptance-proof.d.ts +60 -0
  25. package/dist/src/runner/acceptance-proof.d.ts.map +1 -1
  26. package/dist/src/runner/acceptance-proof.js +113 -0
  27. package/dist/src/runner/acceptance-proof.js.map +1 -1
  28. package/dist/src/runner/android-visual-proof-command.d.ts.map +1 -1
  29. package/dist/src/runner/android-visual-proof-command.js +2 -8
  30. package/dist/src/runner/android-visual-proof-command.js.map +1 -1
  31. package/dist/src/runner/auto-visual-proof-command.d.ts.map +1 -1
  32. package/dist/src/runner/auto-visual-proof-command.js +11 -1
  33. package/dist/src/runner/auto-visual-proof-command.js.map +1 -1
  34. package/dist/src/runner/command-utils.d.ts.map +1 -1
  35. package/dist/src/runner/command-utils.js +19 -10
  36. package/dist/src/runner/command-utils.js.map +1 -1
  37. package/dist/src/runner/completion-report.d.ts +21 -0
  38. package/dist/src/runner/completion-report.d.ts.map +1 -1
  39. package/dist/src/runner/completion-report.js +46 -0
  40. package/dist/src/runner/completion-report.js.map +1 -1
  41. package/dist/src/runner/daemon-command.js +2 -35
  42. package/dist/src/runner/daemon-command.js.map +1 -1
  43. package/dist/src/runner/flutter-sdk-discovery.d.ts +5 -0
  44. package/dist/src/runner/flutter-sdk-discovery.d.ts.map +1 -0
  45. package/dist/src/runner/flutter-sdk-discovery.js +103 -0
  46. package/dist/src/runner/flutter-sdk-discovery.js.map +1 -0
  47. package/dist/src/runner/handoff-evidence.d.ts +12 -1
  48. package/dist/src/runner/handoff-evidence.d.ts.map +1 -1
  49. package/dist/src/runner/handoff-evidence.js +93 -1
  50. package/dist/src/runner/handoff-evidence.js.map +1 -1
  51. package/dist/src/runner/ios-visual-proof-command.d.ts.map +1 -1
  52. package/dist/src/runner/ios-visual-proof-command.js +2 -4
  53. package/dist/src/runner/ios-visual-proof-command.js.map +1 -1
  54. package/dist/src/runner/issue-tree.d.ts.map +1 -1
  55. package/dist/src/runner/issue-tree.js +28 -2
  56. package/dist/src/runner/issue-tree.js.map +1 -1
  57. package/dist/src/runner/local-execution-session.d.ts.map +1 -1
  58. package/dist/src/runner/local-execution-session.js +80 -7
  59. package/dist/src/runner/local-execution-session.js.map +1 -1
  60. package/dist/src/runner/plan-auto-command.d.ts +1 -0
  61. package/dist/src/runner/plan-auto-command.d.ts.map +1 -1
  62. package/dist/src/runner/plan-auto-command.js +83 -32
  63. package/dist/src/runner/plan-auto-command.js.map +1 -1
  64. package/dist/src/runner/prompt.d.ts.map +1 -1
  65. package/dist/src/runner/prompt.js +24 -5
  66. package/dist/src/runner/prompt.js.map +1 -1
  67. package/dist/src/runner/proof-strategy.d.ts +12 -0
  68. package/dist/src/runner/proof-strategy.d.ts.map +1 -0
  69. package/dist/src/runner/proof-strategy.js +19 -0
  70. package/dist/src/runner/proof-strategy.js.map +1 -0
  71. package/dist/src/runner/review-gate-policy.d.ts +1 -1
  72. package/dist/src/runner/review-gate-policy.d.ts.map +1 -1
  73. package/dist/src/runner/review-gate-policy.js +75 -3
  74. package/dist/src/runner/review-gate-policy.js.map +1 -1
  75. package/dist/src/runner/review-gates.d.ts +5 -0
  76. package/dist/src/runner/review-gates.d.ts.map +1 -1
  77. package/dist/src/runner/review-gates.js +113 -1
  78. package/dist/src/runner/review-gates.js.map +1 -1
  79. package/dist/src/runner/rework-policy.d.ts +1 -0
  80. package/dist/src/runner/rework-policy.d.ts.map +1 -1
  81. package/dist/src/runner/rework-policy.js +7 -1
  82. package/dist/src/runner/rework-policy.js.map +1 -1
  83. package/dist/src/runner/runner-handoff-decision.d.ts +47 -0
  84. package/dist/src/runner/runner-handoff-decision.d.ts.map +1 -0
  85. package/dist/src/runner/runner-handoff-decision.js +66 -0
  86. package/dist/src/runner/runner-handoff-decision.js.map +1 -0
  87. package/dist/src/runner/scope-isolation-policy.d.ts +17 -0
  88. package/dist/src/runner/scope-isolation-policy.d.ts.map +1 -0
  89. package/dist/src/runner/scope-isolation-policy.js +82 -0
  90. package/dist/src/runner/scope-isolation-policy.js.map +1 -0
  91. package/dist/src/runner/scoped-auto-command.d.ts +10 -1
  92. package/dist/src/runner/scoped-auto-command.d.ts.map +1 -1
  93. package/dist/src/runner/scoped-auto-command.js +90 -55
  94. package/dist/src/runner/scoped-auto-command.js.map +1 -1
  95. package/dist/src/runner/scoped-recovery.d.ts.map +1 -1
  96. package/dist/src/runner/scoped-recovery.js +121 -23
  97. package/dist/src/runner/scoped-recovery.js.map +1 -1
  98. package/dist/src/runner/visual-proof-runner.d.ts +4 -0
  99. package/dist/src/runner/visual-proof-runner.d.ts.map +1 -1
  100. package/dist/src/runner/visual-proof-runner.js +85 -39
  101. package/dist/src/runner/visual-proof-runner.js.map +1 -1
  102. package/dist/src/setup/project-config.d.ts +3 -1
  103. package/dist/src/setup/project-config.d.ts.map +1 -1
  104. package/dist/src/setup/project-config.js +71 -14
  105. package/dist/src/setup/project-config.js.map +1 -1
  106. package/docs/deep-dive.md +43 -4
  107. package/package.json +1 -1
  108. package/prompts/workflows/breakdown-review.md +2 -1
  109. package/prompts/workflows/issue-breakdown.md +31 -0
  110. package/prompts/workflows/issue-tree-orchestration.md +21 -15
  111. package/prompts/workflows/prd.md +11 -0
  112. package/prompts/workflows/scoped-implementation.md +22 -1
  113. package/prompts/workflows/triage.md +11 -0
package/docs/deep-dive.md CHANGED
@@ -242,6 +242,36 @@ text or changed paths indicate UI, API, worker, CLI, or smoke-verifiable work.
242
242
  `reviewGates.acceptanceProof` is canonical; `reviewGates.visualProof` remains a
243
243
  configuration migration adapter for existing screenshot and mobile proof policy.
244
244
 
245
+ The risk-routing gate checks declared review metadata instead of inferring risk
246
+ with an LLM. `reviewGates.riskRouting` defaults to enabled `warn` mode so older
247
+ repositories surface findings without unexpected publication blocks. In
248
+ `block` mode, the same findings become publication blockers.
249
+
250
+ For scoped runs, risk routing checks the completion report `reviewHandoff`:
251
+
252
+ - required handoff fields must be present and non-empty;
253
+ - low-risk claims must use an allowed low-risk flow;
254
+ - configured `riskyChangedPathGlobs` can flag low-risk claims that changed
255
+ risky paths;
256
+ - high-risk claims require passed code-review validation when
257
+ `highRiskRequiresCodeReview` is true.
258
+
259
+ Low-risk routing never weakens existing quality, acceptance proof, visual proof,
260
+ deny, or configured-check gates. It can only add warnings or blockers.
261
+ Scoped risk-routing blockers are retryable only when
262
+ `loopPolicy.rework.retryableBlockers` explicitly includes
263
+ `risk-routing-policy`.
264
+
265
+ For parent `agent:plan-auto` runs, risk routing checks the planning report after
266
+ it is read and before parent content or child issues are mutated. Parent
267
+ `sizeRisk` must partition every child stable id exactly once across
268
+ `small`, `medium`, and `high`; `parentReviewHandoff` must include risks, proof
269
+ strategy, and human review focus. In warn mode, findings render under
270
+ `Risk routing warnings` in the parent PR body and review report while execution
271
+ continues. In block mode, the parent stops at that point and does not create
272
+ child issues, execute children, create a draft PR, or attempt parent planning
273
+ rework.
274
+
245
275
  ## Acceptance Proof
246
276
 
247
277
  Acceptance proof is intentionally runner-owned. Codex can implement product
@@ -373,7 +403,11 @@ Setup now uses the package-owned
373
403
  `codex-orchestrator visual-proof auto --issue ${issueNumber}` command. Auto
374
404
  dispatch uses one shared policy owner: web/frontend paths route to browser
375
405
  proof, while Android, iOS, Flutter, and mobile app paths remain device-backed.
376
- Explicit legacy proof command overrides are preserved.
406
+ When acceptance proof is required but the changed paths are backend/API/CLI-only
407
+ and visual proof is not desirable, auto proof does not force a browser or mobile
408
+ target; the runner evaluates the prepared machine-readable
409
+ `acceptance-proof-report.json` and its non-visual artifacts instead. Explicit
410
+ legacy proof command overrides are preserved.
377
411
 
378
412
  For web UI work, `codex-orchestrator visual-proof browser` reads a proof-owned
379
413
  browser scenario, drives Playwright Core against an explicit base URL, and
@@ -520,8 +554,13 @@ Recoverable runs are classified with explicit states:
520
554
  - `completed-pending-handoff` means the completed report, worktree, branch, and
521
555
  base evidence are sufficient to retry runner-owned publication.
522
556
  - `failed-pending-block` means the stale runner-owned run cannot satisfy
523
- publication preconditions and should be moved to `agent:blocked` with
524
- concrete evidence.
557
+ publication preconditions and should be moved to `agent:blocked` with concrete
558
+ evidence. Missing completion reports get one bounded recovery retry first when
559
+ `missing-completion-report` is configured as retryable, the stale attempt has
560
+ remaining rework budget, same-host stale ownership is proven, and the branch
561
+ has no committed, staged, unstaged, or untracked changes since the recovered
562
+ base SHA. If any of those checks fail, recovery blocks instead of resetting or
563
+ building on unreported work.
525
564
  - `unknown-or-foreign` means ownership or safety cannot be proven, so the runner
526
565
  does not mutate GitHub.
527
566
 
@@ -566,7 +605,7 @@ The top-level config areas are:
566
605
  - `project` for config and prompt directories;
567
606
  - `workflows` for prompt or skill routing;
568
607
  - `checks` and `checksPolicy` for validation commands;
569
- - `reviewGates` for quality and acceptance proof requirements;
608
+ - `reviewGates` for quality, risk-routing, and acceptance proof requirements;
570
609
  - `loopPolicy` for issue selection, rework, review, summaries, and suggestions;
571
610
  - `deny` for secret and unsafe-action protection;
572
611
  - `branches` for branch templates;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "codex-orchestrator",
3
- "version": "0.1.36",
3
+ "version": "0.1.39",
4
4
  "description": "Reusable GitHub Issues runner for Codex.",
5
5
  "type": "module",
6
6
  "license": "MIT",
@@ -24,6 +24,7 @@ Use or request:
24
24
  - **Source-of-truth risk:** flag duplicated business rules, competing ownership, or repeated normalization, dispatch, persistence, or compatibility logic across issues.
25
25
  - **Parallelization risk:** flag issues likely to collide if assigned to agents in the same wave.
26
26
  - **Spec proportionality:** flag repeated issue-level spec gates when one shared wave-level spec would be safer.
27
+ - **Risk/proof routing:** flag small issues that are over-orchestrated instead of routed to `small-task-implementer`, and risky issues that lack spec gates, review focus, or proof strategy.
27
28
  - **Final coverage:** complex plans should include a final integration/regression slice when individual slices do not prove the full scenario together.
28
29
 
29
30
  ## Decision Rules
@@ -41,7 +42,7 @@ Always answer in this structure:
41
42
  3. `Blockers before publishing` with concrete findings and failure mechanics.
42
43
  4. `Split / Merge / Reorder` with exact issue-level changes.
43
44
  5. `Acceptance Criteria fixes` with criteria that need to become more testable.
44
- 6. `AFK readiness and orchestration risk` with collision risks, spec-gate proportionality, and safer wave order.
45
+ 6. `AFK readiness and orchestration risk` with collision risks, spec-gate proportionality, risk/proof routing, and safer wave order.
45
46
  7. `Hard questions` with only questions that block publication.
46
47
 
47
48
  If there are no blockers, say so explicitly. Avoid generic praise and speculative objections.
@@ -46,6 +46,16 @@ Consider a spec gate when the slice involves:
46
46
  - required browser, mobile, manual, or live smoke validation;
47
47
  - known rejected approaches or unresolved product or technical decisions.
48
48
 
49
+ ## Size / Risk
50
+
51
+ Each child issue must declare the intended implementation path:
52
+
53
+ - `Small / low risk`: use `small-task-implementer`; keep the issue narrow, deterministic, and directly verifiable.
54
+ - `Medium`: use scoped implementation with TDD and configured review gates.
55
+ - `High risk`: require issue-level or wave-level spec gates before implementation.
56
+
57
+ Do not split a small issue into a parent orchestration tree just to satisfy process. Do not mark risky shared-contract work as small to avoid review.
58
+
49
59
  ## Review Gate
50
60
 
51
61
  Before publishing child issues, run an issue-breakdown-review pass over the proposed breakdown. The review must check tracer-bullet quality, dependency correctness, AFK readiness, acceptance criteria, scope control, source-of-truth risk, spec proportionality, and orchestration risk.
@@ -58,6 +68,7 @@ Present the proposed breakdown as a numbered list. For each slice include:
58
68
 
59
69
  - **Title**
60
70
  - **Type**: AFK or HITL
71
+ - **Size / Risk**: small / medium / high, with intended path (`small-task-implementer`, scoped implementation, issue-level spec, or wave-level spec)
61
72
  - **Blocked by**
62
73
  - **User stories covered** when available
63
74
  - **Spec required**: none / issue-level / wave-level, with one short reason
@@ -93,6 +104,16 @@ Reason:
93
104
 
94
105
  - One concise reason, or "None - straightforward implementation with existing local patterns."
95
106
 
107
+ ## Size / risk
108
+
109
+ Size: small / medium / high
110
+
111
+ Intended path: small-task-implementer / scoped implementation / issue-level spec / wave-level spec
112
+
113
+ Reason:
114
+
115
+ - One concise reason tied to ownership, contracts, and validation.
116
+
96
117
  ## External contracts
97
118
 
98
119
  Status: confirmed / needs proof / not applicable
@@ -104,11 +125,21 @@ Status: confirmed / needs proof / not applicable
104
125
 
105
126
  ## Verification
106
127
 
128
+ - Proof Strategy: auto / visual / browser-visual / mobile-visual / non-visual-smoke / none
107
129
  - Automated:
108
130
  - Architecture:
109
131
  - Manual/live:
110
132
  - Required fixtures or files:
111
133
 
134
+ Choose a concrete proof strategy for every AFK issue:
135
+
136
+ - `non-visual-smoke` for backend, CLI, telemetry, analytics, event dispatch, data, logging, or workflow behavior where tests, command output, logs, API responses, or machine-readable artifacts are the proof surface.
137
+ - `browser-visual` for web UI/layout/copy/responsive behavior.
138
+ - `mobile-visual` for Android/iOS/Flutter UI behavior where device-backed launch, screenshot, UI dump, or logs are the proof surface.
139
+ - `visual` when visual proof is required but browser vs mobile should be selected from changed paths.
140
+ - `none` only when normal tests/checks are sufficient and no extra acceptance proof artifacts are required.
141
+ - `auto` when no stronger explicit choice is known.
142
+
112
143
  ## codex-orchestrator metadata
113
144
 
114
145
  Ownership:
@@ -37,7 +37,8 @@ Allowed proof includes test result, smoke result, command output, API or browser
37
37
  4. Build a dependency graph: ready, blocked, final integration/regression.
38
38
  5. Identify protected paths, likely source-of-truth owners, and files that must not be edited concurrently.
39
39
  6. Extract each issue's spec gate metadata.
40
- 7. Mark unresolved external contracts, credentials, live fixtures, or human-only decisions as blocked until proven or cleared.
40
+ 7. Extract each issue's size/risk and intended path (`small-task-implementer`, scoped implementation, issue-level spec, or wave-level spec).
41
+ 8. Mark unresolved external contracts, credentials, live fixtures, or human-only decisions as blocked until proven or cleared.
41
42
 
42
43
  ## Plan Waves
43
44
 
@@ -49,6 +50,7 @@ Create the smallest safe parallel wave first:
49
50
  - keep final cross-cutting regression or cleanup issues for the last wave;
50
51
  - define each wave exit gate: worker reports, integration diff review, combined validation, and blocker reconciliation;
51
52
  - decide whether the wave requires a spec gate.
53
+ - keep small low-risk children on the compact `small-task-implementer` path when their ownership and validation are genuinely narrow.
52
54
 
53
55
  ## Spec Gates
54
56
 
@@ -78,6 +80,7 @@ For each issue in the active wave, assign one worker with a narrow ownership sco
78
80
  - accepted implementation spec reference when a spec gate was run;
79
81
  - repo policy and relevant docs to read;
80
82
  - instruction to use TDD for behavior changes;
83
+ - instruction to use `$small-task-implementer` only for child issues explicitly classified as small/low-risk, and to escalate if hidden risk appears;
81
84
  - exact ownership boundaries;
82
85
  - warning that other agents may be editing the repo;
83
86
  - instruction not to revert or overwrite user or worker changes;
@@ -85,6 +88,7 @@ For each issue in the active wave, assign one worker with a narrow ownership sco
85
88
  - required preconditions and verification commands;
86
89
  - stop conditions;
87
90
  - required final report: changed files, proof per acceptance criterion, tests run, skipped checks, risks, blockers, and unresolved acceptance criteria.
91
+ - required `reviewHandoff`: flow used, risk level, implemented contract, proof by acceptance criterion, review focus, and human review checklist.
88
92
 
89
93
  While workers run, the integrator should do non-overlapping work: read docs, inspect ownership, map integration points, and prepare validation. Do not duplicate a worker's implementation.
90
94
 
@@ -98,14 +102,15 @@ Treat each wave as a hard execution gate:
98
102
  4. Remove duplicated logic, competing source-of-truth changes, and workaround-shaped code.
99
103
  5. Check for architecture drift: shallow modules, one-adapter seams, duplicated rules, or tests coupled to implementation details.
100
104
  6. Reconcile every active child issue as complete, blocked with evidence, or deferred.
101
- 7. Verify TDD evidence for behavior-changing issues.
102
- 8. Run the smallest meaningful combined validation for the wave.
103
- 9. Run repo architecture checks when available and applicable.
104
- 10. Verify implementation stayed inside the accepted spec, or document why a spec amendment was required.
105
- 11. Run cleanup-review and code-review when repo policy or change size requires them.
106
- 12. Fix high-confidence review findings and rerun validation.
107
- 13. Create focused commits for completed child issues, or one documented wave commit when safe separation is impossible.
108
- 14. Update the dependency graph before starting the next wave.
105
+ 7. Verify each child's declared path: small-task-implementer stayed compact, and medium/high-risk work satisfied spec/review/proof expectations.
106
+ 8. Verify TDD evidence for behavior-changing issues.
107
+ 9. Run the smallest meaningful combined validation for the wave.
108
+ 10. Run repo architecture checks when available and applicable.
109
+ 11. Verify implementation stayed inside the accepted spec, or document why a spec amendment was required.
110
+ 12. Run cleanup-review and code-review when repo policy or change size requires them.
111
+ 13. Fix high-confidence review findings and rerun validation.
112
+ 14. Create focused commits for completed child issues, or one documented wave commit when safe separation is impossible.
113
+ 15. Update the dependency graph before starting the next wave.
109
114
 
110
115
  If a worker reports an ambiguous or risky decision, pause and ask the user a targeted question.
111
116
 
@@ -117,12 +122,13 @@ After all child issues are implemented, blocked, or deferred:
117
122
  2. Reconcile all child acceptance criteria, skipped checks, blockers, and out-of-scope protections.
118
123
  3. Run cleanup-review before final code-review when repo policy requires it.
119
124
  4. Fix grounded findings and rerun relevant checks.
120
- 5. Summarize completed issues, verification, skipped checks, and follow-up risks.
121
- 6. Comment on completed child issues with result summaries and verification.
122
- 7. Close completed child issues only after completion evidence is posted.
123
- 8. Comment on the parent with completed children, validation, out-of-scope items preserved, and residual risks.
124
- 9. Open or prepare a pull request when requested or expected by repo workflow.
125
- 10. Stop before any human-only action such as PR approval, merge, product decision, manual validation, missing access, or secrets.
125
+ 5. Write a Parent Risk/Proof Mini-Report summarizing child outcomes, flow used per child, risk levels, proof coverage, skipped checks, and the exact human review focus.
126
+ 6. Summarize completed issues, verification, skipped checks, and follow-up risks.
127
+ 7. Comment on completed child issues with result summaries and verification.
128
+ 8. Close completed child issues only after completion evidence is posted.
129
+ 9. Comment on the parent with completed children, validation, Parent Risk/Proof Mini-Report, out-of-scope items preserved, and residual risks.
130
+ 10. Open or prepare a pull request when requested or expected by repo workflow.
131
+ 11. Stop before any human-only action such as PR approval, merge, product decision, manual validation, missing access, or secrets.
126
132
 
127
133
  ## Stop Conditions
128
134
 
@@ -53,6 +53,16 @@ List testing decisions. Include:
53
53
  - required smoke, browser, mobile, API, or live validation;
54
54
  - cases where a deterministic test seam is missing and must be created or explicitly accepted as risk.
55
55
 
56
+ ### Risk And Proof
57
+
58
+ Classify the initiative so downstream automation can choose the right implementation path:
59
+
60
+ - small low-risk work that should stay on the Small Task Implementer path;
61
+ - medium scoped implementation work that needs TDD and review gates;
62
+ - high-risk contracts that need issue-level or wave-level spec gates before coding.
63
+
64
+ List the proof strategy a maintainer should expect after implementation: tests, smoke checks, artifacts, review focus, and any proof that cannot be automated locally.
65
+
56
66
  ### Out of Scope
57
67
 
58
68
  List adjacent behavior that should not be included in this PRD.
@@ -65,5 +75,6 @@ Record open questions, migration notes, rollout notes, compatibility concerns, a
65
75
 
66
76
  - The PRD must be durable: useful even if files move.
67
77
  - The PRD must be specific enough to break into vertical implementation issues.
78
+ - The PRD must make risk and proof expectations explicit enough that small tasks are not over-orchestrated and risky work is not under-specified.
68
79
  - Do not hide unresolved decisions inside implementation work.
69
80
  - Do not mark work as agent-ready when the PRD still depends on unconfirmed external contracts, credentials, manual design decisions, or ambiguous acceptance criteria.
@@ -12,6 +12,16 @@ Implement one scoped issue or approved implementation spec. Follow the issue, re
12
12
  6. Do not add pass-through modules, one-adapter seams, or test-only helpers unless they improve real locality or leverage.
13
13
  7. Add comments only where they clarify non-obvious behavior.
14
14
 
15
+ ## Task Sizing
16
+
17
+ Before editing, classify the issue:
18
+
19
+ - `Small / low risk`: clear behavior, narrow ownership, no schema/persistence/auth/background/shared-contract change, and a targeted validation path exists. Use the Small Task Implementer path (`$small-task-implementer`): compact contract, smallest edit, targeted proof, explicit escalation if hidden risk appears.
20
+ - `Medium`: behavior-changing runtime work that touches shared flow, multiple files, or meaningful tests, but does not require a parent issue tree. Use this scoped implementation workflow with TDD and the configured review gates.
21
+ - `High risk`: schemas, persistence, queues, retries, idempotency, auth, permissions, billing, caching, external contracts, multi-service work, or broad source-of-truth changes. Use the scoped/spec path only if the issue already provides deterministic contracts; otherwise return `needs-promotion` with evidence.
22
+
23
+ Do not run a heavy plan/spec process for a genuinely small task. Do not force a small task path when the change reveals hidden shared-contract or validation risk.
24
+
15
25
  ## TDD And Behavior Proof
16
26
 
17
27
  For runtime behavior changes, use strict TDD red-to-green:
@@ -44,6 +54,17 @@ Run code-review before completion when the issue or repo policy requires it, or
44
54
 
45
55
  For compact low-risk changes, focused validation plus a clear completion report is enough unless repo policy says otherwise.
46
56
 
57
+ ## Review Handoff
58
+
59
+ For completed work, include `reviewHandoff` in the runner JSON report:
60
+
61
+ - `flowUsed`: `small-task-implementer`, `scoped-implementation`, `spec-implementer`, `issue-tree-child`, or `other`;
62
+ - `riskLevel`: `low`, `medium`, or `high`;
63
+ - `implementedContract`: what behavior or invariant changed;
64
+ - `proofByAcceptanceCriteria`: acceptance criteria mapped to test/smoke/artifact evidence;
65
+ - `reviewFocus`: the exact files, states, contracts, or edge cases a human should inspect;
66
+ - `humanReviewChecklist`: the shortest useful manual review path.
67
+
47
68
  ## Acceptance Proof
48
69
 
49
70
  Prepare runner-owned acceptance proof artifacts when configured. For visual work,
@@ -97,5 +118,5 @@ Do not mark the issue complete until:
97
118
  - validation commands and behavior proof have run, or skipped checks have concrete reasons;
98
119
  - applicable cleanup-review and code-review gates have run;
99
120
  - protected paths stayed untouched and rejected approaches were avoided;
100
- - changed files, validation, skipped checks, residual risks, and blockers are reported;
121
+ - changed files, validation, review handoff, skipped checks, residual risks, and blockers are reported;
101
122
  - the structured completion report is written exactly where the runner requested it.
@@ -79,6 +79,17 @@ Describe what should happen, including edge cases and error conditions.
79
79
  - Interface or contract name - what needs to change and why
80
80
  - Config shape or command - expected behavior
81
81
 
82
+ **Proof strategy:**
83
+ Proof Strategy: auto / visual / browser-visual / mobile-visual / non-visual-smoke / none
84
+
85
+ Choose the explicit strategy from the acceptance criteria:
86
+ - `non-visual-smoke` for backend, CLI, telemetry, analytics, event dispatch, data, or logging work where tests, command output, logs, API responses, or machine-readable artifacts prove behavior better than screenshots.
87
+ - `browser-visual` for web UI/layout/copy/responsive work.
88
+ - `mobile-visual` for Android/iOS/Flutter UI behavior where device-backed launch, screenshot, UI dump, or logs are the proof surface.
89
+ - `visual` when visual proof is required but browser vs mobile should be selected from changed paths.
90
+ - `none` only when normal tests/checks are sufficient and no extra acceptance proof artifacts are required.
91
+ - `auto` when no stronger explicit choice is known.
92
+
82
93
  **Acceptance criteria:**
83
94
  - [ ] Specific, testable criterion
84
95
  - [ ] Specific, testable criterion