nccgs 1.3.0 → 2.0.0-rc.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (180) hide show
  1. package/.claude/agents/nccgs-adversarial-reviewer.md +17 -6
  2. package/.claude/agents/nccgs-engine-programmer.md +2 -2
  3. package/.claude/agents/nccgs-fast-implementer.md +30 -0
  4. package/.claude/agents/nccgs-game-designer.md +26 -0
  5. package/.claude/agents/nccgs-gameplay-programmer.md +3 -4
  6. package/.claude/agents/nccgs-qa-engineer.md +2 -2
  7. package/.claude/agents/nccgs-task-scout.md +24 -0
  8. package/.claude/agents/nccgs-ui-programmer.md +2 -2
  9. package/.claude/agents/nccgs-unity-build-specialist.md +2 -2
  10. package/.claude/agents/nccgs-unity-implementer.md +7 -6
  11. package/.claude/agents/nccgs-unity-systems-specialist.md +2 -2
  12. package/.claude/agents/nccgs-unity-ui-specialist.md +2 -2
  13. package/.claude/agents/nccgs-verification-engineer.md +2 -2
  14. package/.claude/nccgs/MEMORY.md +59 -0
  15. package/.claude/nccgs/THIRD_PARTY_NOTICES.md +16 -0
  16. package/.claude/nccgs/VERSION +1 -1
  17. package/.claude/nccgs/agent-library/nccgs-adversarial-reviewer.md +16 -5
  18. package/.claude/nccgs/agent-library/nccgs-fast-implementer.md +28 -0
  19. package/.claude/nccgs/agent-library/nccgs-game-designer.md +1 -2
  20. package/.claude/nccgs/agent-library/nccgs-gameplay-programmer.md +1 -2
  21. package/.claude/nccgs/agent-library/nccgs-task-scout.md +22 -0
  22. package/.claude/nccgs/agent-library/nccgs-unity-implementer.md +5 -4
  23. package/.claude/nccgs/constitution.md +44 -12
  24. package/.claude/nccgs/hooks/agent-audit.mjs +2 -2
  25. package/.claude/nccgs/hooks/closure-guard.mjs +31 -8
  26. package/.claude/nccgs/hooks/common.mjs +2 -1
  27. package/.claude/nccgs/hooks/runtime-observation.mjs +2 -2
  28. package/.claude/nccgs/hooks/session-start.mjs +11 -4
  29. package/.claude/nccgs/packages/ponytail/LICENSE +21 -0
  30. package/.claude/nccgs/packages/ponytail/NOTICE.md +32 -0
  31. package/.claude/nccgs/packages/ponytail/README.md +46 -0
  32. package/.claude/nccgs/packages/ponytail/complexity-review.md +47 -0
  33. package/.claude/nccgs/packages/ponytail/engineering.md +68 -0
  34. package/.claude/nccgs/packages/ponytail/manifest.json +24 -0
  35. package/.claude/nccgs/pipeline.json +92 -8
  36. package/.claude/nccgs/protocols/agent-contract.md +50 -1
  37. package/.claude/nccgs/protocols/astra-claude.md +46 -12
  38. package/.claude/nccgs/protocols/bounded-repair.md +118 -0
  39. package/.claude/nccgs/protocols/context-packets.md +5 -0
  40. package/.claude/nccgs/protocols/delivery-acceptance.md +105 -0
  41. package/.claude/nccgs/protocols/design-lifecycle.md +139 -0
  42. package/.claude/nccgs/protocols/evidence.md +11 -1
  43. package/.claude/nccgs/protocols/model-routing.md +44 -42
  44. package/.claude/nccgs/protocols/orchestration.md +46 -5
  45. package/.claude/nccgs/protocols/review-handoff.md +65 -0
  46. package/.claude/nccgs/protocols/standalone-gates.md +273 -0
  47. package/.claude/nccgs/protocols/standalone-pilot.md +144 -0
  48. package/.claude/nccgs/protocols/standalone.md +187 -0
  49. package/.claude/nccgs/protocols/task-autonomy.md +55 -0
  50. package/.claude/nccgs/protocols/task-gates.md +73 -7
  51. package/.claude/nccgs/routing.generated.md +25 -7
  52. package/.claude/nccgs/settings.fragment.json +153 -141
  53. package/.claude/nccgs/studio.json +135 -101
  54. package/.claude/nccgs/tools/autonomy.mjs +27 -0
  55. package/.claude/nccgs/tools/command-process.mjs +44 -0
  56. package/.claude/nccgs/tools/compile-policy.mjs +34 -4
  57. package/.claude/nccgs/tools/configure-models.mjs +13 -5
  58. package/.claude/nccgs/tools/criteria.mjs +121 -0
  59. package/.claude/nccgs/tools/delivery-budget.mjs +52 -0
  60. package/.claude/nccgs/tools/design-readiness.mjs +158 -0
  61. package/.claude/nccgs/tools/evidence.mjs +153 -0
  62. package/.claude/nccgs/tools/gates.mjs +151 -24
  63. package/.claude/nccgs/tools/inputs.mjs +73 -0
  64. package/.claude/nccgs/tools/lifecycle.mjs +521 -0
  65. package/.claude/nccgs/tools/native-completion.mjs +21 -0
  66. package/.claude/nccgs/tools/orchestration.mjs +430 -0
  67. package/.claude/nccgs/tools/paths.mjs +60 -0
  68. package/.claude/nccgs/tools/policy.mjs +58 -13
  69. package/.claude/nccgs/tools/process-lock.mjs +74 -0
  70. package/.claude/nccgs/tools/repair-packet.mjs +87 -0
  71. package/.claude/nccgs/tools/review-handoff.mjs +32 -0
  72. package/.claude/nccgs/tools/runtime.mjs +33 -7
  73. package/.claude/nccgs/tools/task.mjs +62 -31
  74. package/.claude/nccgs/tools/tool-guard.mjs +117 -0
  75. package/.claude/nccgs/tools/verification-jobs.mjs +161 -0
  76. package/.claude/nccgs/tools/verification-worker.mjs +71 -0
  77. package/.claude/nccgs/tools/windows-job.cs +96 -0
  78. package/.claude/nccgs/workflow-catalog.json +286 -26
  79. package/.claude/skills/accessibility-review/SKILL.md +12 -5
  80. package/.claude/skills/architecture-decision/SKILL.md +12 -5
  81. package/.claude/skills/asset-audit/SKILL.md +12 -5
  82. package/.claude/skills/audit/SKILL.md +12 -5
  83. package/.claude/skills/balance-review/SKILL.md +12 -5
  84. package/.claude/skills/bug-triage/SKILL.md +12 -5
  85. package/.claude/skills/bug-triage/references/astra.md +12 -5
  86. package/.claude/skills/closure/SKILL.md +12 -5
  87. package/.claude/skills/compatibility-review/SKILL.md +12 -5
  88. package/.claude/skills/context-pack/SKILL.md +12 -5
  89. package/.claude/skills/dependency-review/SKILL.md +12 -5
  90. package/.claude/skills/design/SKILL.md +14 -5
  91. package/.claude/skills/design-review/SKILL.md +14 -5
  92. package/.claude/skills/evidence-review/SKILL.md +12 -5
  93. package/.claude/skills/hotfix/SKILL.md +12 -5
  94. package/.claude/skills/hotfix/references/astra.md +12 -5
  95. package/.claude/skills/incident-recovery/SKILL.md +12 -5
  96. package/.claude/skills/incident-recovery/references/astra.md +12 -5
  97. package/.claude/skills/localize-game/SKILL.md +12 -5
  98. package/.claude/skills/localize-game/references/astra.md +12 -5
  99. package/.claude/skills/migrate-project/SKILL.md +12 -5
  100. package/.claude/skills/migrate-project/references/astra.md +12 -5
  101. package/.claude/skills/migrate-project/references/procedure.md +2 -1
  102. package/.claude/skills/milestone-review/SKILL.md +12 -5
  103. package/.claude/skills/nccgs-code-review/SKILL.md +12 -5
  104. package/.claude/skills/nccgs-code-review/references/astra.md +8 -2
  105. package/.claude/skills/performance-audit/SKILL.md +12 -5
  106. package/.claude/skills/plan-feature/SKILL.md +12 -5
  107. package/.claude/skills/playtest/SKILL.md +12 -5
  108. package/.claude/skills/playtest/references/astra.md +16 -5
  109. package/.claude/skills/project-stage/SKILL.md +12 -5
  110. package/.claude/skills/prototype-feature/SKILL.md +12 -5
  111. package/.claude/skills/prototype-feature/references/astra.md +16 -5
  112. package/.claude/skills/qa-plan/SKILL.md +12 -5
  113. package/.claude/skills/release/SKILL.md +12 -5
  114. package/.claude/skills/release/references/astra.md +12 -5
  115. package/.claude/skills/release-readiness/SKILL.md +12 -5
  116. package/.claude/skills/retrospective/SKILL.md +12 -5
  117. package/.claude/skills/review/SKILL.md +12 -5
  118. package/.claude/skills/review/references/astra.md +8 -2
  119. package/.claude/skills/security-audit/SKILL.md +12 -5
  120. package/.claude/skills/sprint-plan/SKILL.md +12 -5
  121. package/.claude/skills/status/SKILL.md +12 -5
  122. package/.claude/skills/status/references/astra.md +6 -0
  123. package/.claude/skills/story-readiness/SKILL.md +14 -5
  124. package/.claude/skills/test/SKILL.md +12 -5
  125. package/.claude/skills/test/references/astra.md +12 -5
  126. package/.claude/skills/ui-review/SKILL.md +12 -5
  127. package/.claude/skills/work/SKILL.md +12 -5
  128. package/.claude/skills/work/references/astra.md +17 -5
  129. package/CHANGELOG.md +42 -0
  130. package/CLAUDE.md +2 -0
  131. package/MEMORY.md +55 -0
  132. package/README.md +132 -180
  133. package/THIRD_PARTY_NOTICES.md +15 -0
  134. package/UPGRADING.md +91 -69
  135. package/VERSION +1 -1
  136. package/docs/ARCHITECTURE.md +19 -5
  137. package/docs/BOUNDED-REPAIR.md +11 -0
  138. package/docs/HUONG-DAN-MIGRATE-VA-SU-DUNG.md +46 -34
  139. package/docs/NCCGS-1.3.md +10 -122
  140. package/docs/NCCGS-1.4.1.md +149 -0
  141. package/docs/NCCGS-1.4.md +184 -0
  142. package/docs/NCCGS-1.5-IMPLEMENTATION-PLAN.md +378 -0
  143. package/docs/NCCGS-1.5-PILOT.md +130 -0
  144. package/docs/NCCGS-1.5.md +237 -0
  145. package/docs/PHASE-1-OPERATIONS.md +137 -0
  146. package/docs/PHASE-2-OPERATIONS.md +183 -0
  147. package/docs/PROJECT-POLICY.md +21 -9
  148. package/docs/RELEASE-STATUS.md +42 -0
  149. package/docs/WORKFLOWS.md +42 -24
  150. package/package.json +7 -3
  151. package/scaffold/.nccgs/autonomy.json +4 -0
  152. package/scaffold/.nccgs/execution.json +10 -0
  153. package/scaffold/.nccgs/inputs.json +6 -0
  154. package/scaffold/.nccgs/project.yaml +4 -2
  155. package/scaffold/.nccgs/templates/agent-handoff.md +2 -0
  156. package/scaffold/.nccgs/templates/design-contract.json +34 -0
  157. package/scaffold/.nccgs/templates/feature-contract.md +21 -5
  158. package/scaffold/.nccgs/templates/implementation-report.md +4 -1
  159. package/scaffold/.nccgs/templates/repair-packet.json +14 -0
  160. package/scripts/benchmark-snapshot.mjs +205 -0
  161. package/scripts/benchmark-v15.mjs +173 -0
  162. package/scripts/cli.mjs +129 -2
  163. package/scripts/doctor.mjs +128 -22
  164. package/scripts/install.mjs +96 -34
  165. package/scripts/run.mjs +72 -0
  166. package/scripts/supervise.mjs +124 -0
  167. package/scripts/validate.mjs +45 -4
  168. package/tests/autonomy.test.mjs +113 -0
  169. package/tests/benchmark-v15.test.mjs +65 -0
  170. package/tests/design-readiness.test.mjs +200 -0
  171. package/tests/evidence.test.mjs +186 -0
  172. package/tests/framework.test.mjs +333 -16
  173. package/tests/gates.test.mjs +313 -4
  174. package/tests/lifecycle-v15.test.mjs +300 -0
  175. package/tests/orchestration-v15.test.mjs +545 -0
  176. package/tests/phase1.test.mjs +324 -0
  177. package/tests/phase2.test.mjs +250 -0
  178. package/tests/policy-v15.test.mjs +220 -0
  179. package/tests/ponytail-package.test.mjs +75 -0
  180. package/tests/runtime-v15.test.mjs +624 -0
@@ -1,6 +1,6 @@
1
1
  # Astra planning, Claude implementation, Fable review
2
2
 
3
- This protocol is mandatory when `models.profile` is `astra-claude`, including when
3
+ This historical protocol is mandatory for compatibility execution when `models.profile` is `astra-claude`, including when
4
4
  the package default is used before project policy exists. It takes precedence over
5
5
  generic NCCGS workflow instructions. Other profiles retain their existing workflow.
6
6
 
@@ -13,8 +13,10 @@ status labels, determine whether reporting or DONE is allowed.
13
13
 
14
14
  - **Astra** owns planning, feature contracts, context packets, architecture decisions,
15
15
  canonical documents, and the final check of Claude's report.
16
- - **Claude Opus 5.5 / xhigh** implements the supplied plan, writes tests, runs relevant
17
- checks, fixes defects, and records implementation evidence and the handoff report.
16
+ - **Claude's selected implementer** implements the supplied plan, writes tests,
17
+ runs relevant checks, fixes defects, and records evidence and the handoff report.
18
+ FAST uses Sonnet/medium, STANDARD uses Opus 5.5/medium, and CONTROLLED uses
19
+ Opus 5.5/high, as resolved from `pipeline.json`.
18
20
  - **Fable / high**, through `nccgs-adversarial-reviewer`, independently reviews the
19
21
  completed work. The reviewer remains read-only and does not implement fixes.
20
22
  - **Product Owner** performs final acceptance after Astra's check. Only this explicit
@@ -31,18 +33,27 @@ remain available as technical consultants, not substitutes for Astra ownership.
31
33
  1. Read Astra's plan, requirements, acceptance criteria, document references, and
32
34
  assigned write set. A concise Astra handoff is enough for FAST work. Missing or
33
35
  contradictory scope is a planning blocker; return the exact question to Astra.
34
- 2. Confirm the implementation session or agent actually uses `claude-opus-5-5` at
35
- `xhigh`. Implement and collect compilation, test, runtime, and regression evidence
36
- appropriate to the change. A role name alone does not prove its runtime model.
37
- 3. Immediately after implementation and its checks, invoke a fresh, independent
38
- `nccgs-adversarial-reviewer` with the review route in pipeline.json. Supply raw requirements, the
36
+ 2. Initialize the task once with `--risk FAST|STANDARD|CONTROLLED` (default STANDARD).
37
+ Resume that same task for remediation; never create a new ID to reset its budget.
38
+ Confirm the implementation session or agent actually uses the task's recorded
39
+ implementation route and effort. Implement and collect compilation, test,
40
+ runtime, and regression evidence appropriate to the change. A role name alone
41
+ does not prove its runtime model.
42
+ 3. After recording implementation and its checks, run
43
+ `task check --id ID --target review`. Only when allowed, invoke a fresh,
44
+ independent `nccgs-adversarial-reviewer` with the review route in pipeline.json.
45
+ The bound-task pre-agent guard also checks readiness and remaining rounds.
46
+ Keep only one Fable review in flight per task; the guard does not reserve rounds
47
+ or meter invocations, tokens, or cost.
48
+ Supply raw requirements, the
39
49
  final diff/working-tree snapshot, affected callers, and evidence. Do not supply
40
50
  an expected verdict. No final completion report is handed to Astra before this
41
51
  review. Progress updates and blocker notices are always allowed.
42
52
  4. Fable returns PASS, CHANGES_REQUIRED, or BLOCKED with evidence and findings.
43
53
  Open BLOCKER/HIGH findings or unmet required acceptance criteria prevent PASS.
44
- Opus fixes findings within approved scope, reruns affected checks, then requests
45
- Fable review again. Any code/test/asset change after a review invalidates that
54
+ The selected implementer fixes actionable findings within approved scope, reruns
55
+ affected checks, then requests Fable review within the remaining review budget.
56
+ Any code/test/asset change after a review invalidates that
46
57
  review. Bind each review to a commit plus dirty/untracked diff or file hashes;
47
58
  a commit ID alone is insufficient when the working tree is dirty.
48
59
  5. After Fable PASS on the final snapshot, Claude writes the implementation report
@@ -52,7 +63,8 @@ remain available as technical consultants, not substitutes for Astra ownership.
52
63
  a real delivery mechanism. Do not simulate Astra's response.
53
64
  6. Astra checks plan compliance, evidence, Fable findings and their resolution,
54
65
  documents, and residual risk. Astra may return work for correction, which repeats
55
- steps 2-5, or record its check and prepare `AWAITING_PO_ACCEPTANCE`.
66
+ implementation/check/review/report in steps 2-5 on the same task and remaining
67
+ budget, or record its check and prepare `AWAITING_PO_ACCEPTANCE`.
56
68
  7. The Product Owner accepts the reviewed snapshot. Only then may Astra finalize the
57
69
  closure record as DONE if all other configured gates also pass. Unknown, missing,
58
70
  stale, or inferred sign-offs never count. Changes after either sign-off require
@@ -63,9 +75,31 @@ VALIDATED, DONE, and BLOCKED. Neither FAST classification nor lean closure skips
63
75
  Fable review, Astra checking, or explicit Product Owner acceptance. Hotfixes,
64
76
  prototypes, migrations with code changes, and release fixes follow the same chain.
65
77
 
78
+ ## Bounded execution and stopping
79
+
80
+ Use one accountable implementer. Add the read-only task scout only for a specific
81
+ missing fact, and add specialist reviewers only for distinct risks. Stay within
82
+ approved scope; do not turn speculative improvements into requirements or broaden
83
+ a completed change. FAST keeps the full acceptance chain with a concise plan and
84
+ report.
85
+
86
+ Stop implementation and analysis when approved criteria and relevant tests pass
87
+ and no actionable blocking findings remain. Re-run checks only for changed inputs,
88
+ failures, or new evidence. For subjective uncertainty, prepare a playable test with
89
+ observable questions for the Product Owner instead of repeating speculative debate.
90
+
91
+ Task gates enforce review-round limits: FAST 2, STANDARD 3, CONTROLLED 3. A valid
92
+ recorded PASS, CHANGES_REQUIRED, or BLOCKED consumes one round. Re-recording
93
+ implementation does not reset the counter. A PASS on the last allowed round can
94
+ continue to reporting; an exhausted budget without a usable PASS stops work with
95
+ the evidence, unresolved findings, and next decision. It never becomes auto-PASS.
96
+ An external operator may explicitly extend the budget with a reason using
97
+ `task extend-review`; see [task gates](task-gates.md). Context and elapsed-time
98
+ budgets guide scope and escalation; they are advisory, not enforced runtime caps.
99
+
66
100
  ## Model failures and evidence
67
101
 
68
- Do not silently substitute Sonnet, another Opus version, or inherited models for
102
+ Do not silently substitute a different model, effort, or inherited routing for
69
103
  the required stages. If access, forced routing, effort caps, or runtime fallback
70
104
  prevents the requested model/effort, record BLOCKED with the observed identity and
71
105
  resume only when the required stage can run or the Product Owner changes policy.
@@ -0,0 +1,118 @@
1
+ # Bounded source repair — 2.0 RC
2
+
3
+ The Unity phase-2 retest exhausted a repair worker's 22 native turns while exploring
4
+ parameters in a scratch harness. It made no gameplay change and left a PARTIAL
5
+ handoff. A protected dispatch slot alone did not reserve time to deliver an edit.
6
+ Fresh repair reservations now require a concrete source-edit packet and durable
7
+ tool admission limits. Existing failed cases and legacy reservations stay unchanged.
8
+
9
+ ## Prepare the decision before spending the repair slot
10
+
11
+ The Claude coordinator first uses the recorded review or measured failure to choose
12
+ one correction. It must preserve the approved requirements and measurement rules.
13
+ If diagnosis is unresolved, use eligible read-only analysis within the existing
14
+ budget or report PARTIAL; do not spend the repair slot on another parameter search.
15
+ General task autonomy does not expand an individual worker's assigned scope.
16
+
17
+ The coordinator applies [Ponytail engineering guidance](../packages/ponytail/engineering.md)
18
+ to this diagnosis and solution choice: trace relevant callers, reuse suitable
19
+ existing behavior, and preserve every requirement and check. Include the material
20
+ rationale and limits in the existing packet fields. The worker does not spend
21
+ additional reads loading that package or reopen exploration to optimize diff size.
22
+
23
+ Copy `.nccgs/templates/repair-packet.json` into
24
+ `.nccgs/evidence/work-packets/UNIQUE.json` and replace every placeholder:
25
+
26
+ - `taskId`, `contractHash`, `sourceSnapshot`: current task and snapshot from the CLI.
27
+ - `problem`, `solution`: concise failure and already-chosen correction.
28
+ - `changes`: 1–4 exact existing source files, each with a concrete instruction.
29
+ Directory ownership, new files, governance and evidence edits are rejected.
30
+ - `context`: up to four existing `.nccgs/evidence/` artifact references produced by
31
+ `task ref`. Supply only relevant measured facts and rationale, not entire logs.
32
+ - `verification`: all required check names from the contract. At least one is
33
+ required. The coordinator runs them after the worker returns.
34
+ - `budget`: optional lower limits. Defaults and ceilings are 12 work tool calls,
35
+ including 6 reads/searches, 3 separate handoff writes, and 600000 ms from claim.
36
+ The packet is at most 16 KiB. Limits cannot be raised in the request.
37
+
38
+ The reservation must contain the packet path, exactly matching writePaths, and one
39
+ fresh handoff path. It still needs recorded negative feedback and eligible remaining
40
+ repair capacity within the original aggregate budget:
41
+
42
+ ```json
43
+ {
44
+ "requestKey": "repair-1",
45
+ "kind": "implementation",
46
+ "purpose": "repair",
47
+ "agent": "nccgs-gameplay-programmer",
48
+ "reason": "Apply the chosen correction from the recorded negative review",
49
+ "workPacket": ".nccgs/evidence/work-packets/repair-1.json",
50
+ "writePaths": ["Assets/ChosenFile.cs"],
51
+ "handoffPaths": [".nccgs/evidence/analysis/repair-1.md"]
52
+ }
53
+ ```
54
+
55
+ ```text
56
+ node .claude/nccgs/tools/task.mjs reserve --id TASK --request .nccgs/evidence/requests/repair-1.json
57
+ node .claude/nccgs/tools/task.mjs dispatch-prompt --id TASK --dispatch RETURNED_ID
58
+ ```
59
+
60
+ Use the returned `prompt` for the foreground Agent call. The pre-tool hook replaces
61
+ that call's prompt with the same pinned packet so the worker receives the chosen
62
+ solution, exact files, limits and handoff without first searching for them. Model
63
+ route, risk-selected effort and native turn limits remain unchanged. Packet/context
64
+ hashes are checked before launch, work calls, completion and stage validation.
65
+ Direct writes to pinned packet/context evidence are refused even after completion.
66
+
67
+ ## Execute and hand off
68
+
69
+ The worker may Read/Grep/Glob inside the project and Write/Edit/MultiEdit the exact
70
+ source files. The assigned handoff is the only evidence file it can edit. Shells,
71
+ MCP, web tools, nested agents and other tools are refused for this repair protocol.
72
+ Editor/asset operations requiring Unity tools need a different properly scoped
73
+ assignment; do not relabel repair as delivery to evade these limits.
74
+
75
+ Write `NCCGS-HANDOFF: PARTIAL` early. Apply the chosen correction, then replace that
76
+ line with `NCCGS-HANDOFF: COMPLETE` only when the assigned edits are finished.
77
+ Include changed files, rationale, remaining unverified checks, risks and next owner.
78
+ COMPLETE means the edit assignment is finished; it does not mean product PASS.
79
+ If the fix needs more investigation, preserve PARTIAL and return promptly.
80
+
81
+ Every correlated tool attempt reaching the ledger consumes its bucket, including
82
+ scope denials. Identical hook redelivery with the same tool-use ID and input is
83
+ idempotent. A changed payload under that ID is rejected. The ledger stores a payload
84
+ hash and counts, not the raw command or tool content. Atomic lock protection keeps
85
+ parallel hooks from both taking the last call. Lock contention fails closed; retry
86
+ the same ID after the other owner finishes. It is not an admitted/spent call until
87
+ the transaction can record it. Missing/corrupt accounting is never reset.
88
+
89
+ Reads count within the work allowance. Exhausting reads leaves remaining edit calls;
90
+ exhausting work or reaching the deadline still permits remaining handoff writes.
91
+ The deadline is checked at tool admission and cannot interrupt a model request or
92
+ tool already running. These counts are **tool calls, not native turns or token
93
+ limits**; they cannot guarantee three native turns remain. Missing/partial handoffs
94
+ still fail the dispatch, and quota is never refunded. Counter exhaustion does not
95
+ complete, release, resume or replace an active worker.
96
+
97
+ The coordinator then runs declared checks with `task verify`, collects their original
98
+ receipts, records the work stage and dispatches protected QA and independent review
99
+ when eligible. Failing measurements still block PASS and DONE. No helper exit code
100
+ or handoff marker substitutes for measured acceptance.
101
+
102
+ ## Validation and rollout boundary
103
+
104
+ Apply only to a fresh authorized install/task. Updating a historical reservation
105
+ does not retroactively enforce a packet or rewrite its result; idempotent requests
106
+ return the original reservation. Old limits, attempts, evidence and failures persist.
107
+ Do not upgrade/resume frozen game cases merely to obtain a passing result.
108
+
109
+ Local tests exercise executable hooks, concurrency, exact ownership, stale context,
110
+ separate handoff allocation and a complete synthetic repair/QA/re-review cycle under
111
+ the original cap. They establish framework behavior, not actual Claude performance,
112
+ Unity gameplay viability, provider usage or token savings. A new real Claude/Unity
113
+ acceptance run is still required. NCCGS is a workflow guard, not an OS sandbox or
114
+ protection against malicious code, falsified measurements or externally edited ledgers.
115
+
116
+ Native hook output uses `PreToolUse.hookSpecificOutput.updatedInput` for the packet
117
+ and `additionalContext` for remaining counts, without granting native permissions.
118
+ See the [official Claude Code hooks reference](https://code.claude.com/docs/en/hooks#pretooluse-decision-control).
@@ -11,3 +11,8 @@ source revisions, and contradictions.
11
11
  Prefer anchors, IDs, paths, and hashes over pasted prose. Update the packet when its
12
12
  sources change. Read a full source only when the packet records why its summary is
13
13
  insufficient.
14
+
15
+ Keep context proportional to the approved risk tier. Context and elapsed-time
16
+ budgets are advisory: crossing one is a reason to narrow the next read or surface
17
+ the exact missing fact, not proof of failure or permission to discard evidence.
18
+ They do not reset or extend the separately enforced task review-round limit.
@@ -0,0 +1,105 @@
1
+ # Delivery acceptance and separate product research
2
+
3
+ Owner decision, 2026-10-04: evaluate NCCGS on supporting game development from
4
+ brief through implementation, testing and acceptance. Balance is researched
5
+ separately for each product. Apply this policy when authoring new or explicitly
6
+ revised criteria; preserve frozen contracts, measurements, verdicts and spent budgets.
7
+
8
+ ## Define what is being accepted
9
+
10
+ For each brief item, identify its source and choose one of these scopes in the
11
+ design handoff's prose and requirement mapping. These labels are guidance, not
12
+ new JSON fields or additional machine verdicts.
13
+
14
+ - **Delivery obligation:** specified behavior, formulas/configuration, UI/content,
15
+ integration, compatibility, build, tests and required delivery evidence. Map to
16
+ executable acceptance or a scoped owner decision where objectively necessary.
17
+ - **Technical diagnostic:** a probe used to locate defects, cover a state or measure
18
+ load. State setup, interventions and limitations. Do not promote it to production
19
+ evidence or an unrelated mandatory gate.
20
+ - **Product research:** difficulty, fairness, enjoyment, pacing preferences, build
21
+ viability, win rate, strategy strength and choosing tuning targets. Record in a
22
+ separate product research brief with its own questions, methods and owner. Do
23
+ not import those questions as mandatory delivery criteria or default manual gates.
24
+
25
+ Implementing an approved damage formula, drop probability, spawn schedule or tuning
26
+ value is delivery work. Deciding whether that value feels fair or produces a desired
27
+ win rate is research. Report suspected balance issues without silently changing the
28
+ approved numbers. If a required value is genuinely unspecified, obtain its authority
29
+ before dependent implementation; continue independent work. Never discard a stated
30
+ behavior, accessibility requirement or technical performance budget as "balance".
31
+
32
+ NCCGS can support a separately requested `/balance-review` or `/playtest`; judge its
33
+ research deliverable against that research brief. Such a task does not make a bot's
34
+ victory or a positive balance finding a universal condition for studio acceptance.
35
+
36
+ ## Delivery review checklist
37
+
38
+ 1. **Brief fidelity:** each required item has a stable ID, actual source, implementable
39
+ rule and test/observation; ambiguities and scope changes are recorded with authority.
40
+ 2. **Implementation:** trace the item to changed code/assets and actual behavior.
41
+ Preserve unrelated functionality. A build or test count alone proves no coverage.
42
+ 3. **Testing:** cover the relevant normal, boundary and failure paths; make expected
43
+ outcomes independent of implementation self-report. A correct loss/death is a
44
+ passing test when the requirement is correct loss handling.
45
+ 4. **Evidence integrity:** bind measurements to the source/build, environment, inputs
46
+ and attempt; reject stale/missing/foreign data. Exercise negative controls for a
47
+ helper's consequential pass logic. Keep actual failures and intervention records.
48
+ 5. **Integration and delivery:** build the specified target, exercise required input/UI
49
+ and system interactions, package reproduction steps and disclose remaining gaps.
50
+ 6. **Independent review and repair:** compare brief, code and evidence on a matching
51
+ snapshot. Fix blocking findings and repeat affected checks. Do not invent a finding
52
+ to consume a repair slot or call every successful invocation completed work.
53
+ 7. **Coordination:** complete useful bounded packets under cumulative resource limits,
54
+ own and settle command processes, reuse valid evidence, and report any required
55
+ operator assistance. A missing measurement is neither zero usage nor permission
56
+ to renew a budget. A partial repair is useful when honestly labelled; inability
57
+ to fit an entire playtest does not by itself forbid an authorized small code fix.
58
+ 8. **Honest acceptance:** report each delivery criterion's evidence, review, limitations
59
+ and scoped owner decisions. Delivery acceptance does not claim enjoyable/balanced
60
+ gameplay or general productivity superiority. Record research status separately.
61
+
62
+ The generic engine still enforces every criterion in the imported contract. This
63
+ checklist guides the designer/reviewer; it does not silently filter old criteria,
64
+ grant manual approval, or convert old FAIL/UNVERIFIED into PASS.
65
+
66
+ ## Choose tests that measure the requirement
67
+
68
+ Use short deterministic scenarios for state transitions, damage/formulas, schedules,
69
+ exactly-once rewards, death precedence, pause, reset and UI flows. Establish a known
70
+ precondition, apply the normal operation under test, and observe the result. A fixture
71
+ may set health, place entities, select a reference build or advance simulation time
72
+ when necessary. Disclose those interventions; verify shipped defaults separately.
73
+ The fixture must not directly set the result it claims to test (e.g. injecting a
74
+ Victory state cannot prove that the normal victory condition works).
75
+
76
+ Use an actual player build for integration/launch/input/platform behavior. Choose
77
+ duration and workload to expose the specific risk. Neither a universal 600-second
78
+ run nor bot survival/victory is a default delivery requirement. A real-time soak,
79
+ performance threshold or human UI inspection may be mandatory when the delivery
80
+ brief explicitly requires it; explain the authority, workload and failure rule.
81
+ Keep unrelated implementation work progressing if it does not depend on that result.
82
+ Never relabel an assisted fixture as an unassisted natural play session.
83
+
84
+ Bot losses, seeds and actions are diagnostics unless a separately scoped requirement
85
+ is the bot itself. Separate a product defect, test/collector defect, bot limitation,
86
+ and research question before assigning repair work. Retry or change input only to
87
+ test a stated hypothesis; do not repeatedly play until an unexplained PASS appears.
88
+
89
+ ## Early checkpoint and revision
90
+
91
+ Use `checkpoint: "early"` only for a small delivery prerequisite whose failure
92
+ actually prevents dependent work: for example build/start plus one core input/state
93
+ path. State the dependency and fit it into the first bounded implementation slice.
94
+ Do not select bot win rate, survival time, enjoyment, balance or damage headroom as
95
+ a universal expansion gate. The current engine blocks all further delivery dispatches
96
+ for an unmet early criterion; therefore do not mark a late integration/soak requirement
97
+ early when independent features can still be implemented. Ordinary final criteria
98
+ remain mandatory for final delivery acceptance.
99
+
100
+ For a previously pinned contract, write an explicit old-to-new criterion comparison
101
+ against the owner's scope decision, preserve required game behavior and the old
102
+ evidence, then follow design validation/comparison and independent review before
103
+ importing a revised executable handoff. Preserve spent attempts and cumulative limits.
104
+ Do not edit `.nccgs/tasks`, receipts or a frozen design to make this policy appear
105
+ retroactively satisfied. A changed acceptance scope is not a new budget authorization.
@@ -0,0 +1,139 @@
1
+ # Standalone design, independent document review and readiness
2
+
3
+ Use with [standalone](standalone.md) for `/design`, `/design-review` and
4
+ `/story-readiness`. Claude Code owns these responsibilities. Ponytail complexity
5
+ guidance is not their review standard. Schema validation does not judge game design.
6
+ Apply [delivery acceptance](delivery-acceptance.md): NCCGS delivery is judged on
7
+ brief fidelity, implementation, testing, review and handoff. Balance/viability/feel
8
+ research belongs to a separately scoped product study, not default delivery gates.
9
+
10
+ ## /design — Game Designer responsibility
11
+
12
+ 1. Read the brief, canonical GDD sections, recorded owner decisions and relevant
13
+ existing behavior. Identify concrete ambiguities, contradictions and missing
14
+ rules. Do not assume that a numeric GDD is necessarily complete.
15
+ 2. Define the assigned feature's states, rules/formulas, player feedback, boundaries,
16
+ interactions, failure/edge cases, dependencies and out-of-scope behavior. Give
17
+ each requirement a stable ID and a source section/excerpt with its actual hash.
18
+ 3. Separate fixed behavior from delegated tuning. Record tuning parameters, bounds
19
+ and the source granting authority. Resolve routine choices within that authority.
20
+ Batch only meaningful unresolved product decisions for the owner; continue
21
+ independent work. An invented assumption is not an owner decision.
22
+ 4. Map every requirement to concrete acceptance: procedure, environment, pass
23
+ condition, rationale and authority for that threshold. Automatic criteria need
24
+ actual measurement assertions. Keep product balance/experience research outside
25
+ the delivery contract and list its disposition in the handoff/outOfScope; do not
26
+ silently omit a required behavior. Manual delivery gates require a specific
27
+ brief obligation and named decision, not a default "fun/viability" gate. Keep
28
+ diagnostics separate, with their limitations. "Two viable builds" is an unresolved
29
+ product-research question unless the owner supplies a scoped definition; it does
30
+ not imply every bot/seed wins, and random offers do not imply early access.
31
+ Equal seeds alone do not establish equal action traces or timestep schedules.
32
+ 5. Produce one bounded contract using the installed
33
+ `.nccgs/templates/design-contract.json` schema. The JSON is a versioned handoff
34
+ referencing canonical truth, not a replacement GDD tree. Supporting explanatory
35
+ prose belongs in its declared handoff. Preserve open decisions and return
36
+ BLOCKED when they affect implementation. READY_WITH_RISKS cannot hide an open
37
+ product decision. Human product acceptance can still be pending on a READY design.
38
+
39
+ The coordinator initializes a `mode: "plan", workflow: "design"` task with
40
+ `requiredChecks: ["design-contract"]`, `acceptance: []`, the canonical `sources`,
41
+ and `analysisArtifacts` listing the output JSON and each worker's fresh handoff.
42
+ The plan task's acceptance is about delivering/reviewing the document; do not copy
43
+ the future game's executable criteria into the plan task. Reserve a designer as
44
+ `kind: "analysis"`, normally `nccgs-game-designer`; use technical analysis only for
45
+ a concrete feasibility question. The worker writes only declared analysis outputs.
46
+ `handoffPaths` should name a separate Markdown handoff with the required marker;
47
+ the JSON contract itself is not a handoff-marker file.
48
+
49
+ After the designer returns, the coordinator executes the fixed read-only validator:
50
+
51
+ ```text
52
+ node .claude/nccgs/tools/task.mjs design-validate --id DESIGN --file .nccgs/evidence/analysis/DESIGN-contract.json
53
+ ```
54
+
55
+ Use its returned `evidence` as the `design-contract` check and the contract JSON's
56
+ `task ref` as the analysis stage's `evidence`; record the completed designer's actual
57
+ runtime and dispatch. No arbitrary shell check is granted to a plan worker.
58
+ A structurally valid BLOCKED document can be recorded and reviewed honestly.
59
+
60
+ For a bounded revision, pin the baseline JSON in the plan intake's `sources` and
61
+ run the fixed comparison before sending the candidate to the reviewer:
62
+
63
+ ```text
64
+ node .claude/nccgs/tools/task.mjs design-compare --id DESIGN --baseline .nccgs/evidence/context/baseline.json --file .nccgs/evidence/analysis/DESIGN-contract.json
65
+ ```
66
+
67
+ This command captures changed/added/removed requirement and criterion IDs, order
68
+ changes and other changed top-level fields into immutable design-check evidence.
69
+ It does not approve scope, interpret prose or claim PASS. Supply its receipt with
70
+ the actual baseline and candidate to the reviewer; arbitrary plan shell remains
71
+ blocked. A missing comparison cannot be deferred until after a conditional PASS.
72
+
73
+ ## /design-review — independent document acceptance
74
+
75
+ Reserve the existing independent reviewer (`kind: "review"`) after analysis is
76
+ recorded. Review the actual sources and full scoped contract, not just its schema:
77
+
78
+ - Is the intended player experience translated into implementable behavior? Are
79
+ rules, states, formulas, interaction/edge cases and tuning ownership complete?
80
+ - Does each required behavior have the right test, environment and justified
81
+ threshold? Are all requirements covered without adding unauthorized demands?
82
+ - Are delivery obligations distinguished from product research? Does a bot loss
83
+ wrongly fail a functional test or block unrelated implementation? Does an early
84
+ checkpoint represent a real prerequisite rather than a late soak/balance result?
85
+ Do controlled scenarios test normal logic without directly injecting the expected
86
+ outcome? Are shipped values checked separately from test setup interventions?
87
+ - Are bot diagnostics, technical checks, natural gameplay and human judgement
88
+ distinguished? Can a helper falsely pass while the player requirement fails?
89
+ - Are decisions genuinely authorized, contradictions resolved, dependencies feasible
90
+ and deferred behavior really out of scope? Cite exact sources for findings.
91
+
92
+ Return PASS, CHANGES_REQUIRED or BLOCKED with prioritized evidence and missing
93
+ decisions. Be independent of the designer. A PASS requires READY/READY_WITH_RISKS
94
+ and no unresolved blocking issue. A complete negative review is useful evidence.
95
+ Every recorded design finding must include `acceptanceImpact`: `none`, `false-pass`,
96
+ `missing-coverage` or `authority-gap`. A demonstrated way to pass a criterion while
97
+ violating a scoped fixed requirement or delegated bound is a blocking acceptance
98
+ gap, even if labelled MEDIUM or accepted as a risk. Missing required coverage or
99
+ threshold authority is also blocking. Return CHANGES_REQUIRED/BLOCKED until the
100
+ document is corrected and independently reviewed; never defer these gaps to QA
101
+ after importing the deficient criteria. PASS rejects all unresolved acceptance
102
+ impacts regardless of severity or ACCEPTED status. Ordinary nonblocking risks may
103
+ use `none`. Semantic detection remains the reviewer's responsibility; this field
104
+ does not turn structural validation into a semantic oracle.
105
+ Use the existing review budget; do not mint tasks to reset attempts. A document
106
+ revision requires fresh validation and independent review before a new handoff.
107
+ Follow [review handoff](review-handoff.md). New policies bind verdict, criteriaMet,
108
+ findings and remaining conditions to the completed reviewer's pinned JSON block.
109
+ Pending acceptance work requires BLOCKED/CHANGES_REQUIRED, not a conditional PASS.
110
+
111
+ ## /story-readiness — gate before dependent implementation
112
+
113
+ Check the approved scope, stable requirements, test mapping, decisions, dependencies,
114
+ ownership, feasibility, protected paths and current source revisions. Preserve
115
+ residual risks. A READY verdict describes the document, not a verified game.
116
+ After positive independent review, record the plan's honest success report and
117
+ close its gates through the CLI. Unresolved product decisions remain BLOCKED.
118
+
119
+ The implementation intake references the closed plan:
120
+
121
+ ```json
122
+ { "schemaVersion": 1, "mode": "implement", "design": { "taskId": "DESIGN" } }
123
+ ```
124
+
125
+ Initialization verifies the complete plan gates and pins its task, contract, sources,
126
+ validation receipt and independent runtime evidence. It imports objective, scope,
127
+ target, criteria, required checks and manual gates exactly. A conflicting override
128
+ is rejected. Evidence changes invalidate the handoff; edits to the implementation
129
+ source alone do not rewrite or invalidate the historical document review.
130
+
131
+ New installs enable `designReviewRequired: true` in execution.json. With this policy,
132
+ feature/cross-system implementation cannot reserve, launch, write source, verify
133
+ or close without a current reviewed handoff. A ready local fix with explicit
134
+ acceptance can proceed without a separate design task; do not misclassify new
135
+ gameplay as a local fix. Structural controls cannot infer hidden semantic scope.
136
+ Missing/false policy retains older obligations and is not evidence of design
137
+ readiness. Existing projects opt in explicitly for fresh tasks; do not change a
138
+ frozen task's pinned execution policy to retrofit approval. Compatibility branches
139
+ retain their original design/approval process.
@@ -14,8 +14,18 @@ build state. A changed baseline must be explained before closure.
14
14
  Subjective experience remains a Product Owner gate. Provide a reproducible scenario
15
15
  and what to observe; never fabricate validation.
16
16
 
17
- For astra-claude, every code change requires Fable review, including FAST work.
17
+ For standalone execution, use [the standalone protocol](standalone.md): reports
18
+ may honestly be partial, blocked or failed; only successful closure needs all gates.
19
+ For compatibility execution with astra-claude, every code change requires Fable review, including FAST work.
18
20
  Record requested versus observed model/effort and the final dirty/untracked snapshot,
19
21
  not only a role label or commit ID. A Fable PASS permits reporting to Astra, not
20
22
  DONE. Astra's check and the Product Owner's explicit acceptance remain separate
21
23
  required evidence. See [the handoff protocol](astra-claude.md).
24
+
25
+ Match verification to the criterion and concrete risk. Stop expanding evidence
26
+ when approved criteria and relevant checks pass with no actionable blockers. Rerun
27
+ only for changed inputs, failures, or new evidence; a repeated green run adds no
28
+ new proof by itself. Reviewers cover distinct risks, not duplicate reassurance.
29
+ If subjective questions remain, supply a playable observation scenario for the
30
+ Product Owner. Exhausted review rounds require a recorded stop or explicit external
31
+ extension; they never justify PASS or weaken the acceptance chain.
@@ -1,44 +1,46 @@
1
1
  # Model routing protocol
2
2
 
3
- Model choice is a quality and compatibility policy, not a status symbol.
4
-
5
- `.claude/nccgs/pipeline.json` is the source for routing. Catalog and frontmatter
6
- values are generated, checked for drift, and documented in
7
- [routing.generated.md](../routing.generated.md). Ten core roles load by default;
8
- 35 optional roles remain in `agent-library/` until explicitly enabled. Activation
9
- is recorded in `.nccgs/agent-activation.json` and survives updates.
10
-
11
- ## Capability classes
12
-
13
- The default profile is `astra-claude`: all Claude implementation/support roles use
14
- the pinned `claude-opus-5-5` model with `effort: xhigh`; the `review` class uses
15
- `fable` with `effort: high`. Astra is an external planning/document owner, not a
16
- Claude model alias. See [the required handoff sequence](astra-claude.md).
17
-
18
- `configure-models.mjs` applies both model and effort to agent frontmatter. For
19
- `astra-claude`, it also sets the project session model and effort in
20
- `.claude/settings.json`, preserving unrelated settings and backing up replaced
21
- settings. Existing sessions, local settings, environment variables, forced subagent
22
- routing, and provider restrictions may still override configuration; verify the
23
- actual runtime model/effort before recording a successful stage.
24
-
25
- - `lightweight`: indexing, state, formatting, deterministic documentation updates;
26
- - `standard`: implementation, focused design, tests, ordinary domain analysis;
27
- - `deep`: architecture, migration, adversarial reasoning, cross-domain conflicts;
28
- - `leadership`: product-wide creative, technical, or production synthesis.
29
- - `review`: independent adversarial review after implementation.
30
-
31
- Legacy `quality` uses Haiku for lightweight, Sonnet for standard, and Opus for
32
- deep/leadership/review. `balanced` moves routine leadership to Sonnet. `inherit`
33
- removes agent model and effort pins and uses the session settings. Legacy profiles
34
- do not change the main session settings. Selecting them explicitly opts out of the
35
- `astra-claude` chain; ordinary updates preserve an existing project's profile.
36
-
37
- Select the role first, then apply project model policy. Do not call Opus merely
38
- because a role sounds senior. Escalate when evidence shows cross-system ambiguity,
39
- irreversible risk, security implications, conflicting canon, or repeated failed
40
- hypotheses.
41
-
42
- Only legacy profiles may use an inherit fallback when project policy allows it.
43
- For `astra-claude`, unavailable or substituted model/effort blocks that stage; never
44
- present another model's result as Opus 5.5 implementation or Fable review.
3
+ NCCGS runs in Claude Code. Resolve execution obligations independently of models.
4
+ `pipeline.json` is the routing source; generated catalog/frontmatter and
5
+ [routing.generated.md](../routing.generated.md) must match it. Thirteen roles
6
+ load by default; 34 optional specialists remain in `agent-library/` until needed.
7
+
8
+ ## Standalone default
9
+
10
+ | Risk | Implementation | Requested effort | Review rounds |
11
+ |---|---|---|---|
12
+ | FAST | Sonnet, lightweight class | medium | 2 |
13
+ | STANDARD | Opus, standard class | medium | 3 |
14
+ | CONTROLLED | Opus, deep class | high | 3 |
15
+
16
+ Independent review uses Opus/high. These are family aliases, not claims about a
17
+ particular resolved provider version. Runtime observations must identify actual
18
+ Claude models. Unknown effort remains null unless project policy requires observed
19
+ effort, in which case missing evidence blocks the stage. Never infer observed
20
+ effort, model, duration or cost from configuration.
21
+
22
+ Use `nccgs-fast-implementer` for bounded FAST work; use a domain programmer for
23
+ STANDARD and `nccgs-unity-implementer` for CONTROLLED. A scout resolves one missing
24
+ fact; it is optional. Do not promote work merely because a role sounds senior.
25
+ The same model family may implement and review, but review must run in a separate
26
+ observed agent invocation and consume a reserved review attempt.
27
+
28
+ `configure-models.mjs` renders profile model/effort settings. Local settings,
29
+ environment, provider restrictions and existing sessions may override them.
30
+ Restart/recheck the real session after configuration changes. Classify and reserve
31
+ before dispatch, then use the task's route. Do not silently fall back when a model
32
+ is unavailable. Report the actual blocker and measurements.
33
+
34
+ ## Compatibility
35
+
36
+ Missing `.nccgs/execution.json` means compatibility. Historical `astra-claude`
37
+ retains Sonnet/medium for FAST, `claude-opus-5-5` medium/high for larger work,
38
+ Fable/high review, external Astra check and explicit Product Owner acceptance.
39
+ These historical identifiers are preserved configuration, not new availability
40
+ claims. See [its protocol](astra-claude.md).
41
+
42
+ `quality`, `balanced` and `inherit` remain available for historical workflows.
43
+ Changing only a model profile cannot waive task gates. Standalone schema-2 tasks
44
+ require explicit implementation/review families; an unpinned inherit route cannot
45
+ prove them. Ordinary updates retain configuration. Use explicit execution migration
46
+ to adopt standalone; historical task records and spent budgets remain intact.