nccgs 1.3.0 → 2.0.0-rc.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (180) hide show
  1. package/.claude/agents/nccgs-adversarial-reviewer.md +17 -6
  2. package/.claude/agents/nccgs-engine-programmer.md +2 -2
  3. package/.claude/agents/nccgs-fast-implementer.md +30 -0
  4. package/.claude/agents/nccgs-game-designer.md +26 -0
  5. package/.claude/agents/nccgs-gameplay-programmer.md +3 -4
  6. package/.claude/agents/nccgs-qa-engineer.md +2 -2
  7. package/.claude/agents/nccgs-task-scout.md +24 -0
  8. package/.claude/agents/nccgs-ui-programmer.md +2 -2
  9. package/.claude/agents/nccgs-unity-build-specialist.md +2 -2
  10. package/.claude/agents/nccgs-unity-implementer.md +7 -6
  11. package/.claude/agents/nccgs-unity-systems-specialist.md +2 -2
  12. package/.claude/agents/nccgs-unity-ui-specialist.md +2 -2
  13. package/.claude/agents/nccgs-verification-engineer.md +2 -2
  14. package/.claude/nccgs/MEMORY.md +59 -0
  15. package/.claude/nccgs/THIRD_PARTY_NOTICES.md +16 -0
  16. package/.claude/nccgs/VERSION +1 -1
  17. package/.claude/nccgs/agent-library/nccgs-adversarial-reviewer.md +16 -5
  18. package/.claude/nccgs/agent-library/nccgs-fast-implementer.md +28 -0
  19. package/.claude/nccgs/agent-library/nccgs-game-designer.md +1 -2
  20. package/.claude/nccgs/agent-library/nccgs-gameplay-programmer.md +1 -2
  21. package/.claude/nccgs/agent-library/nccgs-task-scout.md +22 -0
  22. package/.claude/nccgs/agent-library/nccgs-unity-implementer.md +5 -4
  23. package/.claude/nccgs/constitution.md +44 -12
  24. package/.claude/nccgs/hooks/agent-audit.mjs +2 -2
  25. package/.claude/nccgs/hooks/closure-guard.mjs +31 -8
  26. package/.claude/nccgs/hooks/common.mjs +2 -1
  27. package/.claude/nccgs/hooks/runtime-observation.mjs +2 -2
  28. package/.claude/nccgs/hooks/session-start.mjs +11 -4
  29. package/.claude/nccgs/packages/ponytail/LICENSE +21 -0
  30. package/.claude/nccgs/packages/ponytail/NOTICE.md +32 -0
  31. package/.claude/nccgs/packages/ponytail/README.md +46 -0
  32. package/.claude/nccgs/packages/ponytail/complexity-review.md +47 -0
  33. package/.claude/nccgs/packages/ponytail/engineering.md +68 -0
  34. package/.claude/nccgs/packages/ponytail/manifest.json +24 -0
  35. package/.claude/nccgs/pipeline.json +92 -8
  36. package/.claude/nccgs/protocols/agent-contract.md +50 -1
  37. package/.claude/nccgs/protocols/astra-claude.md +46 -12
  38. package/.claude/nccgs/protocols/bounded-repair.md +118 -0
  39. package/.claude/nccgs/protocols/context-packets.md +5 -0
  40. package/.claude/nccgs/protocols/delivery-acceptance.md +105 -0
  41. package/.claude/nccgs/protocols/design-lifecycle.md +139 -0
  42. package/.claude/nccgs/protocols/evidence.md +11 -1
  43. package/.claude/nccgs/protocols/model-routing.md +44 -42
  44. package/.claude/nccgs/protocols/orchestration.md +46 -5
  45. package/.claude/nccgs/protocols/review-handoff.md +65 -0
  46. package/.claude/nccgs/protocols/standalone-gates.md +273 -0
  47. package/.claude/nccgs/protocols/standalone-pilot.md +144 -0
  48. package/.claude/nccgs/protocols/standalone.md +187 -0
  49. package/.claude/nccgs/protocols/task-autonomy.md +55 -0
  50. package/.claude/nccgs/protocols/task-gates.md +73 -7
  51. package/.claude/nccgs/routing.generated.md +25 -7
  52. package/.claude/nccgs/settings.fragment.json +153 -141
  53. package/.claude/nccgs/studio.json +135 -101
  54. package/.claude/nccgs/tools/autonomy.mjs +27 -0
  55. package/.claude/nccgs/tools/command-process.mjs +44 -0
  56. package/.claude/nccgs/tools/compile-policy.mjs +34 -4
  57. package/.claude/nccgs/tools/configure-models.mjs +13 -5
  58. package/.claude/nccgs/tools/criteria.mjs +121 -0
  59. package/.claude/nccgs/tools/delivery-budget.mjs +52 -0
  60. package/.claude/nccgs/tools/design-readiness.mjs +158 -0
  61. package/.claude/nccgs/tools/evidence.mjs +153 -0
  62. package/.claude/nccgs/tools/gates.mjs +151 -24
  63. package/.claude/nccgs/tools/inputs.mjs +73 -0
  64. package/.claude/nccgs/tools/lifecycle.mjs +521 -0
  65. package/.claude/nccgs/tools/native-completion.mjs +21 -0
  66. package/.claude/nccgs/tools/orchestration.mjs +430 -0
  67. package/.claude/nccgs/tools/paths.mjs +60 -0
  68. package/.claude/nccgs/tools/policy.mjs +58 -13
  69. package/.claude/nccgs/tools/process-lock.mjs +74 -0
  70. package/.claude/nccgs/tools/repair-packet.mjs +87 -0
  71. package/.claude/nccgs/tools/review-handoff.mjs +32 -0
  72. package/.claude/nccgs/tools/runtime.mjs +33 -7
  73. package/.claude/nccgs/tools/task.mjs +62 -31
  74. package/.claude/nccgs/tools/tool-guard.mjs +117 -0
  75. package/.claude/nccgs/tools/verification-jobs.mjs +161 -0
  76. package/.claude/nccgs/tools/verification-worker.mjs +71 -0
  77. package/.claude/nccgs/tools/windows-job.cs +96 -0
  78. package/.claude/nccgs/workflow-catalog.json +286 -26
  79. package/.claude/skills/accessibility-review/SKILL.md +12 -5
  80. package/.claude/skills/architecture-decision/SKILL.md +12 -5
  81. package/.claude/skills/asset-audit/SKILL.md +12 -5
  82. package/.claude/skills/audit/SKILL.md +12 -5
  83. package/.claude/skills/balance-review/SKILL.md +12 -5
  84. package/.claude/skills/bug-triage/SKILL.md +12 -5
  85. package/.claude/skills/bug-triage/references/astra.md +12 -5
  86. package/.claude/skills/closure/SKILL.md +12 -5
  87. package/.claude/skills/compatibility-review/SKILL.md +12 -5
  88. package/.claude/skills/context-pack/SKILL.md +12 -5
  89. package/.claude/skills/dependency-review/SKILL.md +12 -5
  90. package/.claude/skills/design/SKILL.md +14 -5
  91. package/.claude/skills/design-review/SKILL.md +14 -5
  92. package/.claude/skills/evidence-review/SKILL.md +12 -5
  93. package/.claude/skills/hotfix/SKILL.md +12 -5
  94. package/.claude/skills/hotfix/references/astra.md +12 -5
  95. package/.claude/skills/incident-recovery/SKILL.md +12 -5
  96. package/.claude/skills/incident-recovery/references/astra.md +12 -5
  97. package/.claude/skills/localize-game/SKILL.md +12 -5
  98. package/.claude/skills/localize-game/references/astra.md +12 -5
  99. package/.claude/skills/migrate-project/SKILL.md +12 -5
  100. package/.claude/skills/migrate-project/references/astra.md +12 -5
  101. package/.claude/skills/migrate-project/references/procedure.md +2 -1
  102. package/.claude/skills/milestone-review/SKILL.md +12 -5
  103. package/.claude/skills/nccgs-code-review/SKILL.md +12 -5
  104. package/.claude/skills/nccgs-code-review/references/astra.md +8 -2
  105. package/.claude/skills/performance-audit/SKILL.md +12 -5
  106. package/.claude/skills/plan-feature/SKILL.md +12 -5
  107. package/.claude/skills/playtest/SKILL.md +12 -5
  108. package/.claude/skills/playtest/references/astra.md +16 -5
  109. package/.claude/skills/project-stage/SKILL.md +12 -5
  110. package/.claude/skills/prototype-feature/SKILL.md +12 -5
  111. package/.claude/skills/prototype-feature/references/astra.md +16 -5
  112. package/.claude/skills/qa-plan/SKILL.md +12 -5
  113. package/.claude/skills/release/SKILL.md +12 -5
  114. package/.claude/skills/release/references/astra.md +12 -5
  115. package/.claude/skills/release-readiness/SKILL.md +12 -5
  116. package/.claude/skills/retrospective/SKILL.md +12 -5
  117. package/.claude/skills/review/SKILL.md +12 -5
  118. package/.claude/skills/review/references/astra.md +8 -2
  119. package/.claude/skills/security-audit/SKILL.md +12 -5
  120. package/.claude/skills/sprint-plan/SKILL.md +12 -5
  121. package/.claude/skills/status/SKILL.md +12 -5
  122. package/.claude/skills/status/references/astra.md +6 -0
  123. package/.claude/skills/story-readiness/SKILL.md +14 -5
  124. package/.claude/skills/test/SKILL.md +12 -5
  125. package/.claude/skills/test/references/astra.md +12 -5
  126. package/.claude/skills/ui-review/SKILL.md +12 -5
  127. package/.claude/skills/work/SKILL.md +12 -5
  128. package/.claude/skills/work/references/astra.md +17 -5
  129. package/CHANGELOG.md +42 -0
  130. package/CLAUDE.md +2 -0
  131. package/MEMORY.md +55 -0
  132. package/README.md +132 -180
  133. package/THIRD_PARTY_NOTICES.md +15 -0
  134. package/UPGRADING.md +91 -69
  135. package/VERSION +1 -1
  136. package/docs/ARCHITECTURE.md +19 -5
  137. package/docs/BOUNDED-REPAIR.md +11 -0
  138. package/docs/HUONG-DAN-MIGRATE-VA-SU-DUNG.md +46 -34
  139. package/docs/NCCGS-1.3.md +10 -122
  140. package/docs/NCCGS-1.4.1.md +149 -0
  141. package/docs/NCCGS-1.4.md +184 -0
  142. package/docs/NCCGS-1.5-IMPLEMENTATION-PLAN.md +378 -0
  143. package/docs/NCCGS-1.5-PILOT.md +130 -0
  144. package/docs/NCCGS-1.5.md +237 -0
  145. package/docs/PHASE-1-OPERATIONS.md +137 -0
  146. package/docs/PHASE-2-OPERATIONS.md +183 -0
  147. package/docs/PROJECT-POLICY.md +21 -9
  148. package/docs/RELEASE-STATUS.md +42 -0
  149. package/docs/WORKFLOWS.md +42 -24
  150. package/package.json +7 -3
  151. package/scaffold/.nccgs/autonomy.json +4 -0
  152. package/scaffold/.nccgs/execution.json +10 -0
  153. package/scaffold/.nccgs/inputs.json +6 -0
  154. package/scaffold/.nccgs/project.yaml +4 -2
  155. package/scaffold/.nccgs/templates/agent-handoff.md +2 -0
  156. package/scaffold/.nccgs/templates/design-contract.json +34 -0
  157. package/scaffold/.nccgs/templates/feature-contract.md +21 -5
  158. package/scaffold/.nccgs/templates/implementation-report.md +4 -1
  159. package/scaffold/.nccgs/templates/repair-packet.json +14 -0
  160. package/scripts/benchmark-snapshot.mjs +205 -0
  161. package/scripts/benchmark-v15.mjs +173 -0
  162. package/scripts/cli.mjs +129 -2
  163. package/scripts/doctor.mjs +128 -22
  164. package/scripts/install.mjs +96 -34
  165. package/scripts/run.mjs +72 -0
  166. package/scripts/supervise.mjs +124 -0
  167. package/scripts/validate.mjs +45 -4
  168. package/tests/autonomy.test.mjs +113 -0
  169. package/tests/benchmark-v15.test.mjs +65 -0
  170. package/tests/design-readiness.test.mjs +200 -0
  171. package/tests/evidence.test.mjs +186 -0
  172. package/tests/framework.test.mjs +333 -16
  173. package/tests/gates.test.mjs +313 -4
  174. package/tests/lifecycle-v15.test.mjs +300 -0
  175. package/tests/orchestration-v15.test.mjs +545 -0
  176. package/tests/phase1.test.mjs +324 -0
  177. package/tests/phase2.test.mjs +250 -0
  178. package/tests/policy-v15.test.mjs +220 -0
  179. package/tests/ponytail-package.test.mjs +75 -0
  180. package/tests/runtime-v15.test.mjs +624 -0
@@ -15,17 +15,58 @@ another specialty instead of spawning their own hierarchy.
15
15
 
16
16
  ## Team sizing
17
17
 
18
- - FAST: normally no subagent or one specialist.
19
- - STANDARD: one accountable implementer plus independent verification; add a design
18
+ - FAST: one accountable lightweight implementer with a concise plan and report.
19
+ - STANDARD: one accountable standard implementer plus independent verification; add a design
20
20
  or domain specialist only when the feature actually needs it.
21
- - CONTROLLED: decision owner, affected specialists, implementer, verification, and
21
+ - CONTROLLED: one deep implementer accountable for integration, decision owner,
22
+ affected specialists, verification, and
22
23
  adversarial reviewer. Avoid ceremonial roles with no distinct deliverable.
23
24
 
24
25
  Pass a context packet path and exact responsibility. Do not paste full GDDs, ADRs,
25
26
  or chat history into every prompt. Surface blocked nodes immediately and preserve
26
27
  completed independent outputs.
27
28
 
28
- Under astra-claude, Astra owns the plan and document scope. Claude orchestrates
29
- implementation nodes inside that plan on Opus 5.5/xhigh. Every coding outcome must
29
+ Standalone execution follows [the standalone protocol](standalone.md), including
30
+ durable dispatch reservations, attempt budgets, mode guards and optional external
31
+ review. Path ownership is checked for declared reservations and direct edit tools;
32
+ opaque shell/MCP operations require independent diff review, not a sandbox claim.
33
+
34
+ Under compatibility execution with astra-claude, Astra owns the plan and document scope. Claude orchestrates
35
+ implementation nodes inside that plan using the task's risk-selected model/effort.
36
+ Every coding outcome must
30
37
  pass an independent Fable review before its final report goes to Astra. Even FAST
31
38
  work cannot bypass this sequence. See [the handoff protocol](astra-claude.md).
39
+
40
+ Use `nccgs-task-scout` only to close a named factual gap. Do not launch a scout,
41
+ design consultation, or extra reviewer when existing evidence already answers the
42
+ question. Each reviewer must cover a distinct risk. Preserve one accountable
43
+ implementer even when independent support work runs in parallel.
44
+
45
+ Keep only one reviewer in flight per task. Standalone checks work-stage readiness
46
+ before reserving review and again before launch. A rejected launch is automatically
47
+ marked failed only while demonstrably unclaimed; its quota stays spent. Claimed or
48
+ uncertain invocations still require runtime evidence or external reconciliation.
49
+ When technical checks prevent work-stage recording, use measured negative criteria
50
+ to request an eligible protected repair instead of reserving a premature reviewer.
51
+
52
+ Do not expand scope speculatively. Stop when the approved criteria and relevant
53
+ checks pass with no actionable blockers; rerun only checks affected by changed
54
+ inputs, failures, or new evidence. Turn unresolved subjective questions into a
55
+ playable Product Owner test. Review budgets are enforced by task gates; context
56
+ and time budgets are advisory. Exhaustion requires an evidence-backed stop and an
57
+ explicit external extension, never automatic approval.
58
+
59
+ Native agent `maxTurns` comes from the installed role frontmatter; a larger
60
+ `Agent.max_turns` request is not evidence that the role cap increased. Templates
61
+ are authoritative and policy compilation synchronizes the registry. Designer,
62
+ gameplay/Unity implementation and review roles have bounded allowances of 48,
63
+ 120 and 64 turns respectively; lighter scout/FAST roles keep their short budgets.
64
+
65
+ An external operator can reallocate a pinned dispatch limit with `task
66
+ extend-dispatch --id TASK --request FILE`. The request must contain
67
+ `previousLimit`, a larger finite `maxDispatches`, `reason`, and a hashed
68
+ authorization artifact `reference`. All dispatches must be settled. The ledger
69
+ records before/after limits; spent attempts, ancestor limits and protected slots
70
+ remain unchanged. Claude cannot invoke this command. If a framework update changes
71
+ a plan's source baseline, retain the old failed plan and create a bounded child
72
+ under its preserved aggregate budget; do not reset the old baseline or history.
@@ -0,0 +1,65 @@
1
+ # Reviewer-owned verdict and remaining conditions
2
+
3
+ New standalone installs enable `reviewContractRequired: true`. The completed
4
+ review dispatch must pin the same Markdown handoff used as the review stage's
5
+ `evidence`. The reviewer writes exactly one `nccgs-review` fenced JSON block:
6
+
7
+ ```nccgs-review
8
+ {
9
+ "schemaVersion": 1,
10
+ "taskId": "TASK",
11
+ "contractHash": "CURRENT_CONTRACT_HASH",
12
+ "snapshot": "REVIEWED_SOURCE_SNAPSHOT",
13
+ "verdict": "BLOCKED",
14
+ "criteriaMet": false,
15
+ "findings": [],
16
+ "conditions": [
17
+ { "id": "C-01", "description": "Required baseline comparison has not been supplied or reviewed." }
18
+ ]
19
+ }
20
+ ```
21
+
22
+ Use actual task identity, hash and snapshot supplied by the coordinator. Findings
23
+ retain the required severity, status, description and design acceptanceImpact.
24
+ Finding severity must be BLOCKER, HIGH, MEDIUM, LOW or INFO; status must be OPEN,
25
+ RESOLVED or ACCEPTED. Use ACCEPTED for an acknowledged informational limitation,
26
+ not NOTED. The coordinator must not translate an invalid reviewer-owned value.
27
+ Prose explains the evidence, scope and limits; the block is the authoritative
28
+ verdict, assessment, findings and remaining conditions. Coordinator stage input
29
+ must copy these four fields exactly, including an explicit `conditions` array.
30
+
31
+ PASS requires `conditions: []`. If acceptance still depends on a diff, test,
32
+ source read or other check, return BLOCKED or CHANGES_REQUIRED with that condition.
33
+ A finished negative review may mark its handoff COMPLETE. Do not write a conditional
34
+ PASS or hide pending acceptance work in a prose disclaimer. The framework checks
35
+ the explicit contract; detecting an omitted semantic condition remains a reviewer
36
+ responsibility. Neither a marker nor a schema proves complete semantic coverage.
37
+
38
+ Supply required checks before launching review. When new evidence arrives after
39
+ a completed conditional/negative review, retain that artifact and obtain a fresh
40
+ independent review within the existing dispatch/round/time/cost limits. A coordinator
41
+ cannot clear conditions, change the verdict, edit the captured handoff or relabel
42
+ its own check as the reviewer's completed work. Report PARTIAL/BLOCKED when the
43
+ remaining allowance cannot fund the required follow-up.
44
+
45
+ For a broad delivery review, check the effective native role `maxTurns` before
46
+ dispatch. A number in the assignment prompt does not override the role's limit.
47
+ The shipped reviewer allows up to 200 turns; this is a ceiling, not a target.
48
+ Keep session cost, time and dispatch limits active. Index current receipts, raw
49
+ files and source changes before review, and use focused reads instead of repeatedly
50
+ loading whole histories. Write a truthful PARTIAL handoff early and update it as
51
+ checks finish, so interruption leaves concrete unresolved work. Only the reviewer
52
+ may replace its own draft with the completed verdict before the dispatch ends.
53
+ Capacity changes require an external allocation within the owner's authority;
54
+ never reset failed attempts or loosen the acceptance requirements.
55
+
56
+ The native harness can report an invocation `completed` and still say its worker
57
+ stopped at a turn limit. Harness-designated incomplete notices override a COMPLETE
58
+ file: the attempt fails as incomplete, all spent quota remains, and its evidence
59
+ cannot certify a stage. Unsupported harness notices/metadata fail closed pending
60
+ external inspection. Arbitrary quoted text in a worker's response is not harness
61
+ metadata and does not itself change completion status. Captured observations retain
62
+ the reported invocation status and a diagnosis/hash, not arbitrary response text.
63
+
64
+ Older pinned policies remain unchanged. Upgrade policy only for fresh authorized
65
+ tasks; never retrofit frozen evaluation records or reset attempts to obtain PASS.
@@ -0,0 +1,273 @@
1
+ # NCCGS 2.0 RC — implementation and operation
2
+
3
+ Version: **2.0.0-rc.2**. Target executor: **Claude Code**.
4
+ Framework tests and scoped native delivery evidence exist. Fresh end-to-end
5
+ qualification of this exact RC and published-1.3 upgrade qualification are pending. Read [MEMORY.md](../MEMORY.md) before interpreting evaluation results.
6
+
7
+ ## Policies and migration
8
+
9
+ New installs use `.nccgs/execution.json`:
10
+
11
+ ```json
12
+ {
13
+ "schemaVersion": 1,
14
+ "name": "standalone",
15
+ "externalReview": { "mode": "off" },
16
+ "acceptanceRequired": false,
17
+ "requireObservedEffort": false,
18
+ "maxParallelWorkers": 2,
19
+ "designReviewRequired": true,
20
+ "reviewContractRequired": true
21
+ }
22
+ ```
23
+
24
+ Model profile lives separately in `.nccgs/project.yaml`; the default is
25
+ `claude-standalone`. Its Sonnet/Opus aliases are requests, never resolved observations.
26
+ Missing execution.json resolves to compatibility. Changing a model profile alone
27
+ does not remove an external, acceptance or historical task obligation.
28
+
29
+ Use `nccgs migrate-execution --to standalone --dry-run` after a framework update;
30
+ remove `--dry-run` to apply. Add `--model-profile claude-standalone` to explicitly
31
+ change models too. Configuration archives live under `.nccgs/backups/execution/ID`.
32
+ `--rollback ID --dry-run` checks restoration; rollback refuses changed configuration
33
+ or corrupted backup hashes. Tasks/evidence/budgets are preserved. See
34
+ the packaged UPGRADING.md for failure recovery and legacy task handling.
35
+
36
+ ## Intake, classification and modes
37
+
38
+ Example `.nccgs/evidence/intake/hud-fix.json`:
39
+
40
+ ```json
41
+ {
42
+ "schemaVersion": 1,
43
+ "objective": "Correct the pause-button offset at the declared reference resolution",
44
+ "mode": "implement",
45
+ "scope": "local",
46
+ "systems": ["HUD"],
47
+ "target": "fix",
48
+ "readiness": "ready",
49
+ "acceptance": [{ "id": "HUD-01", "description": "Offset matches the reference", "check": "layout-reference",
50
+ "assertions": [{ "artifact": ".nccgs/evidence/measurements/layout.json", "pointer": "/offsetMatches", "op": "eq", "value": true }] }],
51
+ "requiredChecks": ["layout-reference"],
52
+ "requiredArtifacts": [],
53
+ "manualGates": [],
54
+ "assumptions": ["Gameplay and input behavior remain unchanged"],
55
+ "reassessWhen": ["The layout defect crosses multiple UI systems"]
56
+ }
57
+ ```
58
+
59
+ This is a template, not evidence that the check ran. For `audit` or `plan`, declare
60
+ `analysisArtifacts` such as `.nccgs/evidence/analysis/TASK-result.md`; source edits
61
+ are prohibited. Include real `sources` paths to hash authoritative documents.
62
+ Criteria with `check` add required objective checks; `{id,description,manual:true}`
63
+ adds a manual acceptance gate. Project-level gate manifests remain mandatory.
64
+ Automatic criteria require measured JSON assertions (`eq`, `gte`, `lte`); check-only
65
+ criteria remain UNVERIFIED. The verifier captures changed measurement files into
66
+ hashed per-job evidence. Failed/unverified criteria block PASS review, success report
67
+ and DONE, independently of the helper's exit code. Inspect `task acceptance-status`.
68
+ Use `checkpoint: "early"` only for a small delivery prerequisite that actually
69
+ blocks dependent work. Bot survival/victory, damage headroom and product balance
70
+ are not default expansion gates. See [delivery acceptance](delivery-acceptance.md)
71
+ and [phase-2 operations](../../../docs/PHASE-2-OPERATIONS.md). The engine still
72
+ enforces pinned criteria; revising scope requires an explicit reviewed contract.
73
+
74
+ ```text
75
+ node .claude/nccgs/tools/task.mjs init --id hud-fix --intake .nccgs/evidence/intake/hud-fix.json
76
+ node .claude/nccgs/tools/task.mjs bind --id hud-fix --session ACTUAL_SESSION_ID
77
+ node .claude/nccgs/tools/task.mjs inspect --id hud-fix
78
+ ```
79
+
80
+ Run each task command in its own shell tool call from the project directory.
81
+ Do not append pipes, redirection (including `2>&1`), semicolons, `cd`, or other
82
+ commands. The guard recognizes a plain task invocation; composed shell commands
83
+ are not lifecycle commands and require a correlated implementation worker.
84
+ Use `inspect` to read task state, and Read/Grep for command documentation.
85
+ `status` and `--help` are not recognized lifecycle operations. If a plain command
86
+ is rejected, retain the error and inspect its arguments instead of bypassing the
87
+ guard or requesting an external bind before correcting the invocation.
88
+
89
+ The classifier is deterministic and conservative, not a general semantic model.
90
+ Ready/local/fix can be FAST. Unknown or feature scope defaults to STANDARD.
91
+ Save/serialization, migration, network authority, commerce and cross-system signals
92
+ require CONTROLLED. The coordinator must inspect real risk and supply `riskSignals`,
93
+ reason and evidence; keyword absence is not proof of safety.
94
+
95
+ Explicit `reclassify --classification FILE` requires `reason` and `evidence`; optional
96
+ `risk`, `scope`, `target`, `readiness`, `systems` and `riskSignals` explain the change.
97
+ To change intake content, update the same file and use `refresh-inputs --reason TEXT`.
98
+ Reclassification invalidates affected stages while retaining old evidence and spent
99
+ review rounds. Source/input/policy changes never reuse old approval. Execution policy
100
+ changes cannot be adopted by input refresh: restore pinned policy for current work
101
+ or explicitly scope new work while retaining prior history and spent budgets. Audit/plan
102
+ baseline hashing still detects source edits after canonical inputs are refreshed.
103
+
104
+ ## Dispatch, budget and evidence
105
+
106
+ Save a request under `.nccgs/evidence/requests/`:
107
+
108
+ ```json
109
+ {
110
+ "requestKey": "hud-fix-implementation-1",
111
+ "kind": "implementation",
112
+ "agent": "nccgs-fast-implementer",
113
+ "writePaths": ["Assets/Game/UI/PauseButton.cs"],
114
+ "reason": "One ready local layout fix with a targeted reference check"
115
+ }
116
+ ```
117
+
118
+ ```text
119
+ node .claude/nccgs/tools/task.mjs reserve --id hud-fix --request .nccgs/evidence/requests/hud-fix.json
120
+ ```
121
+
122
+ Use the returned ID in the Agent prompt prefix `[NCCGS-DISPATCH:ID]`, select the
123
+ specified role and task route, and include contract, inputs, ownership and acceptance.
124
+ PreToolUse claims the slot with actual session/tool-use identity. SubagentStart binds
125
+ an unambiguous agent ID. Foreground PostToolUse completion captures final telemetry.
126
+ For this preview use foreground calls; background runs hold slots until correlated
127
+ final evidence or external reconciliation. SubagentStop alone cannot prove model,
128
+ effort or complete usage. Ambiguous same-role starts are not guessed.
129
+
130
+ Review requests use `kind: "review"`, `agent: "nccgs-adversarial-reviewer"`, and
131
+ empty writePaths. Analysis requests also have no source ownership. Dispatch ceilings
132
+ are 6/12/24 for new FAST/STANDARD/CONTROLLED tasks by default; first reservation may explicitly
133
+ set `limits.maxDispatches`, `maxDurationMs`, `maxCostUsd`. Limits are pinned, counted
134
+ across descendants, and cannot be raised by another reservation. Review limits are
135
+ 2/3/3. Failed, cancelled and uncertain attempts remain spent. Extensions of review
136
+ rounds require a reasoned external operator decision and never grant approval.
137
+ New implementation ledgers protect four slots within the total: two reviews, one
138
+ repair and one read-only QA dispatch. Ordinary `purpose: "delivery"` cannot spend
139
+ them. `purpose: "repair"` needs recorded negative review/measured acceptance feedback
140
+ and a bounded `workPacket` with exact files, checks and one fresh handoff. Use
141
+ `task dispatch-prompt --id TASK --dispatch ID` after reservation. See
142
+ [bounded repair](bounded-repair.md) for enforced per-worker allowances;
143
+ `purpose: "qa"` uses a QA/verification role after a completed repair with retained feedback.
144
+ Audit/plan protect one review slot. Existing ledgers retain their pinned allocation.
145
+
146
+ At most two implementation workers can reserve disjoint paths; review/analysis
147
+ cannot overlap source writers. Direct edit tools enforce ownership. Opaque shell/MCP
148
+ operations are not OS-sandboxed by NCCGS; inspect their actual diff and test effects.
149
+ Audit/plan/review block arbitrary shells and unknown tools. Use read tools, declared
150
+ analysis artifacts and externally supplied command evidence when required.
151
+
152
+ `task verify --id TASK --name CHECK -- EXECUTABLE ARGS` records the actual command,
153
+ exit code, bounded timeout, log hash and before/after source snapshot. Claude may
154
+ run it only for implementation tasks. The verifier must not change source.
155
+ Use `verify --timeout-ms 900000 --background` before `--` for a long real-time
156
+ check when no check-specific timeout is configured. The requested timeout is
157
+ recorded and pinned to the request key (100..7200000 ms); it cannot exceed an
158
+ explicit check timeout or extend the supervisor's overall deadline. Windows jobs
159
+ retain all descendants, even detached children: a quick launcher does not let a
160
+ 600-second player escape the default 300-second timeout. Keep long collection in
161
+ the managed job with a sufficient bound, then validate its retained output.
162
+ Do not use WMI or another parent process to escape job ownership. Poll the actual
163
+ job with `job-status` or `job-wait --timeout-ms 30000`; do not create sleep-only
164
+ verification jobs or keep waiting after the collector has exited.
165
+ `task ref --file PATH` produces a `{path,sha256}` reference without inventing evidence.
166
+
167
+ Record a completed work stage with a JSON payload (placeholders below must be
168
+ replaced with returned values, never copied as observations):
169
+
170
+ ```json
171
+ {
172
+ "snapshot": "ACTUAL_FINAL_SNAPSHOT",
173
+ "dispatchId": "ACTUAL_COMPLETED_DISPATCH_ID",
174
+ "runtime": { "path": ".nccgs/evidence/runtime/observations/ACTUAL.json", "sha256": "ACTUAL_HASH" },
175
+ "checks": [{ "name": "layout-reference", "status": "PASS", "evidence": { "path": ".nccgs/evidence/ACTUAL-check.json", "sha256": "ACTUAL_HASH" } }]
176
+ }
177
+ ```
178
+
179
+ Use `record --id TASK --stage implementation --evidence FILE`. Audit/plan use stage
180
+ `analysis` and include `evidence` referencing the declared result artifact. Each
181
+ accepted work/review dispatch can be consumed once, including across history.
182
+
183
+ Review payload adds `verdict` (`PASS|CHANGES_REQUIRED|BLOCKED`), `criteriaMet`,
184
+ `findings`, and a review-artifact `evidence` ref to snapshot/dispatchId/runtime.
185
+ Findings include `severity` (`BLOCKER|HIGH|MEDIUM|LOW|INFO`), `status`
186
+ (`OPEN|RESOLVED|ACCEPTED`) and `description`. An unresolved BLOCKER/HIGH or unmet
187
+ required criterion prevents internal PASS. Review identity must differ from work.
188
+ With reviewContractRequired, the completed reviewer's pinned handoff must contain
189
+ exactly one nccgs-review JSON block; copy its verdict, criteriaMet, findings and
190
+ conditions exactly into the stage payload. Pending conditions block PASS. Follow
191
+ [review handoff](review-handoff.md); a coordinator cannot remove those conditions.
192
+ For design revisions, run fixed `task design-compare --id TASK --baseline PINNED_SOURCE
193
+ --file DECLARED_CANDIDATE` before the final review and supply its immutable receipt.
194
+
195
+ ## Reporting, external review and closure
196
+
197
+ ```text
198
+ node .claude/nccgs/tools/task.mjs report --id TASK --outcome blocked --summary "Actual blocker and missing evidence"
199
+ node .claude/nccgs/tools/task.mjs check --id TASK --target done
200
+ node .claude/nccgs/tools/task.mjs close --id TASK
201
+ ```
202
+
203
+ Reports accept `partial|blocked|failed` even when successful closure is unavailable.
204
+ Success requires verified current work and independent review before any report is
205
+ written. Generated Markdown includes evidence status and unresolved gates.
206
+ `record --stage report` can supply `artifacts` refs for required closure outputs.
207
+ Success wording does not override failed checks. Only `close` with all current gates
208
+ passing establishes DONE. `inspect` remains usable with changed/malformed policy.
209
+
210
+ `export-external --out .nccgs/evidence/external/requests/UNIQUE.json` writes an
211
+ immutable task/contract/snapshot packet. An external actor reviews it and imports
212
+ a receipt through `import-external --file FILE` outside the Claude executor:
213
+
214
+ ```json
215
+ {
216
+ "schemaVersion": 1,
217
+ "id": "REVIEW_ID",
218
+ "taskId": "TASK",
219
+ "contractHash": "EXPORTED_HASH",
220
+ "snapshot": "EXPORTED_SNAPSHOT",
221
+ "actor": "ACTUAL_EXTERNAL_REVIEWER",
222
+ "verdict": "PASS",
223
+ "findings": [],
224
+ "reviewedAt": "ACTUAL_ISO_TIMESTAMP",
225
+ "source": "ACTUAL_REVIEW_REFERENCE"
226
+ }
227
+ ```
228
+
229
+ Off ignores external closure dependency. Advisory retains current findings without
230
+ blocking. Required evaluates all active current receipts; a later empty PASS cannot
231
+ erase earlier blocking findings. To replace a prior decision, include
232
+ `supersedes: ["PRIOR_RECEIPT_ID"]` and `supersessionReason` identifying the new
233
+ evidence. Old findings remain traceable. Stale/duplicate/malformed receipts fail.
234
+
235
+ If acceptance is required, an external Product Owner decision references taskId,
236
+ contractHash, snapshot, `actor: "Product Owner"`, `decision: "ACCEPTED"`, decidedAt,
237
+ statement, reference and each manual gate `{id,decision:"ACCEPTED",reference}`.
238
+ Import via `record --stage acceptance` with its evidence ref. Claude cannot accept
239
+ on the user's behalf. NCCGS neither launches Astra nor sends external messages.
240
+
241
+ ## Recovery and telemetry
242
+
243
+ Supervised verification pins the owning run's absolute deadline in its request and
244
+ receipt. Its managed command tree ends at the earlier of that deadline and the
245
+ check timeout, even if the coordinator exits. A long per-attempt timeout never
246
+ extends the approved run. After expiry, existing request keys remain readable but
247
+ new attempts are refused. Do not escape this ownership through WMI or services.
248
+
249
+ Use `checkpoint --reason TEXT`, `resume`, `dispatch-status` and `telemetry` with
250
+ `--id TASK`. Resume preserves reservations, findings, attempts and budgets. A lost
251
+ callback or dead worker does not refund a slot by timeout. An external operator can
252
+ use `dispatch-reconcile --dispatch ID --request FILE` with `status: failed|cancelled`,
253
+ reason and observed reference after confirming the process is no longer running.
254
+ Never manually reset the ledger. Missing initialized history or altered durable
255
+ evidence fails closed. A crashed lock requires operator inspection before recovery.
256
+
257
+ Telemetry reports known subtotals and coverage for duration, tool calls, tokens and
258
+ cost. Total is null if any started attempt lacks that metric. Observed zero remains
259
+ zero. These are worker measurements; coordinator work, human delay and billing are
260
+ not automatically covered. A configured spend/time ceiling cannot interrupt an
261
+ already running provider request and refuses new dispatch with unknown measurements.
262
+
263
+ Native capture follows the documented Claude Code [hook inputs and Agent tool
264
+ response](https://code.claude.com/docs/en/hooks): resolvedModel/modelsUsed and native
265
+ duration/token/tool-use fields when present. It does not invent effort or cost.
266
+ Actual field availability and correlation must be confirmed by the real pilot.
267
+
268
+ ## Release boundary
269
+
270
+ Framework tests and the synthetic five-size benchmark validate mechanics. They
271
+ cannot measure model reasoning, game quality or end-to-end team productivity.
272
+ Use [the pilot protocol](standalone-pilot.md) before stable promotion. CP0 remains
273
+ reference-only; it is not a valid scored run for this purpose.
@@ -0,0 +1,144 @@
1
+ > Historical development/reference document. Version, status, commands and plans
2
+ > below describe their original scope, not the current package defaults. Read
3
+ > [current release guidance](../MEMORY.md) before use. Research/benchmark proposals
4
+ > do not automatically add requirements to a delivery contract.
5
+
6
+ # Claude Code pilot and benchmark protocol
7
+
8
+ Status: **prepared, not executed**. Framework version: **1.5.0-preview.1**.
9
+ The authorized current work is framework development. Game work resumes later
10
+ through Claude Code. Do not run a provider or change the game merely to fill this report.
11
+ Evaluation scope follows [delivery acceptance](delivery-acceptance.md): measure
12
+ brief fidelity, implementation, testing, review/repair and honest handoff. Balance
13
+ research is a separate product assignment; bot outcomes are not studio scores.
14
+
15
+ ## Provenance and clean starting point
16
+
17
+ Vector Survivors CP0 was built by Codex. Preserve it, label it contaminated, and
18
+ exclude it from scored orchestration/efficiency results. A copy of its solution
19
+ is not a fresh baseline. Prefer unseen tasks from later checkpoints or a new
20
+ equivalent foundation with withheld solution code. Record solution exposure and
21
+ exclude exposed tasks from independent quality/productivity claims.
22
+
23
+ Freeze framework/package hash, game starting snapshot, contract, Unity version,
24
+ machine, Claude Code version, permissions, model routes and available measurements
25
+ before each run. Use real Claude Code agent identities and durable raw observations.
26
+ Requested Sonnet/Opus aliases alone are insufficient. Label all synthetic fixture
27
+ data separately; never import it into a real task as runtime evidence.
28
+
29
+ ## Start with a small real pilot
30
+
31
+ 1. Install the local preview into a fresh disposable Unity project. Review dry-run
32
+ and doctor results; preserve the existing CP0 game untouched.
33
+ 2. Start Claude Code, confirm managed hooks load, and read NCCGS memory.
34
+ 3. Run one fresh FAST fix in foreground with one worker and one reviewer. Confirm
35
+ PreToolUse -> SubagentStart -> PostToolUse IDs actually correlate. Inspect raw
36
+ metadata, model history, missing effort/cost and spent attempts. Do not fabricate
37
+ fields when the client does not provide them.
38
+ 4. Run one audit and one plan task. Attempt no source mutation; verify their outputs
39
+ and source baseline. Check a failed/blocked report as well as successful closure.
40
+ 5. Test an interrupted foreground/background worker, external reconciliation and
41
+ resume. Confirm no duplicate agent or refunded budget. Check one explicit policy
42
+ migration/rollback on a copy with pending legacy tasks.
43
+ 6. Only after these mechanics pass, run the size benchmark below. A blocked runtime
44
+ adapter is a pilot finding, not a reason to substitute Codex orchestration.
45
+
46
+ ## Five task sizes in the survivors game
47
+
48
+ These are example *fresh* tasks. Freeze exact acceptance and withheld checks before
49
+ dispatch; do not reuse a solved task from CP0. Shapes and simple colors are enough.
50
+
51
+ | Size | Example | Expected route, subject to inspection | Required outcome |
52
+ |---|---|---|---|
53
+ | 1: tiny | Correct a HUD offset across fixed resolutions | implement / FAST if ready and local | Reference comparison, unchanged controls, one bounded fix |
54
+ | 2: small prototype | Add one experimental weapon idea with a simple upgrade | implement / STANDARD | Playable loop and explicit learning questions; no production gate inflation |
55
+ | 3: medium integration slice | Implement the specified timeline with progression, pause, loss/restart and feedback | implement / STANDARD; persistence raises risk | Deterministic state/timing tests, actual player integration checks, bug triage and a reproducible build; no bot-win requirement |
56
+ | 4: medium-large system | Redesign spawning/pooling/performance or plan save-compatible progression | implement or plan / CONTROLLED when cross-system | Measured before/after, architecture rationale, relevant regressions and compatibility |
57
+ | 5: large | Finish a large enemy/weapon/progression feature or audit the complete later game | implement or audit / CONTROLLED | Child contracts, integration/load tests, broad evidence and prioritized findings |
58
+
59
+ For size 3, exercise normal operations from controlled, disclosed preconditions;
60
+ check normal victory and defeat conditions independently of the bot's skill. A
61
+ ten-minute game timer does not require every state test to wait ten real minutes.
62
+ Verify shipped timer values and boundary logic, then cover the built player's
63
+ relevant integration paths. Do not inject the expected terminal state itself.
64
+ Human balance/feel studies, including participant count and sampling, belong to a
65
+ separate research brief; do not make five playtesters a default NCCGS delivery gate.
66
+ If a delivery brief explicitly requires human accessibility/usability inspection,
67
+ retain that scoped obligation and record human waiting separately.
68
+ For sizes 4–5, lock workload, seed where relevant, entity counts and
69
+ machine/settings before profiling; measure p50/p95/p99 frame time, GC allocations,
70
+ memory growth, restart stability, and relevant save/upgrade compatibility. Do not
71
+ infer frame performance from framework snapshot timing.
72
+
73
+ ## Compare fairly
74
+
75
+ Use two arms on matched, fresh work: plain Claude Code with a concise team workflow,
76
+ and Claude Code with NCCGS preview. Keep model access, task acceptance and tools
77
+ comparable; record actual route differences. Randomize order and use isolated
78
+ projects/sessions so the second arm does not see the first solution. Blind quality
79
+ review to the arm where practical. Astra is off for the primary benchmark; evaluate
80
+ advisory/required review as a separate experiment, recording added benefit and delay.
81
+
82
+ Start with two matched pairs per size (20 runs total) as an exploratory pilot. This
83
+ is not enough to claim broad statistical superiority. Extend to the larger sample
84
+ in the implementation plan only after runtime mechanics and task comparability pass.
85
+ Failures, retries, abandoned attempts and intervention remain in the denominator.
86
+ Do not replace failed tasks with easier ones or score only completed work.
87
+
88
+ ## Record useful efficiency and overthinking metrics
89
+
90
+ | Dimension | Measure | Interpretation |
91
+ |---|---|---|
92
+ | Correctness | Acceptance pass rate; severity-weighted escaped defects | A cheap but incorrect result is not efficient |
93
+ | Delivery | End-to-end elapsed, active executor time, human wait separately | Never compare worker time with another arm's whole task |
94
+ | Resource use | Observed tokens/cost per accepted outcome, coverage and billing source | Unknown is null; report partial coverage |
95
+ | Coordination | Dispatch count, useful specialist contributions, repeated reads, repeated unchanged reviews | Identify overhead that changed no decision or output |
96
+ | Rework | Failed checks, accepted/rejected findings, fix/review rounds | Count every attempt and actual recovery work |
97
+ | Autonomy | Human interventions and reasons; incorrect vs necessary escalations | Don't reward unsafe action or unexplained blocking |
98
+ | Maintainability | Independent assessment of cohesion, clarity, tests and extension cost | Blind review with concrete evidence |
99
+ | Resilience | Crash/resume, duplicate callback, missing evidence, budget conservation | No false success and no hidden attempt reset |
100
+
101
+ Pre-register what counts as unnecessary analysis: e.g. a repeated unchanged-source
102
+ review with no new evidence, repeated broad document loading with no decision
103
+ dependency, or speculative refactor outside acceptance. A long task alone does
104
+ not establish overthinking. Review whether extra reasoning prevented a real defect.
105
+
106
+ Compute quality-adjusted productivity only when scope is comparable: accepted
107
+ contract outcomes / total elapsed or observed spend. Show numerator, denominator,
108
+ coverage and uncertainty. Do not fabricate monetary savings from model names.
109
+ Preserve raw runs and per-size results; a single pooled average can hide tiny-task
110
+ overhead behind large-task gains.
111
+
112
+ ## Promotion gates
113
+
114
+ - **Preview -> stable 1.5:** real hook/identity/model evidence works on the supported
115
+ client; no known false-DONE, ownership, quota, migration or evidence-loss defect;
116
+ all five sizes have real results and documented failure recovery; scoped delivery
117
+ acceptance passes; measurements distinguish unknown values; owner accepts limitations.
118
+ - **Consider 2.0:** demonstrate useful autonomy and repeatable quality/efficiency
119
+ gains across the agreed workload, including small tasks; assess migration/API
120
+ compatibility and maintenance cost; explicitly approve major-version scope.
121
+
122
+ No version promotion is automatic merely because framework fixtures pass.
123
+
124
+ ## Run report template
125
+
126
+ ```text
127
+ Run ID / arm / task size / mode / risk:
128
+ Fresh baseline hash / solution-exposure declaration:
129
+ Framework package hash / client version / Unity / machine:
130
+ Contract + required checks + manual acceptance:
131
+ Requested routes / actual models / missing fields:
132
+ Dispatch IDs / all attempts / reviews / interventions:
133
+ Outcome / accepted criteria / open findings / escaped defects:
134
+ Requirement-to-code-to-test coverage / fixture interventions / independent review:
135
+ Failure attribution: product code / verification helper / bot / orchestration / research:
136
+ Separate product research questions and their status (not a delivery verdict):
137
+ Elapsed / worker time / human wait:
138
+ Tokens and cost / measurement coverage / raw source:
139
+ Game profiling data when applicable:
140
+ Unnecessary analysis observations and counterexamples:
141
+ Recovery actions / budget before and after:
142
+ Independent reviewer / Product Owner decision:
143
+ Limitations and next concrete action:
144
+ ```