@acrasie/dev-flow 0.0.0-stage → 1.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. package/.codex-plugin/plugin.json +20 -0
  2. package/LICENSE +21 -0
  3. package/README.md +181 -2
  4. package/dist/codex-dev-flow.mjs +3 -0
  5. package/dist/dev-flow.mjs +241 -0
  6. package/docs/adr/0001-hybrid-portable-workflow.md +23 -0
  7. package/docs/adr/0002-share-an-invalidable-context-capsule.md +55 -0
  8. package/docs/adr/0004-scale-assurance-lanes-by-applicable-risk.md +36 -0
  9. package/docs/adr/0006-make-intake-adaptive-user-authoritative-and-token-efficient.md +76 -0
  10. package/docs/adr/0007-collect-opt-in-local-benchmark-feedback.md +82 -0
  11. package/docs/adr/0008-automate-maintainer-releases-with-an-interactive-bun-workflow.md +121 -0
  12. package/docs/adr/0009-separate-intake-decisions-from-shape-discovery.md +200 -0
  13. package/docs/adr/0010-choose-quick-or-plan-after-discovery.md +161 -0
  14. package/docs/adr/0011-separate-fast-local-and-authoritative-ci-quality-gates.md +49 -0
  15. package/docs/adr/0012-use-bun-test-and-require-node-24.md +41 -0
  16. package/docs/adr/0013-layer-source-distribution-and-runtime-tests.md +42 -0
  17. package/docs/adr/0014-ratchet-source-coverage-with-bun.md +51 -0
  18. package/docs/adr/0015-split-fast-and-type-aware-linting.md +41 -0
  19. package/docs/adr/0016-use-husky-with-a-tested-bun-staged-file-adapter.md +45 -0
  20. package/docs/adr/0017-format-conservatively-with-oxfmt.md +45 -0
  21. package/docs/adr/0018-use-a-high-signal-oxlint-policy.md +53 -0
  22. package/docs/adr/0019-gate-deterministic-size-and-observe-timing.md +44 -0
  23. package/docs/adr/0020-support-linux-and-macos-with-targeted-ci.md +41 -0
  24. package/docs/adr/0021-randomize-tests-without-retries.md +35 -0
  25. package/docs/adr/0022-use-one-root-bun-workspace.md +41 -0
  26. package/docs/adr/0024-make-gate-a-minimal-plan-approval.md +74 -0
  27. package/docs/adr/0025-end-the-lifecycle-after-assure.md +55 -0
  28. package/docs/adr/0026-keep-intake-product-stable-and-interview-shape-by-dependency.md +151 -0
  29. package/docs/adr/0027-add-agentic-project-init-and-versioned-engineering-profiles.md +147 -0
  30. package/docs/adr/0028-make-public-documentation-user-first-and-current.md +65 -0
  31. package/docs/adr/0029-make-build-a-native-execution-boundary.md +51 -0
  32. package/docs/adr/0030-unify-product-domain-and-technical-design-interviews.md +240 -0
  33. package/docs/adr/0031-make-assure-the-success-boundary.md +205 -0
  34. package/docs/artifacts.md +47 -0
  35. package/docs/baselines/2026-07-18-p0-lifecycle.json +142 -0
  36. package/docs/design.md +101 -0
  37. package/docs/getting-started.md +204 -0
  38. package/docs/glossary/dev-flow.md +527 -0
  39. package/docs/lifecycle-contract.md +189 -0
  40. package/docs/lifecycle-contract.projection.json +931 -0
  41. package/docs/metrics-protocol.md +113 -0
  42. package/docs/project-profile-contract.md +157 -0
  43. package/docs/runbooks/maintainer-release.md +291 -0
  44. package/docs/target-intake-shape-contract.md +416 -0
  45. package/package.json +68 -4
  46. package/schemas/config.schema.json +104 -0
  47. package/schemas/policy.schema.json +17 -0
  48. package/schemas/project-init-state.schema.json +159 -0
  49. package/schemas/project-profile-local.schema.json +53 -0
  50. package/schemas/project-profile.schema.json +285 -0
  51. package/schemas/state.schema.json +826 -0
  52. package/skills/debug-root-cause/SKILL.md +16 -0
  53. package/skills/design-decisions/SKILL.md +24 -0
  54. package/skills/dev-flow/SKILL.md +306 -0
  55. package/skills/dev-flow/agents/openai.yaml +6 -0
  56. package/skills/discover-change/SKILL.md +31 -0
  57. package/skills/plan-change/SKILL.md +29 -0
  58. package/skills/review-change/SKILL.md +21 -0
@@ -0,0 +1,240 @@
1
+ # ADR 0030: Unify Product, Domain, and Technical Design Interviews
2
+
3
+ ## Status
4
+
5
+ Implemented in TaskState V6 with ADR 0031.
6
+
7
+ ## Context
8
+
9
+ Current INTAKE captures an adaptive brief and current SHAPE discovers repository facts,
10
+ but the workflow has no durable domain-modeling discipline and no ordinary interview for
11
+ material technical choices. ADR 0026 proposed a dependency-directed technical interview,
12
+ yet kept the dependency graph inside SHAPE and did not model terminology, business rules,
13
+ scenarios, documentation promotion, or a shared question boundary.
14
+
15
+ Useful design-interview practices include sharpening ambiguous language, confronting
16
+ claims with code and documentation, probing rules with edge cases, discovering facts
17
+ instead of asking the user, and traversing decisions by prerequisite. Copying an external
18
+ workflow would be wrong for Dev Flow: its lifecycle requires one active resumable
19
+ question, strict phase authority, deterministic state validation, read-only shaping,
20
+ digest-bound approval, and selective invalidation.
21
+
22
+ TaskState V5 shipped in v0.3.1 with the native BUILD boundary. ADR 0031 now requires a
23
+ V6 schema break for ASSURE. This target lands in the same coherent V6 transition so
24
+ SHAPE and the canonical Change Contract are rewritten once. Earlier local task state
25
+ fails explicitly; runtime provides no migration or dual-schema compatibility.
26
+
27
+ ## Decision
28
+
29
+ ### Use one lifecycle-bound semantic protocol
30
+
31
+ Add internal `design-decisions` skill. It is not an autonomous brainstorming command.
32
+ The parent Dev Flow skill supplies typed state, phase ownership, evidence, and current
33
+ frontier; the parent alone persists results.
34
+
35
+ The protocol:
36
+
37
+ - detects material ambiguity and decisions;
38
+ - proposes domain scenarios and technical alternatives;
39
+ - ranks the currently unblocked frontier;
40
+ - emits at most one active question;
41
+ - normalizes explicit answers into typed records; and
42
+ - declares semantic coverage when no material uncertainty remains.
43
+
44
+ The protocol does not require a second routine model call. Initial INTAKE evaluation
45
+ returns Product Baseline progress, domain deltas, frontier analysis, and an optional
46
+ question together. SHAPE Discovery returns evidence, technical decision candidates,
47
+ frontier analysis, and an optional question together. A dedicated reevaluation occurs
48
+ only after an answer, material invalidation, or new evidence.
49
+
50
+ ### Share orchestration, preserve authority
51
+
52
+ TaskState gains transversal `design` state containing one active question and one typed,
53
+ acyclic dependency graph. Payloads remain in their owning records:
54
+
55
+ - INTAKE owns `ProductBaseline`, terminology, business rules, domain scenarios, and
56
+ business-context boundaries needed by the task;
57
+ - SHAPE owns repository evidence and material technical decisions;
58
+ - the canonical Change Contract owns approved implementation and documentation work.
59
+
60
+ Question variants are `product`, `terminology`, `domain_rule`, and `technical`. Product
61
+ and domain questions run in `intaking` or `awaiting_intake_decision`; technical questions
62
+ run in new `awaiting_shape_decision`. Exactly one question may be active across the task.
63
+ Every question persists its owner, prerequisites, prior role, direct consequences,
64
+ recommendation data where applicable, and selection reason before display.
65
+
66
+ The full unblocked frontier remains internal. A model scores impact, irreversibility,
67
+ risk, uncertainty, and branches unlocked with bounded reasons. Runtime rejects blocked
68
+ or authority-invalid candidates, then applies stable deterministic tie-breaking. Resume
69
+ re-renders the exact selected question without another model call unless a dependency
70
+ changed.
71
+
72
+ ### Replace Intake Brief with a stable Product Baseline
73
+
74
+ `ProductBaseline` has five mandatory task-bounded dimensions:
75
+
76
+ 1. objective;
77
+ 2. affected users and product value;
78
+ 3. scope and boundaries;
79
+ 4. observable success and validator; and
80
+ 5. product risks and determining constraints.
81
+
82
+ Each dimension has value or justified `not_applicable`, provenance, status, and graph
83
+ dependencies. Missing and unknown are not complete. Explicit request content can resolve
84
+ dimensions without redundant prompts.
85
+
86
+ INTAKE performs targeted documentary discovery of applicable guidance, an approved
87
+ Project Engineering Profile, domain documents, ADRs, and directly relevant contracts.
88
+ It does not scan implementation generally. A question is required only when a product
89
+ or domain answer can change observable behavior, actors or permissions, an invariant or
90
+ exception, a bounded-context boundary, data meaning, success, scope, risk, or
91
+ compatibility.
92
+
93
+ When viable interpretations exist, questions present two or three non-dominated choices,
94
+ one recommendation, confidence, consequences, and free-form input. If choices would
95
+ invent unknown business meaning, use a targeted open elicitation question. A material
96
+ normalization of free-form input requires targeted confirmation; editorial rewriting
97
+ does not.
98
+
99
+ ### Model domain knowledge as three separate registries
100
+
101
+ The design state owns task-local candidate records:
102
+
103
+ - `DomainGlossaryDelta`: canonical contextual term, definition, accepted aliases,
104
+ rejected or deprecated terms with reasons, related terms, optional examples and
105
+ non-examples, provenance, status, and dependencies. It contains no implementation
106
+ detail.
107
+ - `DomainRule`: context, linked terms, actors, conditions, outcome or invariant,
108
+ exceptions and limits, provenance, status, dependencies, and child `DomainScenario`
109
+ records. Scenarios describe starting situation, action/event, expected or forbidden
110
+ result, and nominal, boundary, exception, or contradiction angle.
111
+ - `DecisionCandidate`: durable decision context, forces, viable and excluded
112
+ alternatives, evidence, recommendation, confidence, explicit selection,
113
+ consequences, ADR qualification, dependencies, status, and proposed destination.
114
+
115
+ Domain modeling activates by default only when a material trigger exists. It is bounded
116
+ to current task. Adjacent observations without material dependency do not enter the
117
+ graph or silently widen scope.
118
+
119
+ A domain rule becomes resolved through adaptive stress testing: positive behavior,
120
+ material boundary or counter-example, and confrontation with available evidence. A
121
+ trivial explicit rule may close without another user question when one strong source
122
+ proves it. A contradiction cannot close silently. Existing domain documentation is the
123
+ product baseline; code and tests prove current behavior, not desired behavior. The user
124
+ may explicitly change the baseline, producing dependent domain and technical deltas.
125
+
126
+ ### Interview material technical decisions inside SHAPE
127
+
128
+ `TechnicalDecision` replaces the narrow Decision Escalation record. It stores
129
+ materiality reason, prerequisites, evidence, viable and excluded options,
130
+ recommendation, confidence, consequences, selection, status, and exact resume role.
131
+
132
+ Architecture boundaries, major frameworks or dependencies, data or migration,
133
+ cross-service contracts, security/privacy, compatibility, concurrency, material
134
+ performance/cost/operations/rollback, and approved profile exceptions are material.
135
+ Repository-conforming reversible implementation details remain agent-owned.
136
+
137
+ Discovery obtains facts before questions. Planning cannot inspect the repository; a new
138
+ fact creates one targeted Discovery Target and a new material choice creates one
139
+ Technical Decision. Evidence proving product infeasibility creates a linked INTAKE
140
+ question rather than duplicating product authority inside SHAPE.
141
+
142
+ ### Make invalidation and completion deterministic
143
+
144
+ The task-level graph contains typed identities and dependencies only; registry payloads
145
+ are never copied into it. Runtime validates acyclicity, allowed ownership edges, one
146
+ active question, frontier eligibility, and transitive invalidation. Changing one answer
147
+ invalidates only dependent decisions, evidence, criteria, tasks, validations, risks, and
148
+ projections.
149
+
150
+ One response may resolve several explicitly answered nodes as one atomic transaction.
151
+ A material inferred normalization requires confirmation. Partial application is
152
+ forbidden. “Choose for me”, silence, deferral, or a request to advance never resolves a
153
+ material decision. User may answer, reduce scope so the branch disappears, or cancel.
154
+
155
+ Interview completion requires both a semantic declaration of no material uncertainty
156
+ and deterministic proof: all required nodes resolved, no active contradiction or stale
157
+ dependency, current evidence/profile, and an explicit disposition for every durable
158
+ domain delta. Planning may reopen only a targeted branch. No question cap, periodic
159
+ checkpoint, estimated-cost gate, or duplicate SHAPE approval exists.
160
+
161
+ ### Approve before projecting domain knowledge
162
+
163
+ Resolved domain records use this lifecycle:
164
+
165
+ ```text
166
+ proposed -> resolved -> approved -> projected
167
+ \-> invalidated
168
+ ```
169
+
170
+ Before GATE, all records remain structured task state. Each durable delta must have one
171
+ disposition: `project`, `already_documented`, `task_local`, or `policy_forbidden`, with
172
+ proof or reason. Material rules trace to acceptance criteria; observable scenarios are
173
+ validation candidates. Decision Candidates become ADR work only when all four gates
174
+ hold: durable beyond the task, materially costly to reverse, real trade-off among viable
175
+ alternatives, and surprising without recorded rationale.
176
+
177
+ GATE includes deltas and dispositions in the canonical contract digest. BUILD applies
178
+ approved documentation work through targeted semantic patches, never whole-file
179
+ regeneration. ASSURE verifies allowed paths, fingerprints, expected identities and
180
+ links, preservation of unrelated human content, and semantic fidelity. Successfully
181
+ projected repository documents become durable authority; task state retains references,
182
+ fingerprints, and receipts.
183
+
184
+ Destination resolution follows policy/configuration, approved Project Engineering
185
+ Profile, detected repository convention, then portable defaults. If no setup exists,
186
+ Dev Flow recommends `$dev-flow init` once but does not block unless policy or an
187
+ irreducible destination conflict requires it. Portable defaults are
188
+ `docs/domain/glossary.md`, `docs/domain/rules.md`, and `docs/adr/`, with context-specific
189
+ subdirectories and `docs/domain/context-map.md` when needed.
190
+
191
+ Project INIT gains adaptive `domain_documentation` profile data for context mode,
192
+ document paths, ADR location/format, language, identities, and link conventions. It
193
+ discovers existing conventions first and asks only unresolved durable choices.
194
+
195
+ ### Bound persistence, failures, visuals, and testing
196
+
197
+ Add routing stage `design`. Its route is resolved once from the immutable run snapshot
198
+ and reused by fused INTAKE/domain evaluations and SHAPE technical-decision evaluations.
199
+ Effort scales with risk, ambiguity, and dependency-graph depth. Discovery keeps its
200
+ separate fact-finding route; a short prompt never justifies silently substituting a less
201
+ capable design route. Routing identity and bounded usage counters persist without
202
+ decision content.
203
+
204
+ Persist normalized records, source references, exact active question, invalidations,
205
+ and compact selection reasons—never transcript, hidden reasoning, raw prompts, large
206
+ code excerpts, credentials, or external payloads. Stable IDs and enums use English;
207
+ questions use the user's dominant language; projections follow repository language.
208
+
209
+ Use the cheapest visual that materially resolves a decision: ASCII, Mermaid, isolated
210
+ HTML/CSS, then Excalidraw or generated image. Full graph is shown only on request or
211
+ when complexity makes it necessary. No fake progress percentage is derived from a
212
+ frontier that may grow.
213
+
214
+ Invalid semantic output receives deterministic syntax repair when meaning is unchanged,
215
+ otherwise one targeted retry. Second failure preserves previous state and creates a
216
+ typed block. Testing combines deterministic schema/DAG/frontier/transition/invalidation
217
+ tests, contractual fixtures, and versioned model evals for materiality, question quality,
218
+ scenario quality, recommendation quality, and non-invention.
219
+
220
+ ### Deliver as one coherent V6 contract
221
+
222
+ Implementation may proceed in verified internal layers, but public activation is
223
+ atomic across runtime, state schema, skills, normative contracts, projections, tests,
224
+ and distribution. TaskState advances to V6 jointly with ADR 0031. No compatibility path
225
+ or migration is provided for earlier local task state; legacy state fails with a clear
226
+ diagnostic. Obsolete structures and branches are removed rather than retained behind
227
+ compatibility code.
228
+
229
+ ## Consequences
230
+
231
+ - INTAKE becomes a task-bounded product and domain design boundary rather than a brief
232
+ field collector.
233
+ - SHAPE gains explicit user ownership for material technical choices without asking for
234
+ discoverable facts.
235
+ - Domain terminology, rules, scenarios, technical decisions, acceptance criteria, and
236
+ documentation share traceable dependencies without sharing payload ownership.
237
+ - Stronger state and eval requirements increase implementation size, but prevent
238
+ invisible assumptions, lossy resume, and uncontrolled documentation writes.
239
+ - The existing target in ADR 0026 is superseded. Current runtime remains authoritative
240
+ until this target lands atomically.
@@ -0,0 +1,205 @@
1
+ # ADR 0031: Make ASSURE the Success Boundary
2
+
3
+ Supersedes ADR 0025.
4
+
5
+ ## Status
6
+
7
+ Implemented in TaskState V6.
8
+
9
+ ## Context
10
+
11
+ Before TaskState V6, public lifecycle presented five macro-phases followed by `finished`:
12
+
13
+ ```text
14
+ INTAKE -> SHAPE -> GATE -> BUILD -> ASSURE -> finished
15
+ ```
16
+
17
+ `finished` is not a phase, but persisting it as a sixth visible lifecycle node duplicates
18
+ the success already proved by a valid Assurance Receipt. ASSURE also spreads its work
19
+ across `reviewing`, `resolving_assure`, and `verifying`, even though review, correction,
20
+ and verification are operations inside one user-facing phase. Re-running fresh BUILD
21
+ checks in ASSURE wastes tokens and time when no later write invalidated them.
22
+
23
+ ## Decision
24
+
25
+ The public lifecycle has exactly five phases and ends in ASSURE:
26
+
27
+ ```text
28
+ INTAKE -> SHAPE -> GATE -> BUILD -> ASSURE
29
+ ```
30
+
31
+ - Remove persisted `finished`. A valid current Assurance Receipt derives successful
32
+ completion; status and receipt cannot disagree.
33
+ - Replace `reviewing`, `resolving_assure`, and `verifying` with one internal `assuring`
34
+ status. Review, correction, and checks are operations, not states.
35
+ - Every acceptance criterion receives during SHAPE an expected outcome, verifier kind
36
+ (`command`, `model-inspection`, or `user-observation`), and evidence requirement.
37
+ - Checks executed during BUILD count when their command identity, result, and workspace
38
+ fingerprint remain current. A later relevant write invalidates their evidence.
39
+ - ASSURE runs only missing or invalidated checks. Deterministic checks run before any
40
+ required model review.
41
+ - Deterministic policy, never unconstrained agent choice, triggers model review from
42
+ verifier requirements, applicable risk, out-of-scope diff, or insufficient
43
+ deterministic evidence.
44
+ - Combine triggered model concerns into one bounded review call. Use separate specialist
45
+ passes only when policy explicitly requires expertise or isolation.
46
+ - Do not impose a global ASSURE correction budget. Track stable `issue_id` identities;
47
+ after two failed corrections for the same stable issue, invoke `debug-root-cause`.
48
+ Use `blocked` only for an external dependency and `failed` only for a demonstrated
49
+ irreparable outcome.
50
+ - Persist only a compact structured evidence index in TaskState: evidence identity,
51
+ covered subjects, verifier kind and identity, result, observed fingerprint, detail
52
+ reference, and digest. Keep command output and explanatory prose in the logical
53
+ artifact or logs. The Assurance Receipt summarizes the evidence index.
54
+ - Use `user-observation` only when proof cannot be automated. Batch all outstanding
55
+ human observations into one request and persist the explicit answer as evidence; a
56
+ model cannot substitute for the user.
57
+ - A policy-triggered model review uses one independent read-only reviewer with a fresh,
58
+ bounded capsule containing only triggered criteria, risks, diff, and evidence. The
59
+ implementing parent owns corrections; full conversation history is excluded.
60
+ - Keep the Assurance Receipt minimal: task identity, contract digest, verified
61
+ snapshot fingerprint, evidence-index digest, completion time, and receipt digest.
62
+ Runtime proves exact coverage before emission; the receipt does not duplicate
63
+ criteria, checks, or evidence references.
64
+ - Evidence freshness is conservative and hybrid. Scoped fingerprints are allowed only
65
+ for a scope declared deterministically by SHAPE or policy; all other evidence uses the
66
+ global workspace fingerprint. The final receipt always binds the complete snapshot.
67
+ - Completing ASSURE appends `assure.completed` without changing status from `assuring`.
68
+ A valid receipt makes that state terminal and immutable. Render the public boundary as
69
+ `[DEV FLOW · ASSURE ✓]`; do not introduce an `assured` or other success status.
70
+ - After success, return a bounded user summary: result, major changes, validation/review
71
+ performed, and material limits. Link to detailed evidence instead of repeating the
72
+ complete criterion-to-evidence map.
73
+ - Metrics and benchmark use derived outcome `success` when the receipt is valid. They do
74
+ not retain `finished` as an outcome alias; explicit irreparable termination remains
75
+ outcome `failed`.
76
+ - Atomically checkpoint the compact evidence index after every costly operation: command
77
+ check, model review, or human observation. Resume reuses each still-fresh checkpoint;
78
+ it does not wait until ASSURE completion or write redundant per-criterion checkpoints.
79
+ - Deduplicate deterministic work by stable check identity. Execute the minimal set of
80
+ checks covering all criteria plus repository/policy-mandatory global checks; one
81
+ evidence record may cover multiple subjects.
82
+ - Independent review returns a bounded structure only: verdict per subject and findings
83
+ containing stable `issue_id`, severity, location, evidence, and required correction.
84
+ It does not restate the plan, diff, or conversational history.
85
+ - Deterministic policy classifies blocking findings. Acceptance violations, correctness
86
+ regressions, security issues, and applicable-risk violations block completion. Style
87
+ or preference without an approved requirement is non-blocking and is not corrected
88
+ automatically; the reviewer cannot redefine this threshold.
89
+ - Before model review, classify out-of-scope diff deterministically. Correct an accidental
90
+ local reversible change natively; send a material contract change through SHAPE and
91
+ GATE; invoke review only when classification remains ambiguous.
92
+ - Execute deterministic checks through a policy/SHAPE-declared dependency DAG ordered by
93
+ cost and signal. Run cheap prerequisites first; a failure skips its dependants and
94
+ model review until corrected instead of spending work on knowingly stale inputs.
95
+ - Apply the same declared-scope invalidation to review evidence. A correction invalidates
96
+ overlapping review subjects; a globally fingerprinted review reruns after any write.
97
+ Unrelated scoped review evidence remains reusable.
98
+ - An approved acceptance criterion cannot become `not_applicable` inside ASSURE. A change
99
+ in applicability returns through SHAPE and GATE. Only a predeclared conditional check
100
+ may resolve `not_applicable`, with deterministic evidence for its condition.
101
+ - If the batched human-observation request receives no answer, persist the exact pending
102
+ request and enter `blocked` with resume target `assuring`. Resume re-renders it without
103
+ model work or rerunning still-fresh checks.
104
+ - A negative human observation creates a stable issue. Correct it locally when possible,
105
+ then request only invalidated observations again. User refusal or an unmet external
106
+ dependency blocks; green automated checks cannot override explicit negative evidence.
107
+ - Remove the routed `verify-change` model skill. SHAPE defines evidence requirements,
108
+ the parent executes deterministic commands, and runtime computes exact coverage and
109
+ emits the receipt. The only ASSURE model call is policy-triggered independent review;
110
+ no final model synthesizer restates deterministic evidence.
111
+ - Retain the `review-change` skill and `review` route, but reduce them to one consolidated,
112
+ structured, policy-triggered read-only review. Remove review lanes and the arbitrary
113
+ three-cycle cap; repeated stable issues follow the shared root-cause rule.
114
+ - ASSURE completion is an immutable historical success for its verified snapshot. Later
115
+ workspace changes neither reopen the task nor invalidate its receipt; further changes
116
+ require a new task and contract.
117
+ - An unplanned BUILD command counts toward required coverage only when runtime can match
118
+ its normalized identity and observed result exactly to a compatible predeclared
119
+ evidence requirement. Otherwise it remains supplemental evidence and cannot replace
120
+ required proof.
121
+ - Complete ASSURE in one atomic transaction: validate exact coverage, freeze the evidence
122
+ index, create the receipt, and append `assure.completed`. No persisted state may expose
123
+ an event without its receipt or a receipt without its event.
124
+ - Before GATE, SHAPE projects acceptance criteria, applicable risks, and mandatory global
125
+ checks into one coverage graph bound to the approved contract. ASSURE cannot invent a
126
+ late requirement; a material policy change invalidates approval and returns through
127
+ SHAPE and GATE.
128
+ - At completion, runtime deterministically selects the canonical current passing evidence
129
+ subset that covers the graph. The receipt binds this compact `completion index`;
130
+ supplemental observations remain referenced outside the receipt.
131
+ - SHAPE compiles the coverage graph before GATE into a canonical deduplicated operation
132
+ set and stable evidence preferences. ASSURE executes missing operations and applies
133
+ those preferences; it runs no exact or heuristic set-cover solver at completion.
134
+ - Checks in one dependency-DAG layer may run concurrently only when declared read-only
135
+ and concurrency-safe. Policy bounds concurrency; all other checks run sequentially.
136
+ - Do not inject full command output into model context. Passing checks return structured
137
+ identity/result only; failures return the shortest decisive excerpt plus a reference
138
+ to the complete log.
139
+ - Each operation declares a deterministic success predicate before GATE: exit code plus
140
+ any required output, artifact, or threshold assertion. Warnings block only when policy
141
+ explicitly classifies them; neither exit code alone nor free model interpretation is
142
+ universally sufficient.
143
+ - Normalize check identity from executable, argv, logical working directory, allowlisted
144
+ environment, and relevant configuration/tool-version fingerprints. Exclude secrets,
145
+ absolute local paths, volatile values, raw shell strings, and human display names from
146
+ canonical identity.
147
+ - Route independent review to the least-cost model qualified for applicable risk. Allow
148
+ one escalation only when its structured output is invalid or confidence falls below a
149
+ policy threshold; the implementing agent cannot freely select a larger model.
150
+ - Build review context deterministically from triggered subjects: related hunks or
151
+ symbols, direct dependencies, and summarized evidence. Never send the full diff merely
152
+ because slicing exceeds budget. If the bounded capsule remains too large, use only the
153
+ specialist passes prescribed by policy; never truncate arbitrarily.
154
+ - Snapshot before GATE the policy ceilings for review input, output, and call count.
155
+ Material policy changes invalidate approval; transient model/service failure remains
156
+ safely resumable inside ASSURE.
157
+ - Advance TaskState from V5 to V6 and fail older state explicitly. Do not add runtime
158
+ migration or dual-schema compatibility for removed ASSURE/success statuses.
159
+ - Implement ADR 0030 and this decision in one TaskState V6 schema transition because
160
+ both reshape SHAPE and the canonical Change Contract. Deliver independently verifiable
161
+ lots inside that change rather than causing two schema breaks and duplicate rewrites.
162
+ - Represent coverage compactly as stable `subjects`, `operations`, and `covers` lists,
163
+ with operation ordering in separate `dependsOn` edges. Persist identifiers and
164
+ deterministic contracts, not a repeated nested tree or narrative verification plan.
165
+ - Replace `verificationRequirements` with digest-bound `assuranceContract`, containing
166
+ the compact coverage graph, canonical operations, activation/success predicates,
167
+ budgets, and relevant policy snapshots.
168
+ - Make the canonical Change Contract the sole owner of `assuranceContract`. Repository
169
+ documentation is a derived projection; TaskState/runtime uses the canonical structure
170
+ and digest without maintaining a second narrative or independently editable copy.
171
+ - Precompile any diff-dependent review as a conditional operation before GATE. ASSURE
172
+ evaluates its deterministic activation predicate; activating it does not invent or
173
+ expand an approved requirement.
174
+ - Keep terminal computation closed and non-configurable: `cancelled` and `failed` are
175
+ terminal, as is `assuring` with the atomically paired valid receipt and completion
176
+ event. Do not add a generic conditional-terminal engine.
177
+ - Render completed work publicly as `ASSURE ✓` with derived result `success`. Keep raw
178
+ internal status `assuring` only in diagnostic JSON, not user-facing lifecycle copy.
179
+ - Verify the optimization contract with committed scenarios: zero false success in the
180
+ fault-injection suite; zero ASSURE model calls without a review trigger; one normal
181
+ consolidated review and at most one qualifying escalation; no rerun of current checks;
182
+ and enforced input/output token ceilings.
183
+ - Against the captured current baseline, require zero ASSURE model tokens on no-review
184
+ scenarios and at least 40% lower review input tokens on each review scenario, not only
185
+ as a favorable aggregate average.
186
+ - In CI, compare deterministic serialized capsules and call counts on fixtures. In a
187
+ controlled benchmark, record actual model-reported token usage for the same scenario
188
+ identities. Link both reports; do not make CI depend on live model calls or accept an
189
+ uncorrelated rough estimate as proof.
190
+ - Fault-injection coverage must reject success for stale evidence, missing subjects,
191
+ failed checks, false activation/success predicates, material scope drift, malformed
192
+ review output, negative human observation, and interrupted completion transactions.
193
+ - Risk and correctness controls remain mandatory, but no model review runs without a
194
+ policy-backed need.
195
+
196
+ ## Consequences
197
+
198
+ - Workflow success has one source of truth: current complete evidence represented by the
199
+ Assurance Receipt.
200
+ - Users see ASSURE complete rather than a misleading extra phase.
201
+ - ASSURE resume needs operation/evidence progress rather than multiple orchestration
202
+ statuses.
203
+ - Benchmark eligibility, metrics, terminal handling, schemas, projections, skills, and
204
+ tests must derive success from the receipt.
205
+ - Removing accepted statuses requires the explicit TaskState V6 schema break.
@@ -0,0 +1,47 @@
1
+ # Dev Flow Artifact Contract
2
+
3
+ ## Canonical owner
4
+
5
+ Persisted `ShapeState.contract` is sole logical owner of implementation contract. It
6
+ contains artifact identity/revision, Intake and Discovery receipt references, selected
7
+ profile, risk, scope, applicable non-goals, stable criteria, ordered tasks, validation
8
+ coverage, applicable risks/controls, dependency nodes, and canonical digest.
9
+
10
+ Quick and Plan share same schema:
11
+
12
+ - **Quick:** compact valid contract in state; minimum may be one criterion, one task, one
13
+ validation. No Markdown required.
14
+ - **Plan:** detailed contract plus one deterministic Markdown projection. Non-goals and
15
+ atomic dependencies required. Implementation Guide allowed only with evidence-linked
16
+ material trigger.
17
+
18
+ ## Projection
19
+
20
+ Parent renders Plan projection at safe repository-relative path, default
21
+ `docs/dev-flow/YYYY-MM-DD-<slug>.md`.
22
+
23
+ 1. Render `Proposed` from canonical state.
24
+ 2. Persist content fingerprint and logical contract digest.
25
+ 3. GATE approves logical digest.
26
+ 4. Mark projection `Approved` without changing logical content.
27
+ 5. Manual drift blocks overwrite/import and returns to Planning.
28
+ 6. Integrating material edit creates new digest and invalidates Approval.
29
+
30
+ Projection is never second source of truth.
31
+
32
+ ## Ownership and invalidation
33
+
34
+ - Discovery/Planning skills remain read-only; parent alone persists state/projection.
35
+ - Dependency DAG links decisions, sources, evidence, criteria, tasks, risks, validation,
36
+ guides, and projection.
37
+ - Change invalidates only transitive dependency closure. Shape is never reset wholesale.
38
+ - Resume reconciles fingerprints and continues first unproved targeted action.
39
+ - Approval binds gate, actor evidence, state revision/fingerprint, artifact identity, and
40
+ digest—not pathname or boolean.
41
+ - Benchmark authority is canonical Shape contract for both Quick and Plan.
42
+
43
+ ## Legacy layouts
44
+
45
+ Earlier `spec.md`, `plan.md`, `risk-and-rollback.md`, and `verification.md` layouts remain
46
+ repository choices only. They are non-canonical and cannot replace persisted contract
47
+ identity/digest.
@@ -0,0 +1,142 @@
1
+ {
2
+ "schemaVersion": 1,
3
+ "baselineId": "p0-taskstate-v6-contract-fixture",
4
+ "capturedAt": "2026-08-13T00:00:00.000Z",
5
+ "contractVersion": "P0",
6
+ "source": "contract_fixture",
7
+ "limitations": [
8
+ "contract_fixture_not_runtime",
9
+ "token_telemetry_unavailable",
10
+ "latency_telemetry_unavailable"
11
+ ],
12
+ "samples": [
13
+ {
14
+ "schemaVersion": 1,
15
+ "sampleIdHash": "26a1300d0c21109a930e6fde3448083cb271de41faa51b47ad82a9133c946274",
16
+ "preparationProfile": "quick",
17
+ "source": "contract_fixture",
18
+ "statuses": [
19
+ "new",
20
+ "intaking",
21
+ "discovering",
22
+ "awaiting_profile_choice",
23
+ "planning",
24
+ "awaiting_approval",
25
+ "executing",
26
+ "assuring"
27
+ ],
28
+ "phases": [
29
+ "INTAKE",
30
+ "SHAPE",
31
+ "GATE",
32
+ "BUILD",
33
+ "ASSURE"
34
+ ],
35
+ "transitions": 7,
36
+ "gates": 1,
37
+ "blocks": 0,
38
+ "resumes": 0,
39
+ "checksRun": 1,
40
+ "evidenceReused": 0,
41
+ "reviewCalls": 0,
42
+ "reviewEscalations": 0,
43
+ "correctionCycles": 0,
44
+ "outcome": "success",
45
+ "terminalStatus": "assuring",
46
+ "falseSuccessDetections": 0,
47
+ "falseSuccessDetectionSource": "contract_tests",
48
+ "tokenUsage": {
49
+ "availability": "unavailable"
50
+ },
51
+ "latency": {
52
+ "availability": "unavailable"
53
+ }
54
+ },
55
+ {
56
+ "schemaVersion": 1,
57
+ "sampleIdHash": "5016ee2afad0fa54c71fa8cf4dbb57afb85ea238cb68911db0e4f5875b99156c",
58
+ "preparationProfile": "plan",
59
+ "source": "contract_fixture",
60
+ "statuses": [
61
+ "new",
62
+ "intaking",
63
+ "discovering",
64
+ "awaiting_profile_choice",
65
+ "planning",
66
+ "awaiting_approval",
67
+ "executing",
68
+ "assuring"
69
+ ],
70
+ "phases": [
71
+ "INTAKE",
72
+ "SHAPE",
73
+ "GATE",
74
+ "BUILD",
75
+ "ASSURE"
76
+ ],
77
+ "transitions": 7,
78
+ "gates": 1,
79
+ "blocks": 0,
80
+ "resumes": 0,
81
+ "checksRun": 1,
82
+ "evidenceReused": 0,
83
+ "reviewCalls": 1,
84
+ "reviewEscalations": 0,
85
+ "correctionCycles": 0,
86
+ "outcome": "success",
87
+ "terminalStatus": "assuring",
88
+ "falseSuccessDetections": 0,
89
+ "falseSuccessDetectionSource": "contract_tests",
90
+ "tokenUsage": {
91
+ "availability": "unavailable"
92
+ },
93
+ "latency": {
94
+ "availability": "unavailable"
95
+ }
96
+ },
97
+ {
98
+ "schemaVersion": 1,
99
+ "sampleIdHash": "29192bd73047a146815ae66c3cad750165c10ff53551d222462a2380001b8ce2",
100
+ "preparationProfile": "plan",
101
+ "source": "contract_fixture",
102
+ "statuses": [
103
+ "new",
104
+ "intaking",
105
+ "discovering",
106
+ "awaiting_profile_choice",
107
+ "planning",
108
+ "awaiting_approval",
109
+ "executing",
110
+ "blocked",
111
+ "executing",
112
+ "assuring"
113
+ ],
114
+ "phases": [
115
+ "INTAKE",
116
+ "SHAPE",
117
+ "GATE",
118
+ "BUILD",
119
+ "ASSURE"
120
+ ],
121
+ "transitions": 9,
122
+ "gates": 1,
123
+ "blocks": 1,
124
+ "resumes": 1,
125
+ "checksRun": 1,
126
+ "evidenceReused": 1,
127
+ "reviewCalls": 1,
128
+ "reviewEscalations": 0,
129
+ "correctionCycles": 0,
130
+ "outcome": "success",
131
+ "terminalStatus": "assuring",
132
+ "falseSuccessDetections": 0,
133
+ "falseSuccessDetectionSource": "contract_tests",
134
+ "tokenUsage": {
135
+ "availability": "unavailable"
136
+ },
137
+ "latency": {
138
+ "availability": "unavailable"
139
+ }
140
+ }
141
+ ]
142
+ }