@acrasie/dev-flow 0.0.0-stage → 1.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.codex-plugin/plugin.json +20 -0
- package/LICENSE +21 -0
- package/README.md +181 -2
- package/dist/codex-dev-flow.mjs +3 -0
- package/dist/dev-flow.mjs +241 -0
- package/docs/adr/0001-hybrid-portable-workflow.md +23 -0
- package/docs/adr/0002-share-an-invalidable-context-capsule.md +55 -0
- package/docs/adr/0004-scale-assurance-lanes-by-applicable-risk.md +36 -0
- package/docs/adr/0006-make-intake-adaptive-user-authoritative-and-token-efficient.md +76 -0
- package/docs/adr/0007-collect-opt-in-local-benchmark-feedback.md +82 -0
- package/docs/adr/0008-automate-maintainer-releases-with-an-interactive-bun-workflow.md +121 -0
- package/docs/adr/0009-separate-intake-decisions-from-shape-discovery.md +200 -0
- package/docs/adr/0010-choose-quick-or-plan-after-discovery.md +161 -0
- package/docs/adr/0011-separate-fast-local-and-authoritative-ci-quality-gates.md +49 -0
- package/docs/adr/0012-use-bun-test-and-require-node-24.md +41 -0
- package/docs/adr/0013-layer-source-distribution-and-runtime-tests.md +42 -0
- package/docs/adr/0014-ratchet-source-coverage-with-bun.md +51 -0
- package/docs/adr/0015-split-fast-and-type-aware-linting.md +41 -0
- package/docs/adr/0016-use-husky-with-a-tested-bun-staged-file-adapter.md +45 -0
- package/docs/adr/0017-format-conservatively-with-oxfmt.md +45 -0
- package/docs/adr/0018-use-a-high-signal-oxlint-policy.md +53 -0
- package/docs/adr/0019-gate-deterministic-size-and-observe-timing.md +44 -0
- package/docs/adr/0020-support-linux-and-macos-with-targeted-ci.md +41 -0
- package/docs/adr/0021-randomize-tests-without-retries.md +35 -0
- package/docs/adr/0022-use-one-root-bun-workspace.md +41 -0
- package/docs/adr/0024-make-gate-a-minimal-plan-approval.md +74 -0
- package/docs/adr/0025-end-the-lifecycle-after-assure.md +55 -0
- package/docs/adr/0026-keep-intake-product-stable-and-interview-shape-by-dependency.md +151 -0
- package/docs/adr/0027-add-agentic-project-init-and-versioned-engineering-profiles.md +147 -0
- package/docs/adr/0028-make-public-documentation-user-first-and-current.md +65 -0
- package/docs/adr/0029-make-build-a-native-execution-boundary.md +51 -0
- package/docs/adr/0030-unify-product-domain-and-technical-design-interviews.md +240 -0
- package/docs/adr/0031-make-assure-the-success-boundary.md +205 -0
- package/docs/artifacts.md +47 -0
- package/docs/baselines/2026-07-18-p0-lifecycle.json +142 -0
- package/docs/design.md +101 -0
- package/docs/getting-started.md +204 -0
- package/docs/glossary/dev-flow.md +527 -0
- package/docs/lifecycle-contract.md +189 -0
- package/docs/lifecycle-contract.projection.json +931 -0
- package/docs/metrics-protocol.md +113 -0
- package/docs/project-profile-contract.md +157 -0
- package/docs/runbooks/maintainer-release.md +291 -0
- package/docs/target-intake-shape-contract.md +416 -0
- package/package.json +68 -4
- package/schemas/config.schema.json +104 -0
- package/schemas/policy.schema.json +17 -0
- package/schemas/project-init-state.schema.json +159 -0
- package/schemas/project-profile-local.schema.json +53 -0
- package/schemas/project-profile.schema.json +285 -0
- package/schemas/state.schema.json +826 -0
- package/skills/debug-root-cause/SKILL.md +16 -0
- package/skills/design-decisions/SKILL.md +24 -0
- package/skills/dev-flow/SKILL.md +306 -0
- package/skills/dev-flow/agents/openai.yaml +6 -0
- package/skills/discover-change/SKILL.md +31 -0
- package/skills/plan-change/SKILL.md +29 -0
- package/skills/review-change/SKILL.md +21 -0
|
@@ -0,0 +1,240 @@
|
|
|
1
|
+
# ADR 0030: Unify Product, Domain, and Technical Design Interviews
|
|
2
|
+
|
|
3
|
+
## Status
|
|
4
|
+
|
|
5
|
+
Implemented in TaskState V6 with ADR 0031.
|
|
6
|
+
|
|
7
|
+
## Context
|
|
8
|
+
|
|
9
|
+
Current INTAKE captures an adaptive brief and current SHAPE discovers repository facts,
|
|
10
|
+
but the workflow has no durable domain-modeling discipline and no ordinary interview for
|
|
11
|
+
material technical choices. ADR 0026 proposed a dependency-directed technical interview,
|
|
12
|
+
yet kept the dependency graph inside SHAPE and did not model terminology, business rules,
|
|
13
|
+
scenarios, documentation promotion, or a shared question boundary.
|
|
14
|
+
|
|
15
|
+
Useful design-interview practices include sharpening ambiguous language, confronting
|
|
16
|
+
claims with code and documentation, probing rules with edge cases, discovering facts
|
|
17
|
+
instead of asking the user, and traversing decisions by prerequisite. Copying an external
|
|
18
|
+
workflow would be wrong for Dev Flow: its lifecycle requires one active resumable
|
|
19
|
+
question, strict phase authority, deterministic state validation, read-only shaping,
|
|
20
|
+
digest-bound approval, and selective invalidation.
|
|
21
|
+
|
|
22
|
+
TaskState V5 shipped in v0.3.1 with the native BUILD boundary. ADR 0031 now requires a
|
|
23
|
+
V6 schema break for ASSURE. This target lands in the same coherent V6 transition so
|
|
24
|
+
SHAPE and the canonical Change Contract are rewritten once. Earlier local task state
|
|
25
|
+
fails explicitly; runtime provides no migration or dual-schema compatibility.
|
|
26
|
+
|
|
27
|
+
## Decision
|
|
28
|
+
|
|
29
|
+
### Use one lifecycle-bound semantic protocol
|
|
30
|
+
|
|
31
|
+
Add internal `design-decisions` skill. It is not an autonomous brainstorming command.
|
|
32
|
+
The parent Dev Flow skill supplies typed state, phase ownership, evidence, and current
|
|
33
|
+
frontier; the parent alone persists results.
|
|
34
|
+
|
|
35
|
+
The protocol:
|
|
36
|
+
|
|
37
|
+
- detects material ambiguity and decisions;
|
|
38
|
+
- proposes domain scenarios and technical alternatives;
|
|
39
|
+
- ranks the currently unblocked frontier;
|
|
40
|
+
- emits at most one active question;
|
|
41
|
+
- normalizes explicit answers into typed records; and
|
|
42
|
+
- declares semantic coverage when no material uncertainty remains.
|
|
43
|
+
|
|
44
|
+
The protocol does not require a second routine model call. Initial INTAKE evaluation
|
|
45
|
+
returns Product Baseline progress, domain deltas, frontier analysis, and an optional
|
|
46
|
+
question together. SHAPE Discovery returns evidence, technical decision candidates,
|
|
47
|
+
frontier analysis, and an optional question together. A dedicated reevaluation occurs
|
|
48
|
+
only after an answer, material invalidation, or new evidence.
|
|
49
|
+
|
|
50
|
+
### Share orchestration, preserve authority
|
|
51
|
+
|
|
52
|
+
TaskState gains transversal `design` state containing one active question and one typed,
|
|
53
|
+
acyclic dependency graph. Payloads remain in their owning records:
|
|
54
|
+
|
|
55
|
+
- INTAKE owns `ProductBaseline`, terminology, business rules, domain scenarios, and
|
|
56
|
+
business-context boundaries needed by the task;
|
|
57
|
+
- SHAPE owns repository evidence and material technical decisions;
|
|
58
|
+
- the canonical Change Contract owns approved implementation and documentation work.
|
|
59
|
+
|
|
60
|
+
Question variants are `product`, `terminology`, `domain_rule`, and `technical`. Product
|
|
61
|
+
and domain questions run in `intaking` or `awaiting_intake_decision`; technical questions
|
|
62
|
+
run in new `awaiting_shape_decision`. Exactly one question may be active across the task.
|
|
63
|
+
Every question persists its owner, prerequisites, prior role, direct consequences,
|
|
64
|
+
recommendation data where applicable, and selection reason before display.
|
|
65
|
+
|
|
66
|
+
The full unblocked frontier remains internal. A model scores impact, irreversibility,
|
|
67
|
+
risk, uncertainty, and branches unlocked with bounded reasons. Runtime rejects blocked
|
|
68
|
+
or authority-invalid candidates, then applies stable deterministic tie-breaking. Resume
|
|
69
|
+
re-renders the exact selected question without another model call unless a dependency
|
|
70
|
+
changed.
|
|
71
|
+
|
|
72
|
+
### Replace Intake Brief with a stable Product Baseline
|
|
73
|
+
|
|
74
|
+
`ProductBaseline` has five mandatory task-bounded dimensions:
|
|
75
|
+
|
|
76
|
+
1. objective;
|
|
77
|
+
2. affected users and product value;
|
|
78
|
+
3. scope and boundaries;
|
|
79
|
+
4. observable success and validator; and
|
|
80
|
+
5. product risks and determining constraints.
|
|
81
|
+
|
|
82
|
+
Each dimension has value or justified `not_applicable`, provenance, status, and graph
|
|
83
|
+
dependencies. Missing and unknown are not complete. Explicit request content can resolve
|
|
84
|
+
dimensions without redundant prompts.
|
|
85
|
+
|
|
86
|
+
INTAKE performs targeted documentary discovery of applicable guidance, an approved
|
|
87
|
+
Project Engineering Profile, domain documents, ADRs, and directly relevant contracts.
|
|
88
|
+
It does not scan implementation generally. A question is required only when a product
|
|
89
|
+
or domain answer can change observable behavior, actors or permissions, an invariant or
|
|
90
|
+
exception, a bounded-context boundary, data meaning, success, scope, risk, or
|
|
91
|
+
compatibility.
|
|
92
|
+
|
|
93
|
+
When viable interpretations exist, questions present two or three non-dominated choices,
|
|
94
|
+
one recommendation, confidence, consequences, and free-form input. If choices would
|
|
95
|
+
invent unknown business meaning, use a targeted open elicitation question. A material
|
|
96
|
+
normalization of free-form input requires targeted confirmation; editorial rewriting
|
|
97
|
+
does not.
|
|
98
|
+
|
|
99
|
+
### Model domain knowledge as three separate registries
|
|
100
|
+
|
|
101
|
+
The design state owns task-local candidate records:
|
|
102
|
+
|
|
103
|
+
- `DomainGlossaryDelta`: canonical contextual term, definition, accepted aliases,
|
|
104
|
+
rejected or deprecated terms with reasons, related terms, optional examples and
|
|
105
|
+
non-examples, provenance, status, and dependencies. It contains no implementation
|
|
106
|
+
detail.
|
|
107
|
+
- `DomainRule`: context, linked terms, actors, conditions, outcome or invariant,
|
|
108
|
+
exceptions and limits, provenance, status, dependencies, and child `DomainScenario`
|
|
109
|
+
records. Scenarios describe starting situation, action/event, expected or forbidden
|
|
110
|
+
result, and nominal, boundary, exception, or contradiction angle.
|
|
111
|
+
- `DecisionCandidate`: durable decision context, forces, viable and excluded
|
|
112
|
+
alternatives, evidence, recommendation, confidence, explicit selection,
|
|
113
|
+
consequences, ADR qualification, dependencies, status, and proposed destination.
|
|
114
|
+
|
|
115
|
+
Domain modeling activates by default only when a material trigger exists. It is bounded
|
|
116
|
+
to current task. Adjacent observations without material dependency do not enter the
|
|
117
|
+
graph or silently widen scope.
|
|
118
|
+
|
|
119
|
+
A domain rule becomes resolved through adaptive stress testing: positive behavior,
|
|
120
|
+
material boundary or counter-example, and confrontation with available evidence. A
|
|
121
|
+
trivial explicit rule may close without another user question when one strong source
|
|
122
|
+
proves it. A contradiction cannot close silently. Existing domain documentation is the
|
|
123
|
+
product baseline; code and tests prove current behavior, not desired behavior. The user
|
|
124
|
+
may explicitly change the baseline, producing dependent domain and technical deltas.
|
|
125
|
+
|
|
126
|
+
### Interview material technical decisions inside SHAPE
|
|
127
|
+
|
|
128
|
+
`TechnicalDecision` replaces the narrow Decision Escalation record. It stores
|
|
129
|
+
materiality reason, prerequisites, evidence, viable and excluded options,
|
|
130
|
+
recommendation, confidence, consequences, selection, status, and exact resume role.
|
|
131
|
+
|
|
132
|
+
Architecture boundaries, major frameworks or dependencies, data or migration,
|
|
133
|
+
cross-service contracts, security/privacy, compatibility, concurrency, material
|
|
134
|
+
performance/cost/operations/rollback, and approved profile exceptions are material.
|
|
135
|
+
Repository-conforming reversible implementation details remain agent-owned.
|
|
136
|
+
|
|
137
|
+
Discovery obtains facts before questions. Planning cannot inspect the repository; a new
|
|
138
|
+
fact creates one targeted Discovery Target and a new material choice creates one
|
|
139
|
+
Technical Decision. Evidence proving product infeasibility creates a linked INTAKE
|
|
140
|
+
question rather than duplicating product authority inside SHAPE.
|
|
141
|
+
|
|
142
|
+
### Make invalidation and completion deterministic
|
|
143
|
+
|
|
144
|
+
The task-level graph contains typed identities and dependencies only; registry payloads
|
|
145
|
+
are never copied into it. Runtime validates acyclicity, allowed ownership edges, one
|
|
146
|
+
active question, frontier eligibility, and transitive invalidation. Changing one answer
|
|
147
|
+
invalidates only dependent decisions, evidence, criteria, tasks, validations, risks, and
|
|
148
|
+
projections.
|
|
149
|
+
|
|
150
|
+
One response may resolve several explicitly answered nodes as one atomic transaction.
|
|
151
|
+
A material inferred normalization requires confirmation. Partial application is
|
|
152
|
+
forbidden. “Choose for me”, silence, deferral, or a request to advance never resolves a
|
|
153
|
+
material decision. User may answer, reduce scope so the branch disappears, or cancel.
|
|
154
|
+
|
|
155
|
+
Interview completion requires both a semantic declaration of no material uncertainty
|
|
156
|
+
and deterministic proof: all required nodes resolved, no active contradiction or stale
|
|
157
|
+
dependency, current evidence/profile, and an explicit disposition for every durable
|
|
158
|
+
domain delta. Planning may reopen only a targeted branch. No question cap, periodic
|
|
159
|
+
checkpoint, estimated-cost gate, or duplicate SHAPE approval exists.
|
|
160
|
+
|
|
161
|
+
### Approve before projecting domain knowledge
|
|
162
|
+
|
|
163
|
+
Resolved domain records use this lifecycle:
|
|
164
|
+
|
|
165
|
+
```text
|
|
166
|
+
proposed -> resolved -> approved -> projected
|
|
167
|
+
\-> invalidated
|
|
168
|
+
```
|
|
169
|
+
|
|
170
|
+
Before GATE, all records remain structured task state. Each durable delta must have one
|
|
171
|
+
disposition: `project`, `already_documented`, `task_local`, or `policy_forbidden`, with
|
|
172
|
+
proof or reason. Material rules trace to acceptance criteria; observable scenarios are
|
|
173
|
+
validation candidates. Decision Candidates become ADR work only when all four gates
|
|
174
|
+
hold: durable beyond the task, materially costly to reverse, real trade-off among viable
|
|
175
|
+
alternatives, and surprising without recorded rationale.
|
|
176
|
+
|
|
177
|
+
GATE includes deltas and dispositions in the canonical contract digest. BUILD applies
|
|
178
|
+
approved documentation work through targeted semantic patches, never whole-file
|
|
179
|
+
regeneration. ASSURE verifies allowed paths, fingerprints, expected identities and
|
|
180
|
+
links, preservation of unrelated human content, and semantic fidelity. Successfully
|
|
181
|
+
projected repository documents become durable authority; task state retains references,
|
|
182
|
+
fingerprints, and receipts.
|
|
183
|
+
|
|
184
|
+
Destination resolution follows policy/configuration, approved Project Engineering
|
|
185
|
+
Profile, detected repository convention, then portable defaults. If no setup exists,
|
|
186
|
+
Dev Flow recommends `$dev-flow init` once but does not block unless policy or an
|
|
187
|
+
irreducible destination conflict requires it. Portable defaults are
|
|
188
|
+
`docs/domain/glossary.md`, `docs/domain/rules.md`, and `docs/adr/`, with context-specific
|
|
189
|
+
subdirectories and `docs/domain/context-map.md` when needed.
|
|
190
|
+
|
|
191
|
+
Project INIT gains adaptive `domain_documentation` profile data for context mode,
|
|
192
|
+
document paths, ADR location/format, language, identities, and link conventions. It
|
|
193
|
+
discovers existing conventions first and asks only unresolved durable choices.
|
|
194
|
+
|
|
195
|
+
### Bound persistence, failures, visuals, and testing
|
|
196
|
+
|
|
197
|
+
Add routing stage `design`. Its route is resolved once from the immutable run snapshot
|
|
198
|
+
and reused by fused INTAKE/domain evaluations and SHAPE technical-decision evaluations.
|
|
199
|
+
Effort scales with risk, ambiguity, and dependency-graph depth. Discovery keeps its
|
|
200
|
+
separate fact-finding route; a short prompt never justifies silently substituting a less
|
|
201
|
+
capable design route. Routing identity and bounded usage counters persist without
|
|
202
|
+
decision content.
|
|
203
|
+
|
|
204
|
+
Persist normalized records, source references, exact active question, invalidations,
|
|
205
|
+
and compact selection reasons—never transcript, hidden reasoning, raw prompts, large
|
|
206
|
+
code excerpts, credentials, or external payloads. Stable IDs and enums use English;
|
|
207
|
+
questions use the user's dominant language; projections follow repository language.
|
|
208
|
+
|
|
209
|
+
Use the cheapest visual that materially resolves a decision: ASCII, Mermaid, isolated
|
|
210
|
+
HTML/CSS, then Excalidraw or generated image. Full graph is shown only on request or
|
|
211
|
+
when complexity makes it necessary. No fake progress percentage is derived from a
|
|
212
|
+
frontier that may grow.
|
|
213
|
+
|
|
214
|
+
Invalid semantic output receives deterministic syntax repair when meaning is unchanged,
|
|
215
|
+
otherwise one targeted retry. Second failure preserves previous state and creates a
|
|
216
|
+
typed block. Testing combines deterministic schema/DAG/frontier/transition/invalidation
|
|
217
|
+
tests, contractual fixtures, and versioned model evals for materiality, question quality,
|
|
218
|
+
scenario quality, recommendation quality, and non-invention.
|
|
219
|
+
|
|
220
|
+
### Deliver as one coherent V6 contract
|
|
221
|
+
|
|
222
|
+
Implementation may proceed in verified internal layers, but public activation is
|
|
223
|
+
atomic across runtime, state schema, skills, normative contracts, projections, tests,
|
|
224
|
+
and distribution. TaskState advances to V6 jointly with ADR 0031. No compatibility path
|
|
225
|
+
or migration is provided for earlier local task state; legacy state fails with a clear
|
|
226
|
+
diagnostic. Obsolete structures and branches are removed rather than retained behind
|
|
227
|
+
compatibility code.
|
|
228
|
+
|
|
229
|
+
## Consequences
|
|
230
|
+
|
|
231
|
+
- INTAKE becomes a task-bounded product and domain design boundary rather than a brief
|
|
232
|
+
field collector.
|
|
233
|
+
- SHAPE gains explicit user ownership for material technical choices without asking for
|
|
234
|
+
discoverable facts.
|
|
235
|
+
- Domain terminology, rules, scenarios, technical decisions, acceptance criteria, and
|
|
236
|
+
documentation share traceable dependencies without sharing payload ownership.
|
|
237
|
+
- Stronger state and eval requirements increase implementation size, but prevent
|
|
238
|
+
invisible assumptions, lossy resume, and uncontrolled documentation writes.
|
|
239
|
+
- The existing target in ADR 0026 is superseded. Current runtime remains authoritative
|
|
240
|
+
until this target lands atomically.
|
|
@@ -0,0 +1,205 @@
|
|
|
1
|
+
# ADR 0031: Make ASSURE the Success Boundary
|
|
2
|
+
|
|
3
|
+
Supersedes ADR 0025.
|
|
4
|
+
|
|
5
|
+
## Status
|
|
6
|
+
|
|
7
|
+
Implemented in TaskState V6.
|
|
8
|
+
|
|
9
|
+
## Context
|
|
10
|
+
|
|
11
|
+
Before TaskState V6, public lifecycle presented five macro-phases followed by `finished`:
|
|
12
|
+
|
|
13
|
+
```text
|
|
14
|
+
INTAKE -> SHAPE -> GATE -> BUILD -> ASSURE -> finished
|
|
15
|
+
```
|
|
16
|
+
|
|
17
|
+
`finished` is not a phase, but persisting it as a sixth visible lifecycle node duplicates
|
|
18
|
+
the success already proved by a valid Assurance Receipt. ASSURE also spreads its work
|
|
19
|
+
across `reviewing`, `resolving_assure`, and `verifying`, even though review, correction,
|
|
20
|
+
and verification are operations inside one user-facing phase. Re-running fresh BUILD
|
|
21
|
+
checks in ASSURE wastes tokens and time when no later write invalidated them.
|
|
22
|
+
|
|
23
|
+
## Decision
|
|
24
|
+
|
|
25
|
+
The public lifecycle has exactly five phases and ends in ASSURE:
|
|
26
|
+
|
|
27
|
+
```text
|
|
28
|
+
INTAKE -> SHAPE -> GATE -> BUILD -> ASSURE
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
- Remove persisted `finished`. A valid current Assurance Receipt derives successful
|
|
32
|
+
completion; status and receipt cannot disagree.
|
|
33
|
+
- Replace `reviewing`, `resolving_assure`, and `verifying` with one internal `assuring`
|
|
34
|
+
status. Review, correction, and checks are operations, not states.
|
|
35
|
+
- Every acceptance criterion receives during SHAPE an expected outcome, verifier kind
|
|
36
|
+
(`command`, `model-inspection`, or `user-observation`), and evidence requirement.
|
|
37
|
+
- Checks executed during BUILD count when their command identity, result, and workspace
|
|
38
|
+
fingerprint remain current. A later relevant write invalidates their evidence.
|
|
39
|
+
- ASSURE runs only missing or invalidated checks. Deterministic checks run before any
|
|
40
|
+
required model review.
|
|
41
|
+
- Deterministic policy, never unconstrained agent choice, triggers model review from
|
|
42
|
+
verifier requirements, applicable risk, out-of-scope diff, or insufficient
|
|
43
|
+
deterministic evidence.
|
|
44
|
+
- Combine triggered model concerns into one bounded review call. Use separate specialist
|
|
45
|
+
passes only when policy explicitly requires expertise or isolation.
|
|
46
|
+
- Do not impose a global ASSURE correction budget. Track stable `issue_id` identities;
|
|
47
|
+
after two failed corrections for the same stable issue, invoke `debug-root-cause`.
|
|
48
|
+
Use `blocked` only for an external dependency and `failed` only for a demonstrated
|
|
49
|
+
irreparable outcome.
|
|
50
|
+
- Persist only a compact structured evidence index in TaskState: evidence identity,
|
|
51
|
+
covered subjects, verifier kind and identity, result, observed fingerprint, detail
|
|
52
|
+
reference, and digest. Keep command output and explanatory prose in the logical
|
|
53
|
+
artifact or logs. The Assurance Receipt summarizes the evidence index.
|
|
54
|
+
- Use `user-observation` only when proof cannot be automated. Batch all outstanding
|
|
55
|
+
human observations into one request and persist the explicit answer as evidence; a
|
|
56
|
+
model cannot substitute for the user.
|
|
57
|
+
- A policy-triggered model review uses one independent read-only reviewer with a fresh,
|
|
58
|
+
bounded capsule containing only triggered criteria, risks, diff, and evidence. The
|
|
59
|
+
implementing parent owns corrections; full conversation history is excluded.
|
|
60
|
+
- Keep the Assurance Receipt minimal: task identity, contract digest, verified
|
|
61
|
+
snapshot fingerprint, evidence-index digest, completion time, and receipt digest.
|
|
62
|
+
Runtime proves exact coverage before emission; the receipt does not duplicate
|
|
63
|
+
criteria, checks, or evidence references.
|
|
64
|
+
- Evidence freshness is conservative and hybrid. Scoped fingerprints are allowed only
|
|
65
|
+
for a scope declared deterministically by SHAPE or policy; all other evidence uses the
|
|
66
|
+
global workspace fingerprint. The final receipt always binds the complete snapshot.
|
|
67
|
+
- Completing ASSURE appends `assure.completed` without changing status from `assuring`.
|
|
68
|
+
A valid receipt makes that state terminal and immutable. Render the public boundary as
|
|
69
|
+
`[DEV FLOW · ASSURE ✓]`; do not introduce an `assured` or other success status.
|
|
70
|
+
- After success, return a bounded user summary: result, major changes, validation/review
|
|
71
|
+
performed, and material limits. Link to detailed evidence instead of repeating the
|
|
72
|
+
complete criterion-to-evidence map.
|
|
73
|
+
- Metrics and benchmark use derived outcome `success` when the receipt is valid. They do
|
|
74
|
+
not retain `finished` as an outcome alias; explicit irreparable termination remains
|
|
75
|
+
outcome `failed`.
|
|
76
|
+
- Atomically checkpoint the compact evidence index after every costly operation: command
|
|
77
|
+
check, model review, or human observation. Resume reuses each still-fresh checkpoint;
|
|
78
|
+
it does not wait until ASSURE completion or write redundant per-criterion checkpoints.
|
|
79
|
+
- Deduplicate deterministic work by stable check identity. Execute the minimal set of
|
|
80
|
+
checks covering all criteria plus repository/policy-mandatory global checks; one
|
|
81
|
+
evidence record may cover multiple subjects.
|
|
82
|
+
- Independent review returns a bounded structure only: verdict per subject and findings
|
|
83
|
+
containing stable `issue_id`, severity, location, evidence, and required correction.
|
|
84
|
+
It does not restate the plan, diff, or conversational history.
|
|
85
|
+
- Deterministic policy classifies blocking findings. Acceptance violations, correctness
|
|
86
|
+
regressions, security issues, and applicable-risk violations block completion. Style
|
|
87
|
+
or preference without an approved requirement is non-blocking and is not corrected
|
|
88
|
+
automatically; the reviewer cannot redefine this threshold.
|
|
89
|
+
- Before model review, classify out-of-scope diff deterministically. Correct an accidental
|
|
90
|
+
local reversible change natively; send a material contract change through SHAPE and
|
|
91
|
+
GATE; invoke review only when classification remains ambiguous.
|
|
92
|
+
- Execute deterministic checks through a policy/SHAPE-declared dependency DAG ordered by
|
|
93
|
+
cost and signal. Run cheap prerequisites first; a failure skips its dependants and
|
|
94
|
+
model review until corrected instead of spending work on knowingly stale inputs.
|
|
95
|
+
- Apply the same declared-scope invalidation to review evidence. A correction invalidates
|
|
96
|
+
overlapping review subjects; a globally fingerprinted review reruns after any write.
|
|
97
|
+
Unrelated scoped review evidence remains reusable.
|
|
98
|
+
- An approved acceptance criterion cannot become `not_applicable` inside ASSURE. A change
|
|
99
|
+
in applicability returns through SHAPE and GATE. Only a predeclared conditional check
|
|
100
|
+
may resolve `not_applicable`, with deterministic evidence for its condition.
|
|
101
|
+
- If the batched human-observation request receives no answer, persist the exact pending
|
|
102
|
+
request and enter `blocked` with resume target `assuring`. Resume re-renders it without
|
|
103
|
+
model work or rerunning still-fresh checks.
|
|
104
|
+
- A negative human observation creates a stable issue. Correct it locally when possible,
|
|
105
|
+
then request only invalidated observations again. User refusal or an unmet external
|
|
106
|
+
dependency blocks; green automated checks cannot override explicit negative evidence.
|
|
107
|
+
- Remove the routed `verify-change` model skill. SHAPE defines evidence requirements,
|
|
108
|
+
the parent executes deterministic commands, and runtime computes exact coverage and
|
|
109
|
+
emits the receipt. The only ASSURE model call is policy-triggered independent review;
|
|
110
|
+
no final model synthesizer restates deterministic evidence.
|
|
111
|
+
- Retain the `review-change` skill and `review` route, but reduce them to one consolidated,
|
|
112
|
+
structured, policy-triggered read-only review. Remove review lanes and the arbitrary
|
|
113
|
+
three-cycle cap; repeated stable issues follow the shared root-cause rule.
|
|
114
|
+
- ASSURE completion is an immutable historical success for its verified snapshot. Later
|
|
115
|
+
workspace changes neither reopen the task nor invalidate its receipt; further changes
|
|
116
|
+
require a new task and contract.
|
|
117
|
+
- An unplanned BUILD command counts toward required coverage only when runtime can match
|
|
118
|
+
its normalized identity and observed result exactly to a compatible predeclared
|
|
119
|
+
evidence requirement. Otherwise it remains supplemental evidence and cannot replace
|
|
120
|
+
required proof.
|
|
121
|
+
- Complete ASSURE in one atomic transaction: validate exact coverage, freeze the evidence
|
|
122
|
+
index, create the receipt, and append `assure.completed`. No persisted state may expose
|
|
123
|
+
an event without its receipt or a receipt without its event.
|
|
124
|
+
- Before GATE, SHAPE projects acceptance criteria, applicable risks, and mandatory global
|
|
125
|
+
checks into one coverage graph bound to the approved contract. ASSURE cannot invent a
|
|
126
|
+
late requirement; a material policy change invalidates approval and returns through
|
|
127
|
+
SHAPE and GATE.
|
|
128
|
+
- At completion, runtime deterministically selects the canonical current passing evidence
|
|
129
|
+
subset that covers the graph. The receipt binds this compact `completion index`;
|
|
130
|
+
supplemental observations remain referenced outside the receipt.
|
|
131
|
+
- SHAPE compiles the coverage graph before GATE into a canonical deduplicated operation
|
|
132
|
+
set and stable evidence preferences. ASSURE executes missing operations and applies
|
|
133
|
+
those preferences; it runs no exact or heuristic set-cover solver at completion.
|
|
134
|
+
- Checks in one dependency-DAG layer may run concurrently only when declared read-only
|
|
135
|
+
and concurrency-safe. Policy bounds concurrency; all other checks run sequentially.
|
|
136
|
+
- Do not inject full command output into model context. Passing checks return structured
|
|
137
|
+
identity/result only; failures return the shortest decisive excerpt plus a reference
|
|
138
|
+
to the complete log.
|
|
139
|
+
- Each operation declares a deterministic success predicate before GATE: exit code plus
|
|
140
|
+
any required output, artifact, or threshold assertion. Warnings block only when policy
|
|
141
|
+
explicitly classifies them; neither exit code alone nor free model interpretation is
|
|
142
|
+
universally sufficient.
|
|
143
|
+
- Normalize check identity from executable, argv, logical working directory, allowlisted
|
|
144
|
+
environment, and relevant configuration/tool-version fingerprints. Exclude secrets,
|
|
145
|
+
absolute local paths, volatile values, raw shell strings, and human display names from
|
|
146
|
+
canonical identity.
|
|
147
|
+
- Route independent review to the least-cost model qualified for applicable risk. Allow
|
|
148
|
+
one escalation only when its structured output is invalid or confidence falls below a
|
|
149
|
+
policy threshold; the implementing agent cannot freely select a larger model.
|
|
150
|
+
- Build review context deterministically from triggered subjects: related hunks or
|
|
151
|
+
symbols, direct dependencies, and summarized evidence. Never send the full diff merely
|
|
152
|
+
because slicing exceeds budget. If the bounded capsule remains too large, use only the
|
|
153
|
+
specialist passes prescribed by policy; never truncate arbitrarily.
|
|
154
|
+
- Snapshot before GATE the policy ceilings for review input, output, and call count.
|
|
155
|
+
Material policy changes invalidate approval; transient model/service failure remains
|
|
156
|
+
safely resumable inside ASSURE.
|
|
157
|
+
- Advance TaskState from V5 to V6 and fail older state explicitly. Do not add runtime
|
|
158
|
+
migration or dual-schema compatibility for removed ASSURE/success statuses.
|
|
159
|
+
- Implement ADR 0030 and this decision in one TaskState V6 schema transition because
|
|
160
|
+
both reshape SHAPE and the canonical Change Contract. Deliver independently verifiable
|
|
161
|
+
lots inside that change rather than causing two schema breaks and duplicate rewrites.
|
|
162
|
+
- Represent coverage compactly as stable `subjects`, `operations`, and `covers` lists,
|
|
163
|
+
with operation ordering in separate `dependsOn` edges. Persist identifiers and
|
|
164
|
+
deterministic contracts, not a repeated nested tree or narrative verification plan.
|
|
165
|
+
- Replace `verificationRequirements` with digest-bound `assuranceContract`, containing
|
|
166
|
+
the compact coverage graph, canonical operations, activation/success predicates,
|
|
167
|
+
budgets, and relevant policy snapshots.
|
|
168
|
+
- Make the canonical Change Contract the sole owner of `assuranceContract`. Repository
|
|
169
|
+
documentation is a derived projection; TaskState/runtime uses the canonical structure
|
|
170
|
+
and digest without maintaining a second narrative or independently editable copy.
|
|
171
|
+
- Precompile any diff-dependent review as a conditional operation before GATE. ASSURE
|
|
172
|
+
evaluates its deterministic activation predicate; activating it does not invent or
|
|
173
|
+
expand an approved requirement.
|
|
174
|
+
- Keep terminal computation closed and non-configurable: `cancelled` and `failed` are
|
|
175
|
+
terminal, as is `assuring` with the atomically paired valid receipt and completion
|
|
176
|
+
event. Do not add a generic conditional-terminal engine.
|
|
177
|
+
- Render completed work publicly as `ASSURE ✓` with derived result `success`. Keep raw
|
|
178
|
+
internal status `assuring` only in diagnostic JSON, not user-facing lifecycle copy.
|
|
179
|
+
- Verify the optimization contract with committed scenarios: zero false success in the
|
|
180
|
+
fault-injection suite; zero ASSURE model calls without a review trigger; one normal
|
|
181
|
+
consolidated review and at most one qualifying escalation; no rerun of current checks;
|
|
182
|
+
and enforced input/output token ceilings.
|
|
183
|
+
- Against the captured current baseline, require zero ASSURE model tokens on no-review
|
|
184
|
+
scenarios and at least 40% lower review input tokens on each review scenario, not only
|
|
185
|
+
as a favorable aggregate average.
|
|
186
|
+
- In CI, compare deterministic serialized capsules and call counts on fixtures. In a
|
|
187
|
+
controlled benchmark, record actual model-reported token usage for the same scenario
|
|
188
|
+
identities. Link both reports; do not make CI depend on live model calls or accept an
|
|
189
|
+
uncorrelated rough estimate as proof.
|
|
190
|
+
- Fault-injection coverage must reject success for stale evidence, missing subjects,
|
|
191
|
+
failed checks, false activation/success predicates, material scope drift, malformed
|
|
192
|
+
review output, negative human observation, and interrupted completion transactions.
|
|
193
|
+
- Risk and correctness controls remain mandatory, but no model review runs without a
|
|
194
|
+
policy-backed need.
|
|
195
|
+
|
|
196
|
+
## Consequences
|
|
197
|
+
|
|
198
|
+
- Workflow success has one source of truth: current complete evidence represented by the
|
|
199
|
+
Assurance Receipt.
|
|
200
|
+
- Users see ASSURE complete rather than a misleading extra phase.
|
|
201
|
+
- ASSURE resume needs operation/evidence progress rather than multiple orchestration
|
|
202
|
+
statuses.
|
|
203
|
+
- Benchmark eligibility, metrics, terminal handling, schemas, projections, skills, and
|
|
204
|
+
tests must derive success from the receipt.
|
|
205
|
+
- Removing accepted statuses requires the explicit TaskState V6 schema break.
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
# Dev Flow Artifact Contract
|
|
2
|
+
|
|
3
|
+
## Canonical owner
|
|
4
|
+
|
|
5
|
+
Persisted `ShapeState.contract` is sole logical owner of implementation contract. It
|
|
6
|
+
contains artifact identity/revision, Intake and Discovery receipt references, selected
|
|
7
|
+
profile, risk, scope, applicable non-goals, stable criteria, ordered tasks, validation
|
|
8
|
+
coverage, applicable risks/controls, dependency nodes, and canonical digest.
|
|
9
|
+
|
|
10
|
+
Quick and Plan share same schema:
|
|
11
|
+
|
|
12
|
+
- **Quick:** compact valid contract in state; minimum may be one criterion, one task, one
|
|
13
|
+
validation. No Markdown required.
|
|
14
|
+
- **Plan:** detailed contract plus one deterministic Markdown projection. Non-goals and
|
|
15
|
+
atomic dependencies required. Implementation Guide allowed only with evidence-linked
|
|
16
|
+
material trigger.
|
|
17
|
+
|
|
18
|
+
## Projection
|
|
19
|
+
|
|
20
|
+
Parent renders Plan projection at safe repository-relative path, default
|
|
21
|
+
`docs/dev-flow/YYYY-MM-DD-<slug>.md`.
|
|
22
|
+
|
|
23
|
+
1. Render `Proposed` from canonical state.
|
|
24
|
+
2. Persist content fingerprint and logical contract digest.
|
|
25
|
+
3. GATE approves logical digest.
|
|
26
|
+
4. Mark projection `Approved` without changing logical content.
|
|
27
|
+
5. Manual drift blocks overwrite/import and returns to Planning.
|
|
28
|
+
6. Integrating material edit creates new digest and invalidates Approval.
|
|
29
|
+
|
|
30
|
+
Projection is never second source of truth.
|
|
31
|
+
|
|
32
|
+
## Ownership and invalidation
|
|
33
|
+
|
|
34
|
+
- Discovery/Planning skills remain read-only; parent alone persists state/projection.
|
|
35
|
+
- Dependency DAG links decisions, sources, evidence, criteria, tasks, risks, validation,
|
|
36
|
+
guides, and projection.
|
|
37
|
+
- Change invalidates only transitive dependency closure. Shape is never reset wholesale.
|
|
38
|
+
- Resume reconciles fingerprints and continues first unproved targeted action.
|
|
39
|
+
- Approval binds gate, actor evidence, state revision/fingerprint, artifact identity, and
|
|
40
|
+
digest—not pathname or boolean.
|
|
41
|
+
- Benchmark authority is canonical Shape contract for both Quick and Plan.
|
|
42
|
+
|
|
43
|
+
## Legacy layouts
|
|
44
|
+
|
|
45
|
+
Earlier `spec.md`, `plan.md`, `risk-and-rollback.md`, and `verification.md` layouts remain
|
|
46
|
+
repository choices only. They are non-canonical and cannot replace persisted contract
|
|
47
|
+
identity/digest.
|
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
{
|
|
2
|
+
"schemaVersion": 1,
|
|
3
|
+
"baselineId": "p0-taskstate-v6-contract-fixture",
|
|
4
|
+
"capturedAt": "2026-08-13T00:00:00.000Z",
|
|
5
|
+
"contractVersion": "P0",
|
|
6
|
+
"source": "contract_fixture",
|
|
7
|
+
"limitations": [
|
|
8
|
+
"contract_fixture_not_runtime",
|
|
9
|
+
"token_telemetry_unavailable",
|
|
10
|
+
"latency_telemetry_unavailable"
|
|
11
|
+
],
|
|
12
|
+
"samples": [
|
|
13
|
+
{
|
|
14
|
+
"schemaVersion": 1,
|
|
15
|
+
"sampleIdHash": "26a1300d0c21109a930e6fde3448083cb271de41faa51b47ad82a9133c946274",
|
|
16
|
+
"preparationProfile": "quick",
|
|
17
|
+
"source": "contract_fixture",
|
|
18
|
+
"statuses": [
|
|
19
|
+
"new",
|
|
20
|
+
"intaking",
|
|
21
|
+
"discovering",
|
|
22
|
+
"awaiting_profile_choice",
|
|
23
|
+
"planning",
|
|
24
|
+
"awaiting_approval",
|
|
25
|
+
"executing",
|
|
26
|
+
"assuring"
|
|
27
|
+
],
|
|
28
|
+
"phases": [
|
|
29
|
+
"INTAKE",
|
|
30
|
+
"SHAPE",
|
|
31
|
+
"GATE",
|
|
32
|
+
"BUILD",
|
|
33
|
+
"ASSURE"
|
|
34
|
+
],
|
|
35
|
+
"transitions": 7,
|
|
36
|
+
"gates": 1,
|
|
37
|
+
"blocks": 0,
|
|
38
|
+
"resumes": 0,
|
|
39
|
+
"checksRun": 1,
|
|
40
|
+
"evidenceReused": 0,
|
|
41
|
+
"reviewCalls": 0,
|
|
42
|
+
"reviewEscalations": 0,
|
|
43
|
+
"correctionCycles": 0,
|
|
44
|
+
"outcome": "success",
|
|
45
|
+
"terminalStatus": "assuring",
|
|
46
|
+
"falseSuccessDetections": 0,
|
|
47
|
+
"falseSuccessDetectionSource": "contract_tests",
|
|
48
|
+
"tokenUsage": {
|
|
49
|
+
"availability": "unavailable"
|
|
50
|
+
},
|
|
51
|
+
"latency": {
|
|
52
|
+
"availability": "unavailable"
|
|
53
|
+
}
|
|
54
|
+
},
|
|
55
|
+
{
|
|
56
|
+
"schemaVersion": 1,
|
|
57
|
+
"sampleIdHash": "5016ee2afad0fa54c71fa8cf4dbb57afb85ea238cb68911db0e4f5875b99156c",
|
|
58
|
+
"preparationProfile": "plan",
|
|
59
|
+
"source": "contract_fixture",
|
|
60
|
+
"statuses": [
|
|
61
|
+
"new",
|
|
62
|
+
"intaking",
|
|
63
|
+
"discovering",
|
|
64
|
+
"awaiting_profile_choice",
|
|
65
|
+
"planning",
|
|
66
|
+
"awaiting_approval",
|
|
67
|
+
"executing",
|
|
68
|
+
"assuring"
|
|
69
|
+
],
|
|
70
|
+
"phases": [
|
|
71
|
+
"INTAKE",
|
|
72
|
+
"SHAPE",
|
|
73
|
+
"GATE",
|
|
74
|
+
"BUILD",
|
|
75
|
+
"ASSURE"
|
|
76
|
+
],
|
|
77
|
+
"transitions": 7,
|
|
78
|
+
"gates": 1,
|
|
79
|
+
"blocks": 0,
|
|
80
|
+
"resumes": 0,
|
|
81
|
+
"checksRun": 1,
|
|
82
|
+
"evidenceReused": 0,
|
|
83
|
+
"reviewCalls": 1,
|
|
84
|
+
"reviewEscalations": 0,
|
|
85
|
+
"correctionCycles": 0,
|
|
86
|
+
"outcome": "success",
|
|
87
|
+
"terminalStatus": "assuring",
|
|
88
|
+
"falseSuccessDetections": 0,
|
|
89
|
+
"falseSuccessDetectionSource": "contract_tests",
|
|
90
|
+
"tokenUsage": {
|
|
91
|
+
"availability": "unavailable"
|
|
92
|
+
},
|
|
93
|
+
"latency": {
|
|
94
|
+
"availability": "unavailable"
|
|
95
|
+
}
|
|
96
|
+
},
|
|
97
|
+
{
|
|
98
|
+
"schemaVersion": 1,
|
|
99
|
+
"sampleIdHash": "29192bd73047a146815ae66c3cad750165c10ff53551d222462a2380001b8ce2",
|
|
100
|
+
"preparationProfile": "plan",
|
|
101
|
+
"source": "contract_fixture",
|
|
102
|
+
"statuses": [
|
|
103
|
+
"new",
|
|
104
|
+
"intaking",
|
|
105
|
+
"discovering",
|
|
106
|
+
"awaiting_profile_choice",
|
|
107
|
+
"planning",
|
|
108
|
+
"awaiting_approval",
|
|
109
|
+
"executing",
|
|
110
|
+
"blocked",
|
|
111
|
+
"executing",
|
|
112
|
+
"assuring"
|
|
113
|
+
],
|
|
114
|
+
"phases": [
|
|
115
|
+
"INTAKE",
|
|
116
|
+
"SHAPE",
|
|
117
|
+
"GATE",
|
|
118
|
+
"BUILD",
|
|
119
|
+
"ASSURE"
|
|
120
|
+
],
|
|
121
|
+
"transitions": 9,
|
|
122
|
+
"gates": 1,
|
|
123
|
+
"blocks": 1,
|
|
124
|
+
"resumes": 1,
|
|
125
|
+
"checksRun": 1,
|
|
126
|
+
"evidenceReused": 1,
|
|
127
|
+
"reviewCalls": 1,
|
|
128
|
+
"reviewEscalations": 0,
|
|
129
|
+
"correctionCycles": 0,
|
|
130
|
+
"outcome": "success",
|
|
131
|
+
"terminalStatus": "assuring",
|
|
132
|
+
"falseSuccessDetections": 0,
|
|
133
|
+
"falseSuccessDetectionSource": "contract_tests",
|
|
134
|
+
"tokenUsage": {
|
|
135
|
+
"availability": "unavailable"
|
|
136
|
+
},
|
|
137
|
+
"latency": {
|
|
138
|
+
"availability": "unavailable"
|
|
139
|
+
}
|
|
140
|
+
}
|
|
141
|
+
]
|
|
142
|
+
}
|