@dzhechkov/harness-core 0.8.36 → 0.8.37

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (106) hide show
  1. package/.dz-manifest.json +195 -75
  2. package/README.md +235 -8
  3. package/dist/agentdb-index.d.ts +87 -7
  4. package/dist/agentdb-index.d.ts.map +1 -1
  5. package/dist/agentdb-index.js +416 -57
  6. package/dist/agentdb-index.js.map +1 -1
  7. package/dist/apply-leg.d.ts +19 -1
  8. package/dist/apply-leg.d.ts.map +1 -1
  9. package/dist/apply-leg.js +187 -36
  10. package/dist/apply-leg.js.map +1 -1
  11. package/dist/codex-rollouts.d.ts +118 -0
  12. package/dist/codex-rollouts.d.ts.map +1 -0
  13. package/dist/codex-rollouts.js +297 -0
  14. package/dist/codex-rollouts.js.map +1 -0
  15. package/dist/cost-ledger.d.ts +56 -4
  16. package/dist/cost-ledger.d.ts.map +1 -1
  17. package/dist/cost-ledger.js +176 -20
  18. package/dist/cost-ledger.js.map +1 -1
  19. package/dist/cross-family-control.d.ts +345 -0
  20. package/dist/cross-family-control.d.ts.map +1 -0
  21. package/dist/cross-family-control.js +802 -0
  22. package/dist/cross-family-control.js.map +1 -0
  23. package/dist/debt-ratchet.d.ts +53 -0
  24. package/dist/debt-ratchet.d.ts.map +1 -0
  25. package/dist/debt-ratchet.js +107 -0
  26. package/dist/debt-ratchet.js.map +1 -0
  27. package/dist/embedding-config.d.ts +42 -0
  28. package/dist/embedding-config.d.ts.map +1 -1
  29. package/dist/embedding-config.js +106 -10
  30. package/dist/embedding-config.js.map +1 -1
  31. package/dist/feature-adr-checkpoints.d.ts +6 -0
  32. package/dist/feature-adr-checkpoints.d.ts.map +1 -1
  33. package/dist/feature-adr-checkpoints.js +29 -0
  34. package/dist/feature-adr-checkpoints.js.map +1 -1
  35. package/dist/feature-adr-decision-recall.d.ts +2 -2
  36. package/dist/feature-adr-decision-recall.d.ts.map +1 -1
  37. package/dist/feature-adr-decision-recall.js +5 -3
  38. package/dist/feature-adr-decision-recall.js.map +1 -1
  39. package/dist/feature-adr-envelope.d.ts +96 -0
  40. package/dist/feature-adr-envelope.d.ts.map +1 -0
  41. package/dist/feature-adr-envelope.js +183 -0
  42. package/dist/feature-adr-envelope.js.map +1 -0
  43. package/dist/feature-adr-routing.d.ts +64 -0
  44. package/dist/feature-adr-routing.d.ts.map +1 -1
  45. package/dist/feature-adr-routing.js +122 -2
  46. package/dist/feature-adr-routing.js.map +1 -1
  47. package/dist/feature-adr-stage-canon.d.ts +79 -0
  48. package/dist/feature-adr-stage-canon.d.ts.map +1 -0
  49. package/dist/feature-adr-stage-canon.js +117 -0
  50. package/dist/feature-adr-stage-canon.js.map +1 -0
  51. package/dist/index.d.ts +19 -9
  52. package/dist/index.d.ts.map +1 -1
  53. package/dist/index.js +13 -5
  54. package/dist/index.js.map +1 -1
  55. package/dist/loop-blobs.generated.js +4 -4
  56. package/dist/loop-blobs.generated.js.map +1 -1
  57. package/dist/mutation-gate.d.ts +51 -0
  58. package/dist/mutation-gate.d.ts.map +1 -1
  59. package/dist/mutation-gate.js +295 -0
  60. package/dist/mutation-gate.js.map +1 -1
  61. package/dist/qe-bridge.d.ts.map +1 -1
  62. package/dist/qe-bridge.js +4 -2
  63. package/dist/qe-bridge.js.map +1 -1
  64. package/dist/qe-findings.d.ts +107 -0
  65. package/dist/qe-findings.d.ts.map +1 -0
  66. package/dist/qe-findings.js +417 -0
  67. package/dist/qe-findings.js.map +1 -0
  68. package/dist/recap.d.ts +1 -1
  69. package/dist/recap.d.ts.map +1 -1
  70. package/dist/recap.js +4 -2
  71. package/dist/recap.js.map +1 -1
  72. package/dist/round.d.ts +74 -1
  73. package/dist/round.d.ts.map +1 -1
  74. package/dist/round.js +112 -4
  75. package/dist/round.js.map +1 -1
  76. package/dist/run-records.d.ts +60 -0
  77. package/dist/run-records.d.ts.map +1 -1
  78. package/dist/run-records.js +244 -2
  79. package/dist/run-records.js.map +1 -1
  80. package/dist/score.d.ts +44 -1
  81. package/dist/score.d.ts.map +1 -1
  82. package/dist/score.js +78 -5
  83. package/dist/score.js.map +1 -1
  84. package/package.json +1 -1
  85. package/sbom.json +374 -74
  86. package/src/agentdb-index.ts +423 -60
  87. package/src/apply-leg.ts +187 -36
  88. package/src/codex-rollouts.ts +374 -0
  89. package/src/cost-ledger.ts +232 -24
  90. package/src/cross-family-control.ts +960 -0
  91. package/src/debt-ratchet.ts +143 -0
  92. package/src/embedding-config.ts +131 -10
  93. package/src/feature-adr-checkpoints.ts +29 -0
  94. package/src/feature-adr-decision-recall.ts +6 -4
  95. package/src/feature-adr-envelope.ts +242 -0
  96. package/src/feature-adr-routing.ts +139 -2
  97. package/src/feature-adr-stage-canon.ts +141 -0
  98. package/src/index.ts +60 -6
  99. package/src/loop-blobs.generated.ts +4 -4
  100. package/src/mutation-gate.ts +316 -0
  101. package/src/qe-bridge.ts +4 -2
  102. package/src/qe-findings.ts +463 -0
  103. package/src/recap.ts +10 -3
  104. package/src/round.ts +165 -6
  105. package/src/run-records.ts +282 -2
  106. package/src/score.ts +115 -6
@@ -0,0 +1,242 @@
1
+ /**
2
+ * The experiment envelope (ADR-001, experiment-envelope). Pure data, built ONCE per run right after
3
+ * the Step-0 router and before Step 1, then carried unchanged into every autorow the pipeline writes
4
+ * (the run-cost ledger, training pairs, the round state). It answers two questions the pipeline used
5
+ * to leave unanswered: what STRATUM was this run (task kind, tier, priority) and what DECISION did
6
+ * routing make (which arms were considered, which one was chosen, by which policy, evaluated by whom).
7
+ *
8
+ * D1 (ADR-001): collecting these fields per-writer let three sources disagree — `mode` and `runId`
9
+ * already drifted across the ledger (133/343 and 34/343 respectively, MEASURED in Step 0). Building
10
+ * the envelope once and threading the same object through every writer removes that class of drift
11
+ * by construction.
12
+ */
13
+
14
+ export const ENVELOPE_SCHEMA = 1 as const;
15
+
16
+ export const TASK_KINDS = ['feature', 'bugfix', 'refactor', 'tooling', 'docs', 'research'] as const;
17
+ export type TaskKind = typeof TASK_KINDS[number];
18
+
19
+ export const PRIORITIES = ['speed', 'balance', 'quality', 'unset'] as const;
20
+ export type EnvelopePriority = typeof PRIORITIES[number];
21
+
22
+ export const TIERS = ['S', 'M', 'L', 'XL'] as const;
23
+ export type EnvelopeTier = typeof TIERS[number];
24
+
25
+ export interface ExperimentEnvelopeArms {
26
+ readonly mode: readonly string[];
27
+ readonly stages: Readonly<Record<string, readonly string[]>>;
28
+ }
29
+
30
+ export interface ExperimentEnvelopeChosen {
31
+ readonly mode: string;
32
+ readonly stages: Readonly<Record<string, string>>;
33
+ /** Lead delta after Codex r2 (HIGH): a stage whose chosen spec was NOT among the offered arms
34
+ * (usage-override, session-inherited fallback) is recorded HERE explicitly — arms stay the set that
35
+ * was actually offered; the winner is never appended to them after the fact. */
36
+ readonly overrides?: Readonly<Record<string, string>>;
37
+ }
38
+
39
+ export interface ExperimentEnvelopePolicy {
40
+ readonly name: string;
41
+ readonly version: string;
42
+ readonly propensity: number | null;
43
+ }
44
+
45
+ export interface ExperimentEnvelopeEvaluator {
46
+ readonly family: 'claude' | 'codex' | null;
47
+ readonly model: string | null;
48
+ readonly source: 'planned' | 'actual';
49
+ }
50
+
51
+ export interface ExperimentEnvelope {
52
+ readonly schema: 1;
53
+ readonly runId: string;
54
+ /**
55
+ * fix-round-1/F4: `null` when the persistent `.fa-state/attempt` counter probe failed — never a
56
+ * fabricated guess (the old `resumedStages.length > 0 ? 2 : 1` heuristic silently reported "2" for
57
+ * every third-and-later retry). Mirrors the `treeSha`/`treeShaReason` null+reason shape below.
58
+ */
59
+ readonly attempt: number | null;
60
+ /** Required (non-empty) exactly when `attempt` is null; null whenever `attempt` is a real count. */
61
+ readonly attemptReason: string | null;
62
+ readonly taskKind: TaskKind;
63
+ readonly tier: EnvelopeTier;
64
+ readonly priority: EnvelopePriority;
65
+ readonly treeSha: string | null;
66
+ readonly treeShaReason: string | null;
67
+ readonly arms: ExperimentEnvelopeArms;
68
+ readonly chosen: ExperimentEnvelopeChosen;
69
+ readonly policy: ExperimentEnvelopePolicy;
70
+ readonly evaluator: ExperimentEnvelopeEvaluator;
71
+ }
72
+
73
+ export interface BuildExperimentEnvelopeInput {
74
+ readonly runId: string;
75
+ readonly attempt: number | null;
76
+ readonly attemptReason?: string | null;
77
+ readonly taskKind: TaskKind;
78
+ readonly tier: EnvelopeTier;
79
+ readonly priority: EnvelopePriority;
80
+ readonly treeSha: string | null;
81
+ readonly treeShaReason?: string | null;
82
+ readonly arms: ExperimentEnvelopeArms;
83
+ readonly chosen: ExperimentEnvelopeChosen;
84
+ readonly policy: ExperimentEnvelopePolicy;
85
+ readonly evaluator: ExperimentEnvelopeEvaluator;
86
+ }
87
+
88
+ /**
89
+ * Assembles the normalized envelope object from already-resolved inputs. This function does not
90
+ * derive routing decisions itself (the caller — the Step-0-adjacent block in the workflow — resolves
91
+ * `arms`/`chosen`/`evaluator` from the routing tables); it only shapes the result consistently and
92
+ * fills the one field that has a computed default: `treeShaReason` is populated only when `treeSha`
93
+ * is null, and cleared when it is not.
94
+ */
95
+ export function buildExperimentEnvelope(input: BuildExperimentEnvelopeInput): ExperimentEnvelope {
96
+ return {
97
+ schema: ENVELOPE_SCHEMA,
98
+ runId: input.runId,
99
+ attempt: input.attempt,
100
+ attemptReason: input.attempt === null ? (input.attemptReason ?? 'unavailable') : null,
101
+ taskKind: input.taskKind,
102
+ tier: input.tier,
103
+ priority: input.priority,
104
+ treeSha: input.treeSha,
105
+ treeShaReason: input.treeSha === null ? (input.treeShaReason ?? 'unavailable') : null,
106
+ arms: { mode: [...input.arms.mode], stages: { ...input.arms.stages } },
107
+ chosen: { mode: input.chosen.mode, stages: { ...input.chosen.stages }, overrides: { ...(input.chosen.overrides ?? {}) } },
108
+ policy: { ...input.policy },
109
+ evaluator: { ...input.evaluator },
110
+ };
111
+ }
112
+
113
+ const HEX40 = /^[0-9a-f]{40}$/i;
114
+
115
+ function isPlainObject(v: unknown): v is Record<string, unknown> {
116
+ return v !== null && typeof v === 'object' && !Array.isArray(v);
117
+ }
118
+
119
+ function isNonEmptyString(v: unknown): v is string {
120
+ return typeof v === 'string' && v.trim() !== '';
121
+ }
122
+
123
+ /**
124
+ * Validates an envelope value field by field, IN ORDER, and returns the FIRST invalid field by name
125
+ * (never a batch of errors — the refusal channel this feeds, `run-records.ts` FR-5, prints one reason
126
+ * line and that line must name something actionable).
127
+ */
128
+ export function validateExperimentEnvelope(value: unknown): { ok: true } | { ok: false; reason: string } {
129
+ if (!isPlainObject(value)) return { ok: false, reason: 'envelope: expected an object' };
130
+ const v = value;
131
+
132
+ if (v.schema !== ENVELOPE_SCHEMA) {
133
+ return { ok: false, reason: `schema: expected ${ENVELOPE_SCHEMA}, got ${JSON.stringify(v.schema)}` };
134
+ }
135
+ if (!isNonEmptyString(v.runId)) {
136
+ return { ok: false, reason: 'runId: expected a non-empty string' };
137
+ }
138
+ if (v.attempt !== null) {
139
+ if (typeof v.attempt !== 'number' || !Number.isInteger(v.attempt) || v.attempt < 1) {
140
+ return { ok: false, reason: 'attempt: expected an integer >= 1 or null' };
141
+ }
142
+ } else if (!isNonEmptyString(v.attemptReason)) {
143
+ return { ok: false, reason: 'attemptReason: required (non-empty) when attempt is null' };
144
+ }
145
+ if (typeof v.taskKind !== 'string' || !(TASK_KINDS as readonly string[]).includes(v.taskKind)) {
146
+ return { ok: false, reason: `taskKind: expected one of ${TASK_KINDS.join('|')}, got ${JSON.stringify(v.taskKind)}` };
147
+ }
148
+ if (typeof v.tier !== 'string' || !(TIERS as readonly string[]).includes(v.tier)) {
149
+ return { ok: false, reason: `tier: expected one of ${TIERS.join('|')}, got ${JSON.stringify(v.tier)}` };
150
+ }
151
+ if (typeof v.priority !== 'string' || !(PRIORITIES as readonly string[]).includes(v.priority)) {
152
+ return { ok: false, reason: `priority: expected one of ${PRIORITIES.join('|')}, got ${JSON.stringify(v.priority)}` };
153
+ }
154
+ if (v.treeSha !== null) {
155
+ if (typeof v.treeSha !== 'string' || !HEX40.test(v.treeSha)) {
156
+ return { ok: false, reason: 'treeSha: expected 40 hex chars or null' };
157
+ }
158
+ } else if (!isNonEmptyString(v.treeShaReason)) {
159
+ return { ok: false, reason: 'treeShaReason: required (non-empty) when treeSha is null' };
160
+ }
161
+
162
+ if (!isPlainObject(v.arms)) return { ok: false, reason: 'arms: expected an object' };
163
+ const arms = v.arms;
164
+ if (!Array.isArray(arms.mode) || arms.mode.length === 0 || !arms.mode.every((m) => isNonEmptyString(m))) {
165
+ return { ok: false, reason: 'arms.mode: expected a non-empty array of non-empty strings' };
166
+ }
167
+ const armsMode = arms.mode as readonly string[];
168
+ if (!isPlainObject(arms.stages)) return { ok: false, reason: 'arms.stages: expected an object' };
169
+ const armsStageEntries = Object.entries(arms.stages);
170
+ if (armsStageEntries.length === 0) return { ok: false, reason: 'arms.stages: expected at least one stage' };
171
+ for (const [stage, specs] of armsStageEntries) {
172
+ // F7: an empty stage NAME (a real, if odd, JS object key) is refused too — a stage nobody can
173
+ // name is a stage nobody can dispatch to member-check against.
174
+ if (stage.trim() === '') return { ok: false, reason: 'arms.stages: stage name must not be empty' };
175
+ if (!Array.isArray(specs) || specs.length === 0 || !specs.every((s) => isNonEmptyString(s))) {
176
+ return { ok: false, reason: `arms.stages.${stage}: expected a non-empty array of non-empty strings` };
177
+ }
178
+ }
179
+
180
+ if (!isPlainObject(v.chosen)) return { ok: false, reason: 'chosen: expected an object' };
181
+ const chosen = v.chosen;
182
+ if (!isNonEmptyString(chosen.mode)) return { ok: false, reason: 'chosen.mode: expected a non-empty string' };
183
+ // F7 (fix-round-1): a valid-SHAPED row can still describe an IMPOSSIBLE decision — `chosen.mode`
184
+ // naming an option `arms.mode` never offered, or a stage's chosen spec absent from what that
185
+ // stage's own arms offered. Membership is checked AFTER shape, so a shape error is still reported
186
+ // first (the more actionable message).
187
+ if (!armsMode.includes(chosen.mode)) {
188
+ return { ok: false, reason: `chosen.mode: "${chosen.mode}" is not a member of arms.mode (${armsMode.join('|')})` };
189
+ }
190
+ if (!isPlainObject(chosen.stages)) return { ok: false, reason: 'chosen.stages: expected an object' };
191
+ const chosenStageEntries = Object.entries(chosen.stages);
192
+ if (chosenStageEntries.length === 0) return { ok: false, reason: 'chosen.stages: expected at least one stage' };
193
+ for (const [stage, spec] of chosenStageEntries) {
194
+ if (stage.trim() === '') return { ok: false, reason: 'chosen.stages: stage name must not be empty' };
195
+ if (!isNonEmptyString(spec)) return { ok: false, reason: `chosen.stages.${stage}: expected a non-empty string` };
196
+ }
197
+ const armsStageNames = new Set(Object.keys(arms.stages as Record<string, unknown>));
198
+ const chosenStageNames = new Set(Object.keys(chosen.stages as Record<string, unknown>));
199
+ if (armsStageNames.size !== chosenStageNames.size || ![...armsStageNames].every((s) => chosenStageNames.has(s))) {
200
+ return { ok: false, reason: 'chosen.stages: stage set disagrees with arms.stages' };
201
+ }
202
+ const armsStages = arms.stages as Record<string, readonly string[]>;
203
+ // Lead delta after Codex r2 (HIGH): arms are IMMUTABLE — the offered set. A chosen spec outside it is
204
+ // legal only when `chosen.overrides` names that stage with the SAME spec (an explicit, auditable
205
+ // "chosen outside the offered arms"), never by appending the winner to arms after the fact.
206
+ const overridesRaw = chosen.overrides === undefined ? {} : chosen.overrides;
207
+ if (!isPlainObject(overridesRaw)) return { ok: false, reason: 'chosen.overrides: expected an object when present' };
208
+ const overrides = overridesRaw as Record<string, unknown>;
209
+ for (const [stage, spec] of Object.entries(overrides)) {
210
+ if (!(stage in (chosen.stages as Record<string, unknown>))) return { ok: false, reason: `chosen.overrides.${stage}: names a stage absent from chosen.stages` };
211
+ if (!isNonEmptyString(spec)) return { ok: false, reason: `chosen.overrides.${stage}: expected a non-empty string` };
212
+ if ((chosen.stages as Record<string, unknown>)[stage] !== spec) return { ok: false, reason: `chosen.overrides.${stage}: "${String(spec)}" disagrees with chosen.stages.${stage}` };
213
+ }
214
+ for (const [stage, spec] of chosenStageEntries) {
215
+ const offered = armsStages[stage]!;
216
+ if (!offered.includes(spec as string) && overrides[stage] !== spec) {
217
+ return { ok: false, reason: `chosen.stages.${stage}: "${String(spec)}" is not a member of arms.stages.${stage} (${offered.join('|')}) and not declared in chosen.overrides` };
218
+ }
219
+ }
220
+
221
+ if (!isPlainObject(v.policy)) return { ok: false, reason: 'policy: expected an object' };
222
+ const policy = v.policy;
223
+ if (!isNonEmptyString(policy.name)) return { ok: false, reason: 'policy.name: expected a non-empty string' };
224
+ if (!isNonEmptyString(policy.version)) return { ok: false, reason: 'policy.version: expected a non-empty string' };
225
+ if (policy.propensity !== null && typeof policy.propensity !== 'number') {
226
+ return { ok: false, reason: 'policy.propensity: expected a number or null' };
227
+ }
228
+
229
+ if (!isPlainObject(v.evaluator)) return { ok: false, reason: 'evaluator: expected an object' };
230
+ const evaluator = v.evaluator;
231
+ if (evaluator.family !== null && evaluator.family !== 'claude' && evaluator.family !== 'codex') {
232
+ return { ok: false, reason: 'evaluator.family: expected claude, codex, or null' };
233
+ }
234
+ if (evaluator.model !== null && typeof evaluator.model !== 'string') {
235
+ return { ok: false, reason: 'evaluator.model: expected a string or null' };
236
+ }
237
+ if (evaluator.source !== 'planned' && evaluator.source !== 'actual') {
238
+ return { ok: false, reason: 'evaluator.source: expected planned or actual' };
239
+ }
240
+
241
+ return { ok: true };
242
+ }
@@ -259,6 +259,57 @@ export function budgetPresetName(axis: BudgetAxis): 'normal' | 'eco' | 'hybrid'
259
259
  return null;
260
260
  }
261
261
 
262
+ /**
263
+ * `args.priority` — a LABEL for the run's learning stratum (ADR-001 D3), applied as a preset one
264
+ * level ABOVE `budget`/`deliveryGate`: an explicit knob always wins. `speed` trades quality for
265
+ * cost; `quality` turns on the delivery gate; `balance` is today's default made nameable. Not yet a
266
+ * full selector (no calibration data) — see the ADR for what is deliberately out of scope.
267
+ */
268
+ export type PriorityName = 'speed' | 'balance' | 'quality';
269
+ export type EnvelopePriorityValue = PriorityName | 'unset';
270
+ export interface PriorityPreset {
271
+ readonly budget: 'normal' | 'eco';
272
+ readonly deliveryGate: boolean;
273
+ }
274
+
275
+ export const PRIORITY_PRESETS: Record<PriorityName, PriorityPreset> = {
276
+ speed: { budget: 'eco', deliveryGate: false },
277
+ balance: { budget: 'normal', deliveryGate: false },
278
+ quality: { budget: 'normal', deliveryGate: true },
279
+ };
280
+
281
+ /** Unknown values return a named `{error}` rather than silently collapsing to `'unset'` (FR-4). */
282
+ export function resolvePriority(raw: unknown): EnvelopePriorityValue | { readonly error: string } {
283
+ // fix-round-1/F5: the literal string 'unset' is accepted as an explicit "not set" — the SAME
284
+ // result as the argument being absent. Before this fix the error message listed 'unset' among the
285
+ // valid values while passing exactly that string was refused (a self-contradicting error).
286
+ if (raw === undefined || raw === null || raw === 'unset') return 'unset';
287
+ if (typeof raw === 'string' && Object.prototype.hasOwnProperty.call(PRIORITY_PRESETS, raw)) {
288
+ return raw as PriorityName;
289
+ }
290
+ const valid = Object.keys(PRIORITY_PRESETS).join('|') + '|unset';
291
+ return { error: 'priority: unknown "' + String(raw) + '" — valid: ' + valid };
292
+ }
293
+
294
+ /**
295
+ * Applies the priority preset UNDER explicit `budget`/`deliveryGate` knobs — an explicit value
296
+ * always wins, the preset only fills what the caller left unspecified. `routingRequested` reports
297
+ * whether a non-`unset` priority alone should turn routing on (NFR-1: without `args.priority` this
298
+ * stays `false`, byte-identical to today).
299
+ */
300
+ export function applyPriorityPreset(
301
+ priority: EnvelopePriorityValue,
302
+ explicit: { readonly budget?: unknown; readonly deliveryGate?: unknown },
303
+ ): { readonly budget: unknown; readonly deliveryGate: unknown; readonly routingRequested: boolean } {
304
+ if (priority === 'unset') {
305
+ return { budget: explicit.budget, deliveryGate: explicit.deliveryGate, routingRequested: false };
306
+ }
307
+ const preset = PRIORITY_PRESETS[priority];
308
+ const budget = explicit.budget !== undefined ? explicit.budget : preset.budget;
309
+ const deliveryGate = explicit.deliveryGate !== undefined ? explicit.deliveryGate : preset.deliveryGate;
310
+ return { budget, deliveryGate, routingRequested: true };
311
+ }
312
+
262
313
  /** The Claude model names the Workflow runtime accepts as `agent()` `model`. */
263
314
  export const CLAUDE_NAMES: Record<string, number> = { fable: 1, opus: 1, sonnet: 1, haiku: 1 };
264
315
 
@@ -1998,7 +2049,12 @@ export function scopedQePrompt(input: ScopedQePromptInput): string {
1998
2049
  if (slug !== '') out += ' They are the changed files of feature ' + slug + '.';
1999
2050
  out += '\n\nAnswer these ' + questions.length + ' questions about them:\n';
2000
2051
  for (let i = 0; i < questions.length; i++) out += i + 1 + '. ' + questions[i] + '\n';
2001
- out += '\nFinish with a single final line: Grade: <A|B|C|D>';
2052
+ // qe-findings-record fix-round-1 finding 8: this is the mirror `export function scopedQePrompt`
2053
+ // of the SAME function inlined in `.claude/workflows/feature-adr.js` (and its byte-identical twin)
2054
+ // — the pipeline gained the QE-VERDICT tail so `dz score`/`dz recap` can read a Codex-graded run
2055
+ // machine-readably, and this mirror had drifted (still asking for only `Grade:`). Text pinned
2056
+ // identical to the pipeline by test/codex-scoped-review.test.ts and test/qe-findings-wiring.test.ts.
2057
+ out += '\nFinish with two final lines: Grade: <A|B|C|D> and QE-VERDICT: <same>';
2002
2058
  return out;
2003
2059
  }
2004
2060
 
@@ -2504,6 +2560,13 @@ export interface PlanGateCmdOpts {
2504
2560
  gateScript?: string
2505
2561
  /** Pin the workspace root instead of resolving it live with `pwd -P` (tests; deterministic pins). */
2506
2562
  workspace?: string
2563
+ /**
2564
+ * FR-2 (ADR-001 plan-inherits-requirements): promotes C8 (01_requirements.md id coverage) from a
2565
+ * WARN-with-count to a per-id FAIL. Opt-in per caller — the pipeline's one call site sets it; the
2566
+ * bare/positional 3-arg form and any opts object without this field stay byte-identical to the
2567
+ * pre-C8 command.
2568
+ */
2569
+ requireRequirements?: boolean
2507
2570
  }
2508
2571
 
2509
2572
  /** Both interpolated knobs are shell-spliced, so both get the same build-time shape check. */
@@ -2538,6 +2601,11 @@ function assertAbsoluteNoTraversal(value: unknown, knob: string): string {
2538
2601
  export function planCompletenessGateCmd(repo: string, featureDir: string, tier?: string | null, opts?: PlanGateCmdOpts): string {
2539
2602
  const q = (s: string) => "'" + String(s).replace(/'/g, "'\\''") + "'"
2540
2603
  const t = (typeof tier === 'string' && tier !== '') ? ' --tier=' + q(tier) : ''
2604
+ // FR-2 (ADR-001 plan-inherits-requirements): opt-in per caller via opts.requireRequirements, not
2605
+ // baked into every invocation — the pipeline's one call site sets it (below), so C8 (01_requirements.md
2606
+ // coverage) is a per-id FAIL there; a caller that omits opts, or opts without the flag, is unaffected —
2607
+ // the bare/positional form stays byte-identical to the pre-C8 command.
2608
+ const req = (opts && opts.requireRequirements) ? ' --require-requirements' : ''
2541
2609
  if (opts === undefined || opts === null) {
2542
2610
  return 'cd ' + q(repo) + ' && node ' + q(PLAN_GATE_SCRIPT) + ' ' + q(featureDir) + t + ' 2>&1; echo K2_EXIT=$?'
2543
2611
  }
@@ -2561,7 +2629,7 @@ export function planCompletenessGateCmd(repo: string, featureDir: string, tier?:
2561
2629
  // twice and the chain silently degenerated from three candidates to two. Saying so turns a
2562
2630
  // puzzling duplicate into an instruction. Not verdict-shaped, so the parser anchoring is untouched.
2563
2631
  '[ "$C2" = "$C3" ] && echo "K2_GATE_NOTE=the workspace candidate resolved to the TARGET repo (WS==repo), so only two distinct candidates were tried; pass args.workspace or args.gateScript when the feature-adr skill is installed outside the target repo"',
2564
- 'if [ -z "$GS" ]; then echo "K2 plan-completeness: NOT-ESTABLISHED — tooling-missing: no gate script at any candidate on the K2_GATE_TRIED line above"; echo "K2_EXIT=3"; else cd ' + q(repo) + ' && node "$GS" ' + q(featureDir) + t + ' 2>&1; echo "K2_EXIT=$?"; fi',
2632
+ 'if [ -z "$GS" ]; then echo "K2 plan-completeness: NOT-ESTABLISHED — tooling-missing: no gate script at any candidate on the K2_GATE_TRIED line above"; echo "K2_EXIT=3"; else cd ' + q(repo) + ' && node "$GS" ' + q(featureDir) + t + req + ' 2>&1; echo "K2_EXIT=$?"; fi',
2565
2633
  ].join('\n')
2566
2634
  }
2567
2635
 
@@ -2599,6 +2667,75 @@ export function parsePlanGateVerdict(raw: string | null | undefined): PlanGateVe
2599
2667
  return { verdict: byName, exit: exitCode, reason: reason, output: output }
2600
2668
  }
2601
2669
 
2670
+ /** Snapshot of a plan taken around the ONE repair round (FR-6 preservation check, plan-inherits-requirements). */
2671
+ export interface PlanSnapshot { present: boolean; len: number; cksum: number | null; headings: string[]; targets: string[] }
2672
+
2673
+ /**
2674
+ * FR-6 preservation check — pure halves of the plan-repair round, mirrored INLINE in
2675
+ * `.claude/workflows/feature-adr.js` (the sandbox cannot import) and body-pinned by the drift guard in
2676
+ * `test/feature-adr-model-routing.test.ts`. See the workflow's own comment block above these functions
2677
+ * for the design, the Codex round-2/3 findings each one answers, and the NAMED LIMIT (task bodies are
2678
+ * not proven preserved by any metric).
2679
+ */
2680
+ export function shellQuote(s: string): string {
2681
+ return "'" + String(s).replace(/'/g, "'\\''") + "'";
2682
+ }
2683
+ export function planBackupCmd(f: string): string {
2684
+ return 'f=' + shellQuote(f) + '; if [ -f "$f" ] && [ ! -L "$f" ] && [ -s "$f" ] && [ ! -e "$f.pre-repair" ] && cp "$f" "$f.pre-repair"; then echo "BACKUP_OK"; else echo "BACKUP_FAILED"; fi'
2685
+ }
2686
+ export function planRestoreCmd(f: string): string {
2687
+ return 'f=' + shellQuote(f) + '; if [ -f "$f.pre-repair" ] && [ ! -L "$f.pre-repair" ] && [ -s "$f.pre-repair" ] && mv "$f.pre-repair" "$f"; then echo "RESTORE_OK"; else echo "RESTORE_FAILED"; fi'
2688
+ }
2689
+ export function planArchiveBackupCmd(f: string, stateDir: string): string {
2690
+ return 'f=' + shellQuote(f) + '; d=' + shellQuote(stateDir) + '; if [ -f "$f.pre-repair" ] && [ ! -L "$f.pre-repair" ] && mkdir -p "$d" && mv "$f.pre-repair" "$d/06_implementation_plan.pre-repair"; then echo "ARCHIVE_OK"; else echo "ARCHIVE_FAILED"; fi'
2691
+ }
2692
+ export function planSnapshotCmd(f: string): string {
2693
+ return 'f=' + shellQuote(f) + '; if [ -f "$f" ] && [ ! -L "$f" ] && [ -s "$f" ]; then ' +
2694
+ 'len=$(wc -c < "$f" | tr -d " "); ck=$(cksum < "$f" | cut -d " " -f 1); ' +
2695
+ 'echo "SNAP_TARGETS_START"; ' +
2696
+ 'awk \'/^EXPECTED_CODE_TARGETS:/{f=1;next} f&&/^[ \\t]*[-*][ \\t]/{s=$0; sub(/^[ \\t]*[-*][ \\t]*/,"",s); sub(/[ \\t]+$/,"",s); print s; next} f&&/^[ \\t]*$/{next} f{exit}\' "$f"; ' +
2697
+ 'echo "SNAP_TARGETS_END"; ' +
2698
+ 'echo "SNAP_HEADS_START"; grep -E "^#{2,4}[[:space:]]" "$f"; echo "SNAP_HEADS_END"; ' +
2699
+ 'echo "SNAP_LEN=$len"; echo "SNAP_CKSUM=$ck"; echo "SNAP_DONE"; ' +
2700
+ 'else echo "SNAP_ABSENT"; echo "SNAP_DONE"; fi'
2701
+ }
2702
+ /** One marker-delimited block: WHOLE-LINE markers, FIRST start to LAST end (a plan line spelling a marker lands inside the block). */
2703
+ export function snapshotBlock(head: string, startMark: string, endMark: string): string[] | null {
2704
+ const startRe = new RegExp('(^|\n)' + startMark + '(\r?\n)')
2705
+ const sm = startRe.exec(head)
2706
+ if (sm === null) return null
2707
+ const s = sm.index + sm[0].length
2708
+ const endRe = new RegExp('(^|\n)' + endMark + '(\r?\n|$)', 'g')
2709
+ let e = -1
2710
+ let em = endRe.exec(head)
2711
+ while (em !== null) { e = em.index + (em[1] === '' ? 0 : 1); em = endRe.exec(head) }
2712
+ if (e < 0 || e < s) return null
2713
+ return head.slice(s, e).split('\n').map((l) => l.trim()).filter((l) => l !== '')
2714
+ }
2715
+ /** The LAST whole-line `<name>=<digits>` in the snapshot. */
2716
+ export function snapshotNumber(head: string, name: string): number | null {
2717
+ const re = new RegExp('(^|\n)' + name + '=(\\d+)[ \t\r]*(\n|$)', 'g')
2718
+ let v = null
2719
+ let m = re.exec(head)
2720
+ while (m !== null) { v = Number(m[2]); m = re.exec(head) }
2721
+ return v
2722
+ }
2723
+ /** null = the probe never completed or a field is unparseable — the caller REJECTS the repair on null, never reads it as "nothing to compare". */
2724
+ export function parsePlanSnapshot(raw: string | null | undefined): PlanSnapshot | null {
2725
+ const text = String(raw === null || raw === undefined ? '' : raw)
2726
+ const doneIdx = text.lastIndexOf('SNAP_DONE')
2727
+ if (doneIdx < 0) return null
2728
+ const head = text.slice(0, doneIdx)
2729
+ if (/(^|\n)SNAP_ABSENT[ \t\r]*\n?[ \t\r]*$/.test(head)) return { present: false, len: 0, cksum: null, headings: [], targets: [] }
2730
+ const len = snapshotNumber(head, 'SNAP_LEN')
2731
+ const cksum = snapshotNumber(head, 'SNAP_CKSUM')
2732
+ if (len === null || cksum === null) return null
2733
+ const targets = snapshotBlock(head, 'SNAP_TARGETS_START', 'SNAP_TARGETS_END')
2734
+ const headings = snapshotBlock(head, 'SNAP_HEADS_START', 'SNAP_HEADS_END')
2735
+ if (targets === null || headings === null) return null
2736
+ return { present: true, len: len, cksum: cksum, headings: headings, targets: targets }
2737
+ }
2738
+
2602
2739
  /**
2603
2740
  * The operator note a refused plan gate carries — ONE reason→text table, so the workflow's inline
2604
2741
  * copy cannot drift into telling an operator to fix a plan that is not broken.
@@ -0,0 +1,141 @@
1
+ /**
2
+ * The canonical stage taxonomy (feature `measurement-integrity`, ADR-001 D1).
3
+ *
4
+ * `cost-ledger.ts` keys its per-stage rows off `stageLabel()` output VERBATIM — deliberately, so the
5
+ * ledger never invents its own taxonomy (the prior feature's FR-2). That is correct for the ledger's
6
+ * own scope, but it leaves nothing to GROUP by: one recorded run (`wf_5a7755c7-f92`, feature-adr,
7
+ * 66 agents) carries 47 distinct verbatim labels (`resolve-root:1`, `runs-record:heartbeat:Design`,
8
+ * `requirements · sonnet`, `ckpt:write:plan`, `trainpair:code`, `ledger:append`, `score:auto`, …)
9
+ * against the pipeline's own 11 canonical stages (`STAGE_EFFORT.override` in `feature-adr.js`:
10
+ * router, requirements, research, adr, ideation, ddd, architecture, plan, code, qe, fleet). Step 0's
11
+ * assessment counted 48 on the same record; a live reproducer today counts 47 — the one-off
12
+ * difference is not chased here (it does not change which prefixes are needed), and every label the
13
+ * reproducer found is in {@link STAGE_LABEL_FIXTURE} below.
14
+ *
15
+ * This module adds ONE thing: a pure, ordered-rule classifier from a verbatim label to a canonical
16
+ * stage. It never replaces or rewrites the verbatim label — `cost-ledger.ts` stamps `stageCanonical`
17
+ * NEXT TO the untouched `stage` field (ADR-001 D1). An unrecognised label is `{stage:'unknown'}`,
18
+ * never silently folded into `infra` — a taxonomy that quietly swallows what it does not recognise
19
+ * would hide exactly the drift this feature exists to surface (ADR-001 rejects that alternative).
20
+ *
21
+ * PURE. No filesystem, no clock, no process — this module must never gain a `node:fs` import; the
22
+ * `core-boundary` ratchet (`test/core-boundary.test.ts`) pins the current count and any I/O added
23
+ * here would grow it.
24
+ *
25
+ * @packageDocumentation
26
+ */
27
+
28
+ /** The pipeline's own 11 design/work stages, plus `infra` for the bookkeeping/plumbing labels that
29
+ * surround them (checkpoints, run-record heartbeats, training-pair capture, the ledger writer
30
+ * itself, `dz score`, `dz round`, the architecture-map refresh, …). 12 values total. */
31
+ export const CANONICAL_STAGES = [
32
+ 'router',
33
+ 'requirements',
34
+ 'research',
35
+ 'adr',
36
+ 'ideation',
37
+ 'ddd',
38
+ 'architecture',
39
+ 'plan',
40
+ 'code',
41
+ 'qe',
42
+ 'fleet',
43
+ 'infra',
44
+ ] as const;
45
+
46
+ export type CanonicalStage = (typeof CANONICAL_STAGES)[number];
47
+
48
+ export interface KnownStageResult {
49
+ readonly stage: CanonicalStage;
50
+ /** The INPUT label, verbatim and unmodified — the classifier never rewrites what it classifies. */
51
+ readonly label: string;
52
+ readonly known: true;
53
+ }
54
+
55
+ export interface UnknownStageResult {
56
+ readonly stage: 'unknown';
57
+ readonly label: string;
58
+ readonly known: false;
59
+ }
60
+
61
+ export type StageCanonResult = KnownStageResult | UnknownStageResult;
62
+
63
+ export interface StageLabelRule {
64
+ /** Tested against the label's PREFIX (everything before a ` · model` suffix, when present). */
65
+ readonly test: RegExp;
66
+ readonly stage: CanonicalStage;
67
+ }
68
+
69
+ /**
70
+ * ONE ordered table (ADR-001 D1 / C-3: "one canon table, one completeness test"). Rules are tried in
71
+ * order and the FIRST match wins, so a more specific rule (an exact router-time probe) must be listed
72
+ * before a broader prefix that would otherwise also claim it (the generic `arch-` → infra catch-all).
73
+ *
74
+ * Every entry below is justified against the 47-label fixture in
75
+ * `test/feature-adr-stage-canon.test.ts` (copied from the live record `wf_5a7755c7-f92`):
76
+ *
77
+ * - `router:*`, the Step-0 decision-recall receipt, and the ONE step-0 architecture-sync probe
78
+ * (`arch-сverka:step0` — a Cyrillic С, copied verbatim from the label the pipeline actually emits)
79
+ * are ROUTER, even though the last one would otherwise fall into the generic `arch-` infra bucket.
80
+ * - `design:*` is folded into `adr`: the S/M-tier pipeline runs ONE design agent that covers
81
+ * ADR+ideation+DDD+architecture at once (`feature-adr-ultracode.md`'s "design-aggregate"), so its
82
+ * probe labels have no separate canonical home — `adr` is the least-wrong single bucket, named here
83
+ * so the choice is not silent.
84
+ * - `qe:baseline*` (the Step-7.5 Codex-landed barrier's baseline hash/targets check) is CODE, not QE:
85
+ * it runs as part of confirming code landed, before the QE stage proper starts.
86
+ * - Everything else that is checkpoint/heartbeat/training-pair/ledger/score/round/architecture-map/
87
+ * resolve-root/usage-probe/project-skills plumbing is INFRA — it surrounds every design/work stage
88
+ * without belonging to any one of them.
89
+ */
90
+ // measurement-integrity fix-round-1/F3 (Codex r1 HIGH #3): every prefix rule below is now BOUNDED —
91
+ // the literal prefix must be followed by `:` or end-of-string (or, for the one dash-separated infra
92
+ // catch-all, `-`) before it may match. The unbounded form (`/^requirements/`) let an unrelated,
93
+ // never-emitted label like `requirementsBROKEN` or `code-new-stage` silently classify as a known
94
+ // stage — exactly the silent-swallow this module's own doc comment says D1 forbids. `roundtrip` is
95
+ // the sharpest case: the OLD infra alternation `/^(...|round)/` had no boundary at all, so any label
96
+ // merely STARTING WITH "round" (never emitted, but never refused either) read as infra plumbing.
97
+ // Negative fixtures for all three land in `test/feature-adr-stage-canon.test.ts`.
98
+ export const STAGE_LABEL_RULES: readonly StageLabelRule[] = [
99
+ { test: /^router(?::|$)/, stage: 'router' },
100
+ { test: /^decision-recall:step0(?::|$)/, stage: 'router' },
101
+ { test: /^arch-сverka:step0$/, stage: 'router' },
102
+ { test: /^requirements(?::|$)/, stage: 'requirements' },
103
+ { test: /^research(?::|$)/, stage: 'research' },
104
+ { test: /^design(?::|$)/, stage: 'adr' },
105
+ { test: /^adr(?::|$)/, stage: 'adr' },
106
+ { test: /^ideation(?::|$)/, stage: 'ideation' },
107
+ { test: /^ddd(?::|$)/, stage: 'ddd' },
108
+ { test: /^architecture(?::|$)/, stage: 'architecture' },
109
+ { test: /^decision-recall:step6(?::|$)/, stage: 'plan' },
110
+ { test: /^plan(?::|$)/, stage: 'plan' },
111
+ // `qe:baseline-hash` / `qe:baseline-targets` use a DASH after the "qe:baseline" root (not a colon),
112
+ // so the boundary set here is `[-:]` or end — never a bare "qe:baselineWHATEVER" glued on.
113
+ { test: /^qe:baseline(?:[-:]|$)/, stage: 'code' },
114
+ { test: /^code(?::|$)/, stage: 'code' },
115
+ { test: /^probe:/, stage: 'code' },
116
+ { test: /^qe(?::|$)/, stage: 'qe' },
117
+ { test: /^fleet(?::|$)/, stage: 'fleet' },
118
+ { test: /^(?:resolve-root|runs-record|ckpt|usage|project-skills|trainpair|ledger|score|round)(?::|$)/, stage: 'infra' },
119
+ // `arch-` is the one DASH-bounded catch-all (the dash IS the boundary — `arch-map:refresh`), kept
120
+ // as its own rule rather than folded into the colon-bounded alternation above.
121
+ { test: /^arch-/, stage: 'infra' },
122
+ ];
123
+
124
+ /**
125
+ * Classify one `stageLabel()` string. A label of the shape `requirements · sonnet` (verbatim label
126
+ * plus a ` · model` suffix — the shape `cost-ledger.ts` rows already carry for a mixed-model bucket)
127
+ * is matched on the part BEFORE ` · `; the returned `label` is always the untouched input.
128
+ *
129
+ * Never throws. A non-string or empty label is `unknown`, same as one that matches no rule.
130
+ */
131
+ export function canonicalStage(label: string): StageCanonResult {
132
+ if (typeof label !== 'string' || label.length === 0) {
133
+ return { stage: 'unknown', label: typeof label === 'string' ? label : '', known: false };
134
+ }
135
+ const sepIndex = label.indexOf(' · ');
136
+ const base = sepIndex === -1 ? label : label.slice(0, sepIndex);
137
+ for (const rule of STAGE_LABEL_RULES) {
138
+ if (rule.test.test(base)) return { stage: rule.stage, label, known: true };
139
+ }
140
+ return { stage: 'unknown', label, known: false };
141
+ }