@dzhechkov/harness-core 0.8.37 → 0.8.39
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.dz-manifest.json +253 -133
- package/README.md +209 -0
- package/dist/agentdb-index.d.ts +3 -3
- package/dist/agentdb-index.js +5 -5
- package/dist/agentdb-index.js.map +1 -1
- package/dist/amendment-trace.d.ts.map +1 -1
- package/dist/amendment-trace.js +4 -1
- package/dist/amendment-trace.js.map +1 -1
- package/dist/backlog.d.ts +2 -1
- package/dist/backlog.d.ts.map +1 -1
- package/dist/backlog.js +3 -2
- package/dist/backlog.js.map +1 -1
- package/dist/brain.d.ts.map +1 -1
- package/dist/brain.js +9 -2
- package/dist/brain.js.map +1 -1
- package/dist/bto-optimize.d.ts.map +1 -1
- package/dist/bto-optimize.js +9 -12
- package/dist/bto-optimize.js.map +1 -1
- package/dist/claim-check.d.ts.map +1 -1
- package/dist/claim-check.js +50 -14
- package/dist/claim-check.js.map +1 -1
- package/dist/cross-family-control.d.ts +35 -0
- package/dist/cross-family-control.d.ts.map +1 -1
- package/dist/cross-family-control.js +49 -3
- package/dist/cross-family-control.js.map +1 -1
- package/dist/experiment-assign.d.ts +110 -0
- package/dist/experiment-assign.d.ts.map +1 -0
- package/dist/experiment-assign.js +229 -0
- package/dist/experiment-assign.js.map +1 -0
- package/dist/feature-adr-checkpoints.d.ts +7 -2
- package/dist/feature-adr-checkpoints.d.ts.map +1 -1
- package/dist/feature-adr-checkpoints.js +20 -5
- package/dist/feature-adr-checkpoints.js.map +1 -1
- package/dist/feature-adr-envelope.d.ts +12 -0
- package/dist/feature-adr-envelope.d.ts.map +1 -1
- package/dist/feature-adr-envelope.js +12 -1
- package/dist/feature-adr-envelope.js.map +1 -1
- package/dist/feature-adr-routing.d.ts +4 -0
- package/dist/feature-adr-routing.d.ts.map +1 -1
- package/dist/feature-adr-routing.js +15 -1
- package/dist/feature-adr-routing.js.map +1 -1
- package/dist/index.d.ts +12 -5
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +11 -4
- package/dist/index.js.map +1 -1
- package/dist/ledger-cost-fill.d.ts +58 -0
- package/dist/ledger-cost-fill.d.ts.map +1 -0
- package/dist/ledger-cost-fill.js +78 -0
- package/dist/ledger-cost-fill.js.map +1 -0
- package/dist/loop-blobs.generated.js +2 -2
- package/dist/loop-blobs.generated.js.map +1 -1
- package/dist/mutation-gate.d.ts +29 -0
- package/dist/mutation-gate.d.ts.map +1 -1
- package/dist/mutation-gate.js +20 -0
- package/dist/mutation-gate.js.map +1 -1
- package/dist/name-check.d.ts +20 -1
- package/dist/name-check.d.ts.map +1 -1
- package/dist/name-check.js +42 -1
- package/dist/name-check.js.map +1 -1
- package/dist/no-stubs.d.ts +10 -0
- package/dist/no-stubs.d.ts.map +1 -1
- package/dist/no-stubs.js +13 -9
- package/dist/no-stubs.js.map +1 -1
- package/dist/patterns.d.ts +24 -0
- package/dist/patterns.d.ts.map +1 -1
- package/dist/patterns.js +49 -9
- package/dist/patterns.js.map +1 -1
- package/dist/publish-source-scope.d.ts +32 -0
- package/dist/publish-source-scope.d.ts.map +1 -0
- package/dist/publish-source-scope.js +53 -0
- package/dist/publish-source-scope.js.map +1 -0
- package/dist/publish.d.ts +5 -0
- package/dist/publish.d.ts.map +1 -1
- package/dist/publish.js +32 -21
- package/dist/publish.js.map +1 -1
- package/dist/qe-bridge.d.ts +12 -1
- package/dist/qe-bridge.d.ts.map +1 -1
- package/dist/qe-bridge.js +21 -11
- package/dist/qe-bridge.js.map +1 -1
- package/dist/rake-analyzer.d.ts +10 -3
- package/dist/rake-analyzer.d.ts.map +1 -1
- package/dist/rake-analyzer.js +84 -22
- package/dist/rake-analyzer.js.map +1 -1
- package/dist/recap.d.ts.map +1 -1
- package/dist/recap.js +8 -4
- package/dist/recap.js.map +1 -1
- package/dist/release.d.ts +2 -1
- package/dist/release.d.ts.map +1 -1
- package/dist/release.js +18 -4
- package/dist/release.js.map +1 -1
- package/dist/reqe.d.ts.map +1 -1
- package/dist/reqe.js +11 -1
- package/dist/reqe.js.map +1 -1
- package/dist/review-cost.d.ts +51 -0
- package/dist/review-cost.d.ts.map +1 -0
- package/dist/review-cost.js +110 -0
- package/dist/review-cost.js.map +1 -0
- package/dist/round.d.ts +136 -3
- package/dist/round.d.ts.map +1 -1
- package/dist/round.js +215 -6
- package/dist/round.js.map +1 -1
- package/dist/run-records.d.ts +51 -0
- package/dist/run-records.d.ts.map +1 -1
- package/dist/run-records.js +148 -2
- package/dist/run-records.js.map +1 -1
- package/dist/score.d.ts.map +1 -1
- package/dist/score.js +8 -1
- package/dist/score.js.map +1 -1
- package/dist/sign.d.ts +23 -0
- package/dist/sign.d.ts.map +1 -1
- package/dist/sign.js +52 -0
- package/dist/sign.js.map +1 -1
- package/dist/skills-verify.d.ts +8 -5
- package/dist/skills-verify.d.ts.map +1 -1
- package/dist/skills-verify.js +59 -7
- package/dist/skills-verify.js.map +1 -1
- package/dist/store-guard-prune.d.ts +22 -0
- package/dist/store-guard-prune.d.ts.map +1 -0
- package/dist/store-guard-prune.js +54 -0
- package/dist/store-guard-prune.js.map +1 -0
- package/dist/sweep-failure-classify.d.ts +45 -0
- package/dist/sweep-failure-classify.d.ts.map +1 -0
- package/dist/sweep-failure-classify.js +90 -0
- package/dist/sweep-failure-classify.js.map +1 -0
- package/dist/workflow-run-dispatch.d.ts +14 -2
- package/dist/workflow-run-dispatch.d.ts.map +1 -1
- package/dist/workflow-run-dispatch.js +14 -2
- package/dist/workflow-run-dispatch.js.map +1 -1
- package/package.json +1 -1
- package/sbom.json +432 -132
- package/src/agentdb-index.ts +5 -5
- package/src/amendment-trace.ts +4 -1
- package/src/backlog.ts +3 -2
- package/src/brain.ts +8 -2
- package/src/bto-optimize.ts +10 -8
- package/src/claim-check.ts +48 -12
- package/src/cross-family-control.ts +81 -3
- package/src/experiment-assign.ts +276 -0
- package/src/feature-adr-checkpoints.ts +20 -5
- package/src/feature-adr-envelope.ts +24 -1
- package/src/feature-adr-routing.ts +15 -1
- package/src/index.ts +23 -4
- package/src/loop-blobs.generated.ts +2 -2
- package/src/mutation-gate.ts +59 -0
- package/src/name-check.ts +43 -1
- package/src/no-stubs.ts +14 -5
- package/src/patterns.ts +49 -8
- package/src/publish-source-scope.ts +53 -0
- package/src/publish.ts +38 -16
- package/src/qe-bridge.ts +30 -11
- package/src/rake-analyzer.ts +75 -20
- package/src/rake-signatures.json +98 -0
- package/src/recap.ts +8 -4
- package/src/release.ts +18 -4
- package/src/reqe.ts +11 -1
- package/src/review-cost.ts +139 -0
- package/src/round.ts +327 -11
- package/src/run-records.ts +175 -1
- package/src/score.ts +8 -1
- package/src/sign.ts +53 -0
- package/src/skills-verify.ts +69 -9
- package/src/store-guard-prune.ts +81 -0
- package/src/sweep-failure-classify.ts +88 -0
- package/src/workflow-run-dispatch.ts +14 -2
package/src/agentdb-index.ts
CHANGED
|
@@ -1168,9 +1168,9 @@ export async function importVectorsToAgentdb(
|
|
|
1168
1168
|
}
|
|
1169
1169
|
|
|
1170
1170
|
/**
|
|
1171
|
-
* lesson-quarantine:
|
|
1172
|
-
*
|
|
1173
|
-
*
|
|
1171
|
+
* lesson-quarantine: mark mirrored rows as promoted after a promotion — the hook daemon reads ONLY
|
|
1172
|
+
* this mirror's metadata, so a promoted lesson must stop being excluded there while retaining its
|
|
1173
|
+
* quarantine history. Best-effort, same custody model as {@link bumpAgentdbUses} (missing db/deps ⇒ no-op).
|
|
1174
1174
|
*/
|
|
1175
1175
|
export function clearAgentdbQuarantine(
|
|
1176
1176
|
projectRoot: string,
|
|
@@ -1194,12 +1194,12 @@ export function clearAgentdbQuarantine(
|
|
|
1194
1194
|
db.pragma('busy_timeout = 5000');
|
|
1195
1195
|
db.exec(REASONING_BANK_SCHEMA);
|
|
1196
1196
|
const stmt = db.prepare(
|
|
1197
|
-
"UPDATE reasoning_patterns SET metadata =
|
|
1197
|
+
"UPDATE reasoning_patterns SET metadata = json_set(metadata, '$.qStatus', 'promoted', '$.promotedAt', ?) WHERE json_extract(metadata, '$.dzId') = ? AND json_extract(metadata, '$.qStatus') = 'quarantined'",
|
|
1198
1198
|
);
|
|
1199
1199
|
const tx = db.transaction(() => {
|
|
1200
1200
|
let cleared = 0;
|
|
1201
1201
|
for (const dzId of dzIds) {
|
|
1202
|
-
const r = stmt.run(dzId);
|
|
1202
|
+
const r = stmt.run(new Date().toISOString(), dzId);
|
|
1203
1203
|
cleared += Number((r as unknown as { changes?: number }).changes ?? 0);
|
|
1204
1204
|
}
|
|
1205
1205
|
return cleared;
|
package/src/amendment-trace.ts
CHANGED
|
@@ -138,7 +138,10 @@ export function extractTestTitles(body: string): string[] {
|
|
|
138
138
|
}
|
|
139
139
|
|
|
140
140
|
export function normalizeTestId(s: string): string {
|
|
141
|
-
|
|
141
|
+
// Unicode letter/number classes: the old `[^a-z0-9]` erased Cyrillic outright, so a Russian test title
|
|
142
|
+
// normalised to '' and tripped the floor as the author's fault (MEASURED 2026-09-04, backlog 191853a2).
|
|
143
|
+
// Re-run over all 519 features on 2026-09-20: zero verdicts changed — this only adds matches.
|
|
144
|
+
return s.toLowerCase().replace(/[^\p{L}\p{N}]+/gu, '');
|
|
142
145
|
}
|
|
143
146
|
|
|
144
147
|
/**
|
package/src/backlog.ts
CHANGED
|
@@ -266,7 +266,7 @@ export function readBacklogConfig(projectRoot: string): BacklogConfig {
|
|
|
266
266
|
}
|
|
267
267
|
|
|
268
268
|
/* ================================================================== */
|
|
269
|
-
/* STORE (AM-1) — .dz/backlog/ideas.jsonl (
|
|
269
|
+
/* STORE (AM-1) — .dz/backlog/ideas.jsonl (JSONL rewritten WHOLE per write, ADR-005 + amendment 2026-09-20) */
|
|
270
270
|
/* ================================================================== */
|
|
271
271
|
|
|
272
272
|
export function backlogDir(projectRoot: string): string {
|
|
@@ -335,7 +335,8 @@ function normaliseIdea(raw: Record<string, unknown>): IdeaRecord | undefined {
|
|
|
335
335
|
return rec;
|
|
336
336
|
}
|
|
337
337
|
|
|
338
|
-
/** Read the
|
|
338
|
+
/** Read the store (one current line per id — `writeIdeas` rewrites it whole and never appends). A corrupt
|
|
339
|
+
* line is SKIPPED (never fatal) — the whole store never throws. */
|
|
339
340
|
export function readIdeas(projectRoot: string): IdeaRecord[] {
|
|
340
341
|
const path = ideasPath(projectRoot);
|
|
341
342
|
if (!existsSync(path)) return [];
|
package/src/brain.ts
CHANGED
|
@@ -21,6 +21,7 @@ import { existsSync, mkdirSync, readFileSync, writeFileSync, renameSync, readdir
|
|
|
21
21
|
import { pathToFileURL } from 'node:url';
|
|
22
22
|
import { createRequire } from 'node:module';
|
|
23
23
|
import { spawn } from 'node:child_process';
|
|
24
|
+
import { openSqliteReadOnly } from '@dzhechkov/memory';
|
|
24
25
|
import { putBookKnowledge, queryBookKnowledge, bookKbPath, type BookKU, type BookKUHit } from './book-kb.js';
|
|
25
26
|
import { indexPatternsToAgentdb, searchAgentdbPatterns, reindexAgentdbRows, type AgentdbRow } from './agentdb-index.js';
|
|
26
27
|
import type { SnapshotRotationReport } from './agentdb-snapshot-rotation.js';
|
|
@@ -193,7 +194,8 @@ export function readBookKus(opts: {
|
|
|
193
194
|
return { kus: [], error: 'better-sqlite3 not installed (run: dz setup --memory agentdb)' };
|
|
194
195
|
}
|
|
195
196
|
try {
|
|
196
|
-
const
|
|
197
|
+
const handle = openSqliteReadOnly(opts.storePath, { Database });
|
|
198
|
+
const db = handle.db as NativeDb;
|
|
197
199
|
try {
|
|
198
200
|
const cols = 'book, ku_id, corpus_version, type, name, problem, content, chapter, pages, metadata';
|
|
199
201
|
const rows = (opts.source !== undefined
|
|
@@ -201,7 +203,11 @@ export function readBookKus(opts: {
|
|
|
201
203
|
: db.prepare(`SELECT ${cols} FROM book_knowledge`).all()) as ProjRow[];
|
|
202
204
|
return { kus: rows.map(rowToKu) };
|
|
203
205
|
} finally {
|
|
204
|
-
|
|
206
|
+
try {
|
|
207
|
+
db.close();
|
|
208
|
+
} finally {
|
|
209
|
+
handle.cleanup();
|
|
210
|
+
}
|
|
205
211
|
}
|
|
206
212
|
} catch (err) {
|
|
207
213
|
return { kus: [], error: `read book KB failed: ${err instanceof Error ? err.message : String(err)}` };
|
package/src/bto-optimize.ts
CHANGED
|
@@ -20,6 +20,8 @@
|
|
|
20
20
|
|
|
21
21
|
import { existsSync, readFileSync } from 'node:fs';
|
|
22
22
|
|
|
23
|
+
import { maskMarkdown } from './markdown-masker.js';
|
|
24
|
+
|
|
23
25
|
export type BtoDimension = 'METHODOLOGY' | 'DEPTH' | 'CORRECTNESS' | 'USABILITY' | 'ROBUSTNESS';
|
|
24
26
|
export const BTO_DIMENSIONS: readonly BtoDimension[] = Object.freeze(['METHODOLOGY', 'DEPTH', 'CORRECTNESS', 'USABILITY', 'ROBUSTNESS']);
|
|
25
27
|
export type DimScores = Record<BtoDimension, number>;
|
|
@@ -203,22 +205,22 @@ const bodyAfterFrontmatter = (t: string): string => {
|
|
|
203
205
|
|
|
204
206
|
/**
|
|
205
207
|
* ALL structural markers on the BODY, not just space-delimited ATX (QE: `##\tNEW`, bare `##`, and setext
|
|
206
|
-
* `Title\n===` / `Title\n---` evaded the old regex).
|
|
207
|
-
*
|
|
208
|
-
*
|
|
208
|
+
* `Title\n===` / `Title\n---` evaded the old regex). The canonical masker skips fenced code blocks before
|
|
209
|
+
* collecting ATX (`#`..`######` + any/no ws) and setext underlines (a non-empty line immediately followed
|
|
210
|
+
* by `=+`/`-+`). Its `unclosed: 'restore'` policy is load-bearing: measured over 7,121 headed Markdown
|
|
211
|
+
* documents, the old parity toggle missed a deletion in 39 documents, masking with `hide` missed 22, and
|
|
212
|
+
* masking with `restore` missed 0.
|
|
209
213
|
*/
|
|
210
214
|
const headings = (t: string): string[] => {
|
|
211
|
-
const
|
|
215
|
+
const body = bodyAfterFrontmatter(t);
|
|
216
|
+
const lines = maskMarkdown(body, { unclosed: 'restore' }).split('\n');
|
|
212
217
|
const out: string[] = [];
|
|
213
|
-
let inFence = false;
|
|
214
218
|
for (let i = 0; i < lines.length; i++) {
|
|
215
219
|
const line = lines[i]!;
|
|
216
|
-
if (/^\s*(```|~~~)/.test(line)) { inFence = !inFence; continue; }
|
|
217
|
-
if (inFence) continue;
|
|
218
220
|
const atx = /^(#{1,6})(?:\s.*)?$/.exec(line.replace(/\s+$/, ''));
|
|
219
221
|
if (atx) { out.push('ATX:' + line.trim()); continue; }
|
|
220
222
|
const next = lines[i + 1];
|
|
221
|
-
if (line.trim() !== '' &&
|
|
223
|
+
if (line.trim() !== '' && next !== undefined && /^(=+|-+)\s*$/.test(next)) {
|
|
222
224
|
out.push('SETEXT:' + line.trim() + '|' + next.trim());
|
|
223
225
|
}
|
|
224
226
|
}
|
package/src/claim-check.ts
CHANGED
|
@@ -122,6 +122,8 @@ function paragraphAround(lines: readonly string[], i: number): string {
|
|
|
122
122
|
/** Tags that make a claim honest (case-insensitive). */
|
|
123
123
|
// `estimated` agrees with the `estimated: true` honest-uncertainty marker `dz usage` already
|
|
124
124
|
// emits — the two honesty systems must not contradict each other.
|
|
125
|
+
import { maskMarkdown } from './markdown-masker.js';
|
|
126
|
+
|
|
125
127
|
const HONEST_TAGS = ['measured', 'claimed', 'synthetic', 'unvalidated', 'baseline', 'estimated'];
|
|
126
128
|
|
|
127
129
|
/**
|
|
@@ -136,17 +138,27 @@ const HONEST_TAGS = ['measured', 'claimed', 'synthetic', 'unvalidated', 'baselin
|
|
|
136
138
|
*/
|
|
137
139
|
export function isFenced(text: string, line: number): boolean {
|
|
138
140
|
if (typeof text !== 'string' || typeof line !== 'number' || !isFinite(line) || line < 1) return false;
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
141
|
+
// DELEGATED to the canonical masker. The local walk this replaces already tracked the marker
|
|
142
|
+
// CHARACTER — the naive-toggle bug the comment above describes was genuinely fixed — but it still
|
|
143
|
+
// broke two further CommonMark rules, and both were MEASURED 2026-09-20 to answer `false` for a
|
|
144
|
+
// line that IS inside a block: a closing fence may carry NO info string, so a second info-string
|
|
145
|
+
// line closed the block; and a closing fence may not be SHORTER than the opening one, so a
|
|
146
|
+
// three-backtick line closed a four-backtick block. Both make the engine scan a QUOTED example as
|
|
147
|
+
// a real claim, and make the hook's deny path stop exempting it.
|
|
148
|
+
//
|
|
149
|
+
// `unclosed: 'hide'` keeps the local walk's policy: an unclosed opener leaves every later line
|
|
150
|
+
// inside the block. One deliberate difference: the OPENING delimiter line now answers `true` (the
|
|
151
|
+
// local walk answered `false` for it and `true` for the closing one) — an info string is not
|
|
152
|
+
// prose, and the two delimiters answering differently was an artifact, not a decision.
|
|
153
|
+
let masked = false;
|
|
154
|
+
try {
|
|
155
|
+
const target = line - 1;
|
|
156
|
+
maskMarkdown(text.replace(/\r\n/g, '\n'), {
|
|
157
|
+
unclosed: 'hide',
|
|
158
|
+
onMasked: (i?: number) => { if (i === target) masked = true; },
|
|
159
|
+
});
|
|
160
|
+
} catch { return false; }
|
|
161
|
+
return masked;
|
|
150
162
|
}
|
|
151
163
|
|
|
152
164
|
const FENCE_RE = /^\s*(`{3,}|~{3,})/;
|
|
@@ -189,9 +201,33 @@ const MAP_METRIC_RE = new RegExp(
|
|
|
189
201
|
* A shell reproducer is STRUCTURAL, never a word. `(MEASURED — reproducer)` is self-certifying and
|
|
190
202
|
* must not pass; a backticked span whose first token is a command this repo actually measures with is
|
|
191
203
|
* evidence. The allowlist boundary is exactly that: an unknown binary is a claim ABOUT evidence.
|
|
204
|
+
*
|
|
205
|
+
* `python3` joined the list for backlog 0b53470a, on the list's OWN criterion rather than by
|
|
206
|
+
* widening it: `packages/@dzhechkov/health-advisor/test/goap-python-suite.test.js` makes
|
|
207
|
+
* `python3 -m unittest discover` a GATE, so it is a command this repo measures with. Before the
|
|
208
|
+
* addition a Python measurer could not cite itself — MEASURED by twins, one line differing only in
|
|
209
|
+
* the backticked binary: `python3 …` → 1 finding "Tagged MEASURED but cites no reproducer",
|
|
210
|
+
* `pytest …` → 0, `git show …` → 0. Bare `python` was deliberately NOT added (cross-family review
|
|
211
|
+
* r1, MEDIUM): the gate this repo runs is `python3`, and the only measured usage in the tree is
|
|
212
|
+
* `python3 -m unittest` — an entry nothing measures with would be exactly the "claim ABOUT
|
|
213
|
+
* evidence" this list refuses. The list stays CLOSED: a measurer in any other language meets the
|
|
214
|
+
* same wall and needs the same deliberate entry. That is the price of the boundary, not a defect.
|
|
215
|
+
*
|
|
216
|
+
* THE BOUNDARY IS `(?![\w-])`, NOT `\b`, and that turned out to matter far beyond python
|
|
217
|
+
* (cross-family review r1, HIGH). `\b` treats a hyphen as a word boundary, so every backticked
|
|
218
|
+
* FILE NAME that begins with a listed command counted as a reproducer. MEASURED 2026-09-19:
|
|
219
|
+
* `pnpm-lock.yaml`, `dz-harness-hub`, `git-workflow` and `npm-shrinkwrap.json` all passed as
|
|
220
|
+
* evidence for a MEASURED claim, while `node_modules` was correctly refused — only because `_` is
|
|
221
|
+
* a word character and `-` is not. The repository holds hundreds of such tokens, so the check has
|
|
222
|
+
* been accepting file names as measurements for as long as the list has existed.
|
|
223
|
+
*
|
|
224
|
+
* WHAT THIS CHECK DOES NOT DO, said plainly because the previous wording implied more: it verifies
|
|
225
|
+
* the SHAPE of a citation, never that a measurement happened. `git --version` in backticks is
|
|
226
|
+
* accepted by construction — validating arguments per binary is a different mechanism with a
|
|
227
|
+
* different cost, and pretending otherwise would be the very laundering this file exists to stop.
|
|
192
228
|
*/
|
|
193
229
|
const SHELL_REPRO_RE =
|
|
194
|
-
/`\s*\$?\s*(?:ps|stat|lsof|time|git|npm|npx|node|pnpm|yarn|dz|curl|wc|grep|find|cargo|make|docker|kubectl|awk|sed|du|df|vitest|pytest)\
|
|
230
|
+
/`\s*\$?\s*(?:ps|stat|lsof|time|git|npm|npx|node|pnpm|yarn|dz|curl|wc|grep|find|cargo|make|docker|kubectl|awk|sed|du|df|vitest|pytest|python3)(?![\w-])[^`]*`/i;
|
|
195
231
|
|
|
196
232
|
/** Reproducer references that count as evidence backing a MEASURED claim. */
|
|
197
233
|
const REPRODUCER_HINTS = [
|
|
@@ -579,10 +579,47 @@ export function buildControlRow(input: BuildControlRowInput): BuildControlRowRes
|
|
|
579
579
|
};
|
|
580
580
|
}
|
|
581
581
|
|
|
582
|
+
/**
|
|
583
|
+
* experiment-instrument FR-4/A6 (ADR-001): a `stage:'control'` row `dz control-review` writes when
|
|
584
|
+
* the run REFUSED before producing a diff — the failure-leaves-a-row half of the ADR's safety
|
|
585
|
+
* property. Distinguished from {@link ControlLedgerRow} by `outcome:'refused'`, which a successful
|
|
586
|
+
* row never carries; the two schemas share nothing else structurally on purpose — a refused run has
|
|
587
|
+
* no diff, no tree hashes, no grades to validate.
|
|
588
|
+
*/
|
|
589
|
+
export interface ControlRefusedRow {
|
|
590
|
+
readonly slug: string;
|
|
591
|
+
readonly stage: 'control';
|
|
592
|
+
readonly outcome: 'refused';
|
|
593
|
+
/** Which half was responsible: the claude review, the codex review, or neither (a setup/tree/
|
|
594
|
+
* aggregation failure that belongs to neither half specifically). */
|
|
595
|
+
readonly half: 'claude' | 'codex' | 'setup';
|
|
596
|
+
readonly reason: string;
|
|
597
|
+
readonly runId: string;
|
|
598
|
+
/** Known only when the coder family was determined before the refusal — absent for the earliest
|
|
599
|
+
* failures (the claude half itself failing before it can report which family it reviewed). When
|
|
600
|
+
* present, the SAME `bucket(coderFamily, reviewerOfInterest)` a successful control row uses. */
|
|
601
|
+
readonly coderFamily?: 'codex' | 'claude';
|
|
602
|
+
readonly minutes: number | null;
|
|
603
|
+
}
|
|
604
|
+
|
|
605
|
+
function isValidControlRefusedRow(obj: Record<string, unknown>): boolean {
|
|
606
|
+
if (typeof obj['slug'] !== 'string' || obj['slug'].trim() === '') return false;
|
|
607
|
+
if (typeof obj['runId'] !== 'string' || obj['runId'].trim() === '') return false;
|
|
608
|
+
if (obj['half'] !== 'claude' && obj['half'] !== 'codex' && obj['half'] !== 'setup') return false;
|
|
609
|
+
if (typeof obj['reason'] !== 'string' || obj['reason'].trim() === '') return false;
|
|
610
|
+
if (!(obj['minutes'] === null || (typeof obj['minutes'] === 'number' && Number.isFinite(obj['minutes'])))) return false;
|
|
611
|
+
if (obj['coderFamily'] !== undefined && obj['coderFamily'] !== 'codex' && obj['coderFamily'] !== 'claude') return false;
|
|
612
|
+
return true;
|
|
613
|
+
}
|
|
614
|
+
|
|
582
615
|
/* ── D5: reading the ledger back (FR-7; A8) ──────────────────────────────────────────────────── */
|
|
583
616
|
|
|
584
617
|
export interface ParsedControlRows {
|
|
585
618
|
readonly rows: readonly ControlLedgerRow[];
|
|
619
|
+
/** experiment-instrument FR-4/A6: `stage:'control'` rows with `outcome:'refused'` — a run that
|
|
620
|
+
* never produced a diff. Schema-validated (`isValidControlRefusedRow`) the same way `rows` is;
|
|
621
|
+
* one that fails validation is `unreadable`, same as a malformed successful row. */
|
|
622
|
+
readonly refusedRows: readonly ControlRefusedRow[];
|
|
586
623
|
readonly roundRows: readonly Record<string, unknown>[];
|
|
587
624
|
readonly fullRows: readonly Record<string, unknown>[];
|
|
588
625
|
/** A8: a line that is not parseable JSON, or parses to something that is not a plain object, or
|
|
@@ -678,6 +715,7 @@ function isValidControlRow(obj: Record<string, unknown>): boolean {
|
|
|
678
715
|
*/
|
|
679
716
|
export function parseControlRows(lines: readonly string[]): ParsedControlRows {
|
|
680
717
|
const rows: ControlLedgerRow[] = [];
|
|
718
|
+
const refusedRows: ControlRefusedRow[] = [];
|
|
681
719
|
const roundRows: Record<string, unknown>[] = [];
|
|
682
720
|
const fullRows: Record<string, unknown>[] = [];
|
|
683
721
|
let unreadable = 0;
|
|
@@ -698,6 +736,17 @@ export function parseControlRows(lines: readonly string[]): ParsedControlRows {
|
|
|
698
736
|
const obj = parsed as Record<string, unknown>;
|
|
699
737
|
const stage = obj['stage'];
|
|
700
738
|
if (stage === 'control') {
|
|
739
|
+
// experiment-instrument FR-4/A6: `outcome:'refused'` is a DIFFERENT schema from a successful
|
|
740
|
+
// control row — checked FIRST, so a refused row is never mistaken for a malformed successful
|
|
741
|
+
// one (which would count it `unreadable`, losing exactly the receipt this feature adds).
|
|
742
|
+
if (obj['outcome'] === 'refused') {
|
|
743
|
+
if (!isValidControlRefusedRow(obj)) {
|
|
744
|
+
unreadable++;
|
|
745
|
+
continue;
|
|
746
|
+
}
|
|
747
|
+
refusedRows.push(obj as unknown as ControlRefusedRow);
|
|
748
|
+
continue;
|
|
749
|
+
}
|
|
701
750
|
if (!isValidControlRow(obj)) {
|
|
702
751
|
unreadable++;
|
|
703
752
|
continue;
|
|
@@ -711,7 +760,7 @@ export function parseControlRows(lines: readonly string[]): ParsedControlRows {
|
|
|
711
760
|
// Every other stage (plan/impl/fix/loop-run/round-exec/…) and the header/comment row (no
|
|
712
761
|
// string `stage`) are readable, just not addressed by this module.
|
|
713
762
|
}
|
|
714
|
-
return { rows, roundRows, fullRows, unreadable };
|
|
763
|
+
return { rows, refusedRows, roundRows, fullRows, unreadable };
|
|
715
764
|
}
|
|
716
765
|
|
|
717
766
|
/* ── D6: the per-family-pair aggregate (FR-7; A6, A8) ────────────────────────────────────────── */
|
|
@@ -769,6 +818,10 @@ export interface FamilyPairAggregate {
|
|
|
769
818
|
readonly refutedShare: { readonly n: number; readonly value: number | 'unknown' };
|
|
770
819
|
readonly costPerConfirmed: { readonly n: number; readonly value: number | 'unknown' };
|
|
771
820
|
readonly draftToShipped: ReadonlyArray<{ readonly slug: string; readonly first: string; readonly final: string; readonly finals: number }>;
|
|
821
|
+
/** experiment-instrument FR-4/A6: `stage:'control'` rows refused for THIS pair (attributed by a
|
|
822
|
+
* known `coderFamily` on the refused row) — a real cost line (the run was attempted and failed),
|
|
823
|
+
* never counted in `n` (which measures completed reviews). */
|
|
824
|
+
readonly refusedRuns: number;
|
|
772
825
|
}
|
|
773
826
|
|
|
774
827
|
export interface FamilyAggregate {
|
|
@@ -781,6 +834,11 @@ export interface FamilyAggregate {
|
|
|
781
834
|
/** A6: the raw count of `stage:'control'` rows folded in — `0` means every `foreignUnique`
|
|
782
835
|
* figure below is a true, honestly-printed absence, not a fabricated non-observation. */
|
|
783
836
|
readonly controlRows: number;
|
|
837
|
+
/** experiment-instrument FR-4/A6: EVERY `stage:'control'` refused row seen, attributed or not —
|
|
838
|
+
* the total accounting figure `refusedRuns` (per pair) can never exceed, and the gap between the
|
|
839
|
+
* sum of per-pair `refusedRuns` and this total is exactly how many refusals had no determinable
|
|
840
|
+
* coderFamily (the earliest failures — printed here rather than silently dropped). */
|
|
841
|
+
readonly refusedControlRows: number;
|
|
784
842
|
}
|
|
785
843
|
|
|
786
844
|
interface MutablePair {
|
|
@@ -801,13 +859,14 @@ interface MutablePair {
|
|
|
801
859
|
costN: number;
|
|
802
860
|
costSum: number;
|
|
803
861
|
drafts: Array<{ slug: string; first: string; final: string; finals: number }>;
|
|
862
|
+
refusedRuns: number;
|
|
804
863
|
}
|
|
805
864
|
|
|
806
865
|
function newPair(): MutablePair {
|
|
807
866
|
return {
|
|
808
867
|
n: 0, grades: {}, shipped: 0, shippedTotal: 0, fixRoundsList: [],
|
|
809
868
|
foreignN: 0, foreignBySeverity: {}, foreignAuto: 0, foreignAdjudicated: 0, foreignAutoRuns: 0, foreignAdjudicatedRuns: 0, foreignIncomplete: 0,
|
|
810
|
-
refutedN: 0, refutedSum: 0, costN: 0, costSum: 0, drafts: [],
|
|
869
|
+
refutedN: 0, refutedSum: 0, costN: 0, costSum: 0, drafts: [], refusedRuns: 0,
|
|
811
870
|
};
|
|
812
871
|
}
|
|
813
872
|
|
|
@@ -886,6 +945,18 @@ export function aggregateByFamily(
|
|
|
886
945
|
else { b.foreignAutoRuns++; b.foreignAuto += foreignTotal; }
|
|
887
946
|
}
|
|
888
947
|
|
|
948
|
+
// experiment-instrument FR-4/A6: a refused row is bucketed the SAME way a successful control row
|
|
949
|
+
// is — by its own coderFamily and the complementary reviewer — but ONLY when coderFamily is known
|
|
950
|
+
// (the earliest failures, before the claude half reports which family it reviewed, cannot be
|
|
951
|
+
// attributed to a pair; they still count toward `refusedControlRows` at the top level below,
|
|
952
|
+
// never silently dropped). `n` is deliberately untouched: `n` measures COMPLETED reviews.
|
|
953
|
+
for (const rr of parsed.refusedRows) {
|
|
954
|
+
if (rr.coderFamily === undefined) continue;
|
|
955
|
+
const reviewerOfInterest = rr.coderFamily === 'codex' ? 'claude' : 'codex';
|
|
956
|
+
const b = bucket(rr.coderFamily, reviewerOfInterest);
|
|
957
|
+
b.refusedRuns++;
|
|
958
|
+
}
|
|
959
|
+
|
|
889
960
|
// draftToShipped: earliest known verdict per slug (a qe-bridge signoff, or this control row's
|
|
890
961
|
// own claude-half grade when no signoff was given) against the slug's final grade — keyed by
|
|
891
962
|
// slug PLUS the normalized (coder,reviewer) family pair (r1-14: two final rows for the same slug
|
|
@@ -952,9 +1023,16 @@ export function aggregateByFamily(
|
|
|
952
1023
|
refutedShare: { n: b.refutedN, value: b.refutedN > 0 ? b.refutedSum / b.refutedN : 'unknown' },
|
|
953
1024
|
costPerConfirmed: { n: b.costN, value: b.costN > 0 ? b.costSum / b.costN : 'unknown' },
|
|
954
1025
|
draftToShipped: b.drafts,
|
|
1026
|
+
refusedRuns: b.refusedRuns,
|
|
955
1027
|
};
|
|
956
1028
|
}
|
|
957
1029
|
|
|
958
1030
|
const incompleteControlRows = parsed.rows.filter((r) => !r.complete).length;
|
|
959
|
-
return {
|
|
1031
|
+
return {
|
|
1032
|
+
pairs: out,
|
|
1033
|
+
incomplete: parsed.unreadable > 0 || incompleteControlRows > 0,
|
|
1034
|
+
incompleteControlRows,
|
|
1035
|
+
controlRows: parsed.rows.length,
|
|
1036
|
+
refusedControlRows: parsed.refusedRows.length,
|
|
1037
|
+
};
|
|
960
1038
|
}
|
|
@@ -0,0 +1,276 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Deterministic, blocked arm assignment for a prospective ablation (ADR-001, ablation-c-start).
|
|
3
|
+
*
|
|
4
|
+
* WHY. The owner's question — does the Step-8 QE mode change task speed — is only answerable
|
|
5
|
+
* causally if the variant a task receives is decided by chance, BEFORE the work starts, and tied to
|
|
6
|
+
* the task's own identity so a repeated query for the same task never reassigns it. This module is
|
|
7
|
+
* that decision: a PURE function of `(experiment, stratum, seed, index)`. No filesystem, no clock,
|
|
8
|
+
* no network — the caller (the cli) owns the journal, the lock, and the "did work already start?"
|
|
9
|
+
* check; this module only computes.
|
|
10
|
+
*
|
|
11
|
+
* ALGORITHM — block randomization, one pair per arm-combination per block. `mulberry32` is the
|
|
12
|
+
* repo's ONLY pseudo-random generator (`compounding.ts` — "no second RNG in this repo"); reused
|
|
13
|
+
* here rather than adding a second stream. Block size is `arms.length * 2` — two occurrences of
|
|
14
|
+
* every arm per block, so a completed block is always perfectly balanced and `propensity` (the
|
|
15
|
+
* arm's actual share of its block) is a fixed `1 / arms.length` — never guessed, never assumed
|
|
16
|
+
* 0.5 by default (NFR-4). The block's own permutation is seeded from
|
|
17
|
+
* `fnv1a(JSON.stringify([experiment, stratum, seed, block]))` (a JSON-tuple, not a delimiter join —
|
|
18
|
+
* the same reason `workOrderDigestInput` in epoch-replay.ts uses tuples: a delimiter in `experiment`
|
|
19
|
+
* or `stratum` could otherwise make two different keys hash identically) — so different strata (and
|
|
20
|
+
* different experiments) never share a stream, and the SAME `(experiment, stratum, seed, index)`
|
|
21
|
+
* always rederives the SAME arm (A1 idempotency, A2 block balance, A3 propensity).
|
|
22
|
+
*/
|
|
23
|
+
|
|
24
|
+
import { fnv1a } from './feature-adr-checkpoints.js';
|
|
25
|
+
import { mulberry32 } from './compounding.js';
|
|
26
|
+
|
|
27
|
+
export interface AssignArmInput {
|
|
28
|
+
readonly experiment: string;
|
|
29
|
+
readonly stratum: string;
|
|
30
|
+
/** Pre-registered seed — fixed once, before any assignment is made. */
|
|
31
|
+
readonly seed: number;
|
|
32
|
+
/** 0-based ordinal of this task within its `(experiment, stratum)` sequence. */
|
|
33
|
+
readonly index: number;
|
|
34
|
+
/** The arms on offer, e.g. `['direct', 'reference']`. At least 2, all non-empty, all unique. */
|
|
35
|
+
readonly arms: readonly string[];
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
export type AssignArmResult =
|
|
39
|
+
| { readonly ok: true; readonly arm: string; readonly propensity: number; readonly block: number; readonly position: number }
|
|
40
|
+
| { readonly ok: false; readonly reason: string };
|
|
41
|
+
|
|
42
|
+
function isNonEmptyString(v: unknown): v is string {
|
|
43
|
+
return typeof v === 'string' && v.trim() !== '';
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
/** Fisher-Yates, driven by the given deterministic RNG. Does not mutate its input. */
|
|
47
|
+
function shuffle<T>(items: readonly T[], rand: () => number): T[] {
|
|
48
|
+
const out = items.slice();
|
|
49
|
+
for (let i = out.length - 1; i > 0; i--) {
|
|
50
|
+
const j = Math.floor(rand() * (i + 1));
|
|
51
|
+
const tmp = out[i]!;
|
|
52
|
+
out[i] = out[j]!;
|
|
53
|
+
out[j] = tmp;
|
|
54
|
+
}
|
|
55
|
+
return out;
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
/**
|
|
59
|
+
* Deterministically assign one arm to task ordinal `index` within `(experiment, stratum)`. Refuses
|
|
60
|
+
* (never throws, never guesses) on any malformed input — empty strings, a non-integer seed/index, a
|
|
61
|
+
* negative index, fewer than two arms, duplicate or blank arm names.
|
|
62
|
+
*/
|
|
63
|
+
export function assignArm(input: AssignArmInput): AssignArmResult {
|
|
64
|
+
if (!isNonEmptyString(input?.experiment)) return { ok: false, reason: 'experiment: expected a non-empty string' };
|
|
65
|
+
if (!isNonEmptyString(input?.stratum)) return { ok: false, reason: 'stratum: expected a non-empty string' };
|
|
66
|
+
if (!Number.isInteger(input?.seed) || input.seed < 0) return { ok: false, reason: 'seed: expected a non-negative integer' };
|
|
67
|
+
if (!Number.isInteger(input?.index) || input.index < 0) return { ok: false, reason: 'index: expected a non-negative integer' };
|
|
68
|
+
if (!Array.isArray(input?.arms) || input.arms.length < 2) {
|
|
69
|
+
return { ok: false, reason: 'arms: expected an array of at least 2 arm names' };
|
|
70
|
+
}
|
|
71
|
+
if (!input.arms.every(isNonEmptyString)) return { ok: false, reason: 'arms: every arm name must be a non-empty string' };
|
|
72
|
+
const seen = new Set<string>();
|
|
73
|
+
for (const a of input.arms) {
|
|
74
|
+
if (seen.has(a)) return { ok: false, reason: `arms: duplicate arm name ${JSON.stringify(a)}` };
|
|
75
|
+
seen.add(a);
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
const blockSize = input.arms.length * 2; // two copies of every arm — always a perfect block
|
|
79
|
+
const block = Math.floor(input.index / input.arms.length / 2); // == Math.floor(index / blockSize)
|
|
80
|
+
const position = input.index % blockSize;
|
|
81
|
+
|
|
82
|
+
const key = JSON.stringify({ experiment: input.experiment, stratum: input.stratum, seed: input.seed, block });
|
|
83
|
+
const rand = mulberry32(parseInt(fnv1a(key), 16) >>> 0);
|
|
84
|
+
// Base card set for this block: every arm twice, in the input's own order — then permuted by the
|
|
85
|
+
// block-seeded stream, so the CONTENT of every block is fixed by construction (perfect balance)
|
|
86
|
+
// and only the ORDER is randomized.
|
|
87
|
+
const cards: string[] = [];
|
|
88
|
+
for (const a of input.arms) { cards.push(a); cards.push(a); }
|
|
89
|
+
const permuted = shuffle(cards, rand);
|
|
90
|
+
const arm = permuted[position]!;
|
|
91
|
+
const propensity = 2 / blockSize; // == 1 / arms.length — the arm's fixed, actual share of its block
|
|
92
|
+
|
|
93
|
+
return { ok: true, arm, propensity, block, position };
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
export interface AssignmentRecord {
|
|
97
|
+
readonly ts: string;
|
|
98
|
+
readonly experiment: string;
|
|
99
|
+
readonly taskId: string;
|
|
100
|
+
readonly stratum: string;
|
|
101
|
+
readonly seed: number;
|
|
102
|
+
readonly index: number;
|
|
103
|
+
readonly block: number;
|
|
104
|
+
readonly position: number;
|
|
105
|
+
readonly arm: string;
|
|
106
|
+
readonly propensity: number;
|
|
107
|
+
readonly assignedAt: string;
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
export interface ReadAssignmentsResult {
|
|
111
|
+
readonly records: readonly AssignmentRecord[];
|
|
112
|
+
/** Non-blank lines that did not parse as a valid assignment record. Never silently dropped: a
|
|
113
|
+
* caller checking "is this task assigned?" must be able to tell "no" from "the journal is
|
|
114
|
+
* unreadable here" (A5). */
|
|
115
|
+
readonly malformedLines: number;
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
/** fix-round-1 (Codex r1 HIGH #7): `new Date().toISOString()` is the ONLY producer of `ts`/
|
|
119
|
+
* `assignedAt` (see the cli's `experimentJournalPath` writer) — a record whose timestamp does not
|
|
120
|
+
* match that exact shape did not come from this code path, so it is corruption, not merely an
|
|
121
|
+
* unusual value. */
|
|
122
|
+
const ISO_TIMESTAMP_RE = /^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}\.\d{3}Z$/;
|
|
123
|
+
|
|
124
|
+
function isValidIsoTimestamp(v: unknown): v is string {
|
|
125
|
+
// Lead delta after Codex r2 (MEDIUM): shape + `Date.parse` accepts impossible calendar dates —
|
|
126
|
+
// `2026-02-30T00:00:00.000Z` parses (it rolls into March) and passed. Round-tripping through
|
|
127
|
+
// `toISOString()` is the exact check: only a real instant re-serializes to the same string.
|
|
128
|
+
if (typeof v !== 'string' || !ISO_TIMESTAMP_RE.test(v)) return false;
|
|
129
|
+
const ms = Date.parse(v);
|
|
130
|
+
return Number.isFinite(ms) && new Date(ms).toISOString() === v;
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
/**
|
|
134
|
+
* PURELY STRUCTURAL: every field is present and the right TYPE. This is `readAssignments`'s gate —
|
|
135
|
+
* it decides whether a journal LINE parses at all (A5's `malformedLines`). It deliberately does NOT
|
|
136
|
+
* validate timestamp FORMAT, propensity RANGE, or arm MEMBERSHIP: a line that is well-typed but
|
|
137
|
+
* semantically wrong (a hand-edited `position:99`, `propensity:2`, a flipped `arm`) is exactly the
|
|
138
|
+
* "syntactically valid corruption" `verifyAssignmentRecord` below exists to catch — folding those
|
|
139
|
+
* checks in here would make a single semantically-tampered record poison `readAssignments` for
|
|
140
|
+
* every caller, including ones (like `status`'s per-task duration math) that need to name WHICH
|
|
141
|
+
* task is bad rather than refuse the whole read (fix-round-1 HIGH #7/#8 — see `verifyAssignmentRecord`).
|
|
142
|
+
*/
|
|
143
|
+
function isValidAssignmentRecord(v: unknown): v is AssignmentRecord {
|
|
144
|
+
if (typeof v !== 'object' || v === null) return false;
|
|
145
|
+
const r = v as Record<string, unknown>;
|
|
146
|
+
return (
|
|
147
|
+
isNonEmptyString(r['ts']) &&
|
|
148
|
+
isNonEmptyString(r['experiment']) &&
|
|
149
|
+
isNonEmptyString(r['taskId']) &&
|
|
150
|
+
isNonEmptyString(r['stratum']) &&
|
|
151
|
+
Number.isInteger(r['seed']) && (r['seed'] as number) >= 0 &&
|
|
152
|
+
Number.isInteger(r['index']) && (r['index'] as number) >= 0 &&
|
|
153
|
+
Number.isInteger(r['block']) && (r['block'] as number) >= 0 &&
|
|
154
|
+
Number.isInteger(r['position']) && (r['position'] as number) >= 0 &&
|
|
155
|
+
isNonEmptyString(r['arm']) &&
|
|
156
|
+
typeof r['propensity'] === 'number' && Number.isFinite(r['propensity'] as number) &&
|
|
157
|
+
isNonEmptyString(r['assignedAt'])
|
|
158
|
+
);
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
export type VerifyAssignmentResult =
|
|
162
|
+
| { readonly ok: true }
|
|
163
|
+
| {
|
|
164
|
+
readonly ok: false;
|
|
165
|
+
readonly reason: string;
|
|
166
|
+
/** Lead delta after Codex r2: `null` when the record failed on its SEED — there is nothing to
|
|
167
|
+
* expect, because the tuple it would be derived from is itself the thing under suspicion. */
|
|
168
|
+
readonly expected: { readonly arm: string; readonly block: number; readonly position: number; readonly propensity: number } | null;
|
|
169
|
+
readonly actual: { readonly arm: string; readonly block: number; readonly position: number; readonly propensity: number };
|
|
170
|
+
};
|
|
171
|
+
|
|
172
|
+
/**
|
|
173
|
+
* fix-round-1 (Codex r1 HIGH #7) — "syntactically valid journal corruption is accepted and returned
|
|
174
|
+
* as the task's assignment". A record that passes `isValidAssignmentRecord`'s SHAPE checks can still
|
|
175
|
+
* have been hand-edited to a different arm/block/position/propensity while keeping every field the
|
|
176
|
+
* right TYPE (e.g. flipping `arm:"direct"` to `arm:"reference"`, or `position:99`). The only way to
|
|
177
|
+
* catch that is to REDERIVE the assignment from the record's own tuple
|
|
178
|
+
* `(experiment, stratum, seed, index)` via the SAME pure `assignArm` that produced it, and compare —
|
|
179
|
+
* never trust the stored arm/block/position/propensity on their own. `arms` is the experiment's own
|
|
180
|
+
* registered arm set (from its `dz experiment init` config), passed in by the caller — this module
|
|
181
|
+
* stays pure and fs-free.
|
|
182
|
+
*/
|
|
183
|
+
export function verifyAssignmentRecord(record: AssignmentRecord, arms: readonly string[], pinnedSeed?: number): VerifyAssignmentResult {
|
|
184
|
+
// Lead delta after Codex r2 (BLOCKER): rederiving from `record.seed` proves the record is
|
|
185
|
+
// internally CONSISTENT, never that it is AUTHENTIC — a row carrying any seed at all passes,
|
|
186
|
+
// because the check compares the row with itself. The experiment's pinned seed (from its
|
|
187
|
+
// `dz experiment init` config, the one value fixed before the first assignment) is the trusted
|
|
188
|
+
// side of the comparison; a row that disagrees with it is tampered, however well-formed.
|
|
189
|
+
if (pinnedSeed !== undefined && record.seed !== pinnedSeed) {
|
|
190
|
+
return {
|
|
191
|
+
ok: false,
|
|
192
|
+
reason: `assignment-tampered: task ${JSON.stringify(record.taskId)} carries seed ${record.seed}, but experiment ${JSON.stringify(record.experiment)} is pinned to seed ${pinnedSeed} — a record's own seed can never authenticate it`,
|
|
193
|
+
expected: null,
|
|
194
|
+
actual: { arm: record.arm, block: record.block, position: record.position, propensity: record.propensity },
|
|
195
|
+
};
|
|
196
|
+
}
|
|
197
|
+
const recomputed = assignArm({ experiment: record.experiment, stratum: record.stratum, seed: record.seed, index: record.index, arms });
|
|
198
|
+
const actual = { arm: record.arm, block: record.block, position: record.position, propensity: record.propensity };
|
|
199
|
+
const expected = recomputed.ok
|
|
200
|
+
? { arm: recomputed.arm, block: recomputed.block, position: recomputed.position, propensity: recomputed.propensity }
|
|
201
|
+
: { arm: '(tuple refused)', block: -1, position: -1, propensity: -1 };
|
|
202
|
+
|
|
203
|
+
if (!recomputed.ok) {
|
|
204
|
+
return {
|
|
205
|
+
ok: false,
|
|
206
|
+
reason: `assignment-tampered: task ${JSON.stringify(record.taskId)}'s stored tuple (experiment=${JSON.stringify(record.experiment)}, stratum=${JSON.stringify(record.stratum)}, seed=${record.seed}, index=${record.index}) no longer rederives a valid assignment (${recomputed.reason}) — the stored record does not match what assignArm would produce from its own tuple`,
|
|
207
|
+
expected,
|
|
208
|
+
actual,
|
|
209
|
+
};
|
|
210
|
+
}
|
|
211
|
+
if (expected.arm !== actual.arm || expected.block !== actual.block || expected.position !== actual.position || expected.propensity !== actual.propensity) {
|
|
212
|
+
return {
|
|
213
|
+
ok: false,
|
|
214
|
+
reason: `assignment-tampered: task ${JSON.stringify(record.taskId)}'s stored assignment does not match what assignArm rederives from its own tuple (experiment=${JSON.stringify(record.experiment)}, stratum=${JSON.stringify(record.stratum)}, seed=${record.seed}, index=${record.index}) — the journal line was modified after it was written`,
|
|
215
|
+
expected,
|
|
216
|
+
actual,
|
|
217
|
+
};
|
|
218
|
+
}
|
|
219
|
+
// fix-round-1 HIGH #7 (named explicitly in the brief, beyond what recompute-and-compare already
|
|
220
|
+
// implies): timestamp FORMAT, propensity RANGE, and arm MEMBERSHIP. Recompute already makes a
|
|
221
|
+
// WRONG arm/propensity/block/position fail above; these three checks catch what recompute cannot
|
|
222
|
+
// — `assignArm` never produces a `ts`/`assignedAt` at all (they are wall-clock, not derived from
|
|
223
|
+
// the tuple), so a hand-edited "not-a-date" needs its own check.
|
|
224
|
+
if (!isValidIsoTimestamp(record.ts) || !isValidIsoTimestamp(record.assignedAt)) {
|
|
225
|
+
return {
|
|
226
|
+
ok: false,
|
|
227
|
+
reason: `assignment-tampered: task ${JSON.stringify(record.taskId)}'s ts/assignedAt is not a valid ISO-8601 UTC timestamp (ts=${JSON.stringify(record.ts)}, assignedAt=${JSON.stringify(record.assignedAt)})`,
|
|
228
|
+
expected,
|
|
229
|
+
actual,
|
|
230
|
+
};
|
|
231
|
+
}
|
|
232
|
+
if (!(record.propensity >= 0 && record.propensity <= 1)) {
|
|
233
|
+
return {
|
|
234
|
+
ok: false,
|
|
235
|
+
reason: `assignment-tampered: task ${JSON.stringify(record.taskId)}'s propensity ${record.propensity} is outside the valid [0,1] range`,
|
|
236
|
+
expected,
|
|
237
|
+
actual,
|
|
238
|
+
};
|
|
239
|
+
}
|
|
240
|
+
if (!arms.includes(record.arm)) {
|
|
241
|
+
return {
|
|
242
|
+
ok: false,
|
|
243
|
+
reason: `assignment-tampered: task ${JSON.stringify(record.taskId)}'s arm ${JSON.stringify(record.arm)} is not one of this experiment's registered arms (${arms.join(', ')})`,
|
|
244
|
+
expected,
|
|
245
|
+
actual,
|
|
246
|
+
};
|
|
247
|
+
}
|
|
248
|
+
return { ok: true };
|
|
249
|
+
}
|
|
250
|
+
|
|
251
|
+
/**
|
|
252
|
+
* Parse a JSONL assignments journal (one record per line) into records, WITHOUT touching the
|
|
253
|
+
* filesystem — the caller reads the file, this only parses its text (NFR-1: the core stays
|
|
254
|
+
* `node:fs`-free). A line that is blank is skipped silently (append-only files end in a trailing
|
|
255
|
+
* newline); a line that is non-blank but does not parse as JSON, or parses to something missing a
|
|
256
|
+
* required field, counts toward `malformedLines` and is otherwise ignored — corruption is named, not
|
|
257
|
+
* folded into "not assigned" (A5).
|
|
258
|
+
*/
|
|
259
|
+
export function readAssignments(text: string): ReadAssignmentsResult {
|
|
260
|
+
const records: AssignmentRecord[] = [];
|
|
261
|
+
let malformedLines = 0;
|
|
262
|
+
for (const raw of String(text ?? '').split('\n')) {
|
|
263
|
+
const line = raw.trim();
|
|
264
|
+
if (line === '') continue;
|
|
265
|
+
let parsed: unknown;
|
|
266
|
+
try {
|
|
267
|
+
parsed = JSON.parse(line);
|
|
268
|
+
} catch {
|
|
269
|
+
malformedLines++;
|
|
270
|
+
continue;
|
|
271
|
+
}
|
|
272
|
+
if (isValidAssignmentRecord(parsed)) records.push(parsed);
|
|
273
|
+
else malformedLines++;
|
|
274
|
+
}
|
|
275
|
+
return { records, malformedLines };
|
|
276
|
+
}
|