ruvnet-brain 4.3.34 → 4.3.36

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -34,6 +34,8 @@ import os from 'node:os';
34
34
  import path from 'node:path';
35
35
  import { fileURLToPath, pathToFileURL } from 'node:url';
36
36
  import { extractZip } from '../../kb/zip-extract.mjs';
37
+ import { validateCoverageLedger } from '../coverage-integrity.mjs';
38
+ import { retiredFixtureStores } from '../fixture-denominator.mjs';
37
39
 
38
40
  const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '../..');
39
41
 
@@ -153,30 +155,50 @@ export function scoreQuestion({ store, expectedPath, results }) {
153
155
  };
154
156
  }
155
157
 
156
- /** Roll per-question rows into the counts the predicates are evaluated on. */
158
+ /**
159
+ * Roll per-question rows into the counts the predicates are evaluated on.
160
+ *
161
+ * ADR-0091 D7.4: a row marked `retired: true` (its repository has no row at all in a complete sealed
162
+ * coverage observation) is excluded from EVERY count, and `retired` is added -- but only when it is
163
+ * non-zero, so a report with no retirements is byte-for-byte the pre-D7 shape and readable in both
164
+ * directions. An older reader of a report WITH retirements re-derives different totals and fails
165
+ * closed, which is the intended behavior (D4's pre-download check turns that into a seed fallback).
166
+ */
157
167
  export function tally(rows) {
158
- const completed = rows.filter((r) => !r.error).length;
168
+ const live = rows.filter((r) => r.retired !== true);
169
+ const retired = rows.length - live.length;
170
+ const completed = live.filter((r) => !r.error).length;
159
171
  return {
160
- questions: rows.length,
172
+ questions: live.length,
161
173
  completed,
162
- errors: rows.length - completed,
163
- repoCoverage: rows.filter((r) => r.repoCovered).length,
164
- hitTop1: rows.filter((r) => r.exactFileRank === 1).length,
165
- hitTop5: rows.filter((r) => r.exactFileRank != null && r.exactFileRank <= DEFAULT_K).length,
174
+ errors: live.length - completed,
175
+ repoCoverage: live.filter((r) => r.repoCovered).length,
176
+ hitTop1: live.filter((r) => r.exactFileRank === 1).length,
177
+ hitTop5: live.filter((r) => r.exactFileRank != null && r.exactFileRank <= DEFAULT_K).length,
178
+ ...(retired > 0 ? { retired } : {}),
166
179
  };
167
180
  }
168
181
 
182
+ const FLOOR_FAILURE = /below the accepted floor/;
183
+ /** ADR-0091 D7.6: the integrity failures. A floor miss is recorded in the report, never enforced. */
184
+ export const blockingFailures = (failures) => failures.filter((failure) => !FLOOR_FAILURE.test(failure));
185
+
169
186
  /**
170
187
  * EVERY blocking predicate, in one place, evaluated over counts alone. Returns the full failure
171
188
  * list rather than the first failure, so one run tells an operator everything that is wrong.
189
+ * The denominator is the frozen fixture minus retired questions (ADR-0091 D7.4) -- the SAME exclusion
190
+ * tally applies, or the totals re-derivation and the gate would disagree.
172
191
  */
173
192
  export function evaluateGate({ totals, floorValue, fixtureCount }) {
174
193
  const failures = [];
175
- if (totals.questions !== fixtureCount) failures.push(`asked ${totals.questions} of ${fixtureCount} frozen questions`);
194
+ const retired = Number.isSafeInteger(totals.retired) && totals.retired > 0 ? totals.retired : 0;
195
+ const expected = fixtureCount - retired;
196
+ const of = retired ? `${expected} (${fixtureCount} frozen, ${retired} retired)` : `${fixtureCount}`;
197
+ if (totals.questions !== expected) failures.push(`asked ${totals.questions} of ${of} frozen questions`);
176
198
  if (totals.errors !== 0) failures.push(`${totals.errors} question(s) failed to complete`);
177
- if (totals.completed !== fixtureCount) failures.push(`${totals.completed} of ${fixtureCount} questions completed`);
178
- if (totals.repoCoverage !== fixtureCount) {
179
- failures.push(`${fixtureCount - totals.repoCoverage} repository(ies) returned nothing of their own`);
199
+ if (totals.completed !== expected) failures.push(`${totals.completed} of ${of} questions completed`);
200
+ if (totals.repoCoverage !== expected) {
201
+ failures.push(`${expected - totals.repoCoverage} repository(ies) returned nothing of their own`);
180
202
  }
181
203
  if (totals.hitTop5 < floorValue) {
182
204
  failures.push(`exact-file Hit@5 regressed to ${totals.hitTop5}, below the accepted floor of ${floorValue}`);
@@ -184,7 +206,73 @@ export function evaluateGate({ totals, floorValue, fixtureCount }) {
184
206
  return { verdict: failures.length === 0 ? 'PASS' : 'FAIL', failures };
185
207
  }
186
208
 
187
- export function validateRecallReport({ report, archive, expectedFixtureSha256 = null, floorValue = null, floorFile = null } = {}) {
209
+ function parseSealedCoverage(bytes, label) {
210
+ let coverage;
211
+ try { coverage = JSON.parse(Buffer.from(bytes).toString('utf8')); }
212
+ catch (error) { fail(`${label} is unreadable (${error.message})`); }
213
+ const checked = validateCoverageLedger(coverage);
214
+ if (coverage?.kind !== 'ruvnet-brain-corpus-coverage' || !checked.valid) {
215
+ fail(`${label} is not a valid sealed ruvnet-brain-corpus-coverage ledger (${checked.failures.join('; ') || coverage?.kind})`);
216
+ }
217
+ return coverage;
218
+ }
219
+
220
+ const sortedLower = (values) => [...new Set(values.map((value) => String(value).toLowerCase()))].sort();
221
+
222
+ /**
223
+ * ADR-0091 D7.3 -- readers verify, never trust. The retired set a report CLAIMS (its `retirement`
224
+ * block and the rows it marks `retired: true`) is recomputed here, independently, from coverage bytes
225
+ * the reader obtained itself, through the one shared retirement function (scripts/fixture-
226
+ * denominator.mjs, also used by the release canary). Returns the failures; empty means consistent.
227
+ *
228
+ * A reader with no verifiable coverage treats ANY claimed retirement as invalid, and every claimed
229
+ * store must be retired by the coverage the claim names. The check is claimed ⊆ recomputed, not
230
+ * equality: retirement only ever REMOVES a question from the denominator, so the unsafe direction is
231
+ * an over-claim (hiding an unanswered question). An under-claim keeps a question in the denominator,
232
+ * which can only make the gate stricter. A report that claims nothing needs no coverage at all, so
233
+ * the pre-D7 path (every report today) reads exactly as it did.
234
+ */
235
+ export function retirementFailures({ report, coverageBytes = null, fixtureStores = null }) {
236
+ const rows = Array.isArray(report?.rows) ? report.rows : [];
237
+ const markedRows = rows.filter((row) => row?.retired === true);
238
+ const claim = report?.retirement;
239
+ const claims = claim !== undefined || markedRows.length > 0 || report?.totals?.retired !== undefined;
240
+ if (!claims) return [];
241
+ if (!claim || typeof claim !== 'object' || !HEX64.test(String(claim.coverageSha256 || ''))
242
+ || !Array.isArray(claim.stores) || !claim.stores.length) {
243
+ return ['repo-recall report marks questions retired without a well-formed retirement block'];
244
+ }
245
+ const failures = [];
246
+ const claimed = sortedLower(claim.stores);
247
+ if (JSON.stringify(claimed) !== JSON.stringify(sortedLower(markedRows.map((row) => row.store)))) {
248
+ failures.push('repo-recall report\'s retirement block does not name exactly the rows it marks retired');
249
+ }
250
+ // A retired question was never asked, so it can carry no evidence of an answer.
251
+ if (markedRows.some((row) => row.repoCovered !== false || row.exactFileRank !== null || row.error !== undefined)) {
252
+ failures.push('repo-recall report credits a retired question with an answer');
253
+ }
254
+ if (coverageBytes == null) {
255
+ return [...failures, 'repo-recall report claims retired question(s) but no coverage was supplied to verify them against'];
256
+ }
257
+ if (sha256(Buffer.from(coverageBytes)) !== claim.coverageSha256) {
258
+ return [...failures, 'repo-recall report\'s retirement was measured against different coverage bytes than the ones supplied'];
259
+ }
260
+ let coverage;
261
+ try { coverage = parseSealedCoverage(coverageBytes, 'retirement coverage'); }
262
+ catch (error) { return [...failures, error.message]; }
263
+ const recomputed = new Set(retiredFixtureStores({ coverage, fixtureStores: fixtureStores ?? rows.map((row) => row.store) }));
264
+ const unsupported = claimed.filter((store) => !recomputed.has(store));
265
+ if (unsupported.length) {
266
+ failures.push(`repo-recall report claims [${unsupported.join(', ')}] retired, but the coverage it names does not `
267
+ + 'retire them (they have a repository row, or the enumeration is not complete)');
268
+ }
269
+ return failures;
270
+ }
271
+
272
+ export function validateRecallReport({
273
+ report, archive, expectedFixtureSha256 = null, floorValue = null, floorFile = null,
274
+ coverageBytes = null, fixtureStores = null,
275
+ } = {}) {
188
276
  const failures = [];
189
277
  if (!report || typeof report !== 'object') fail('repo-recall report is not an object');
190
278
  if (report.schemaVersion !== RECALL_SCHEMA_VERSION || report.kind !== RECALL_KIND) {
@@ -203,6 +291,7 @@ export function validateRecallReport({ report, archive, expectedFixtureSha256 =
203
291
  if (canonical(recomputed) !== canonical(report.totals)) {
204
292
  failures.push('repo-recall report totals do not re-derive from its own rows');
205
293
  }
294
+ failures.push(...retirementFailures({ report, coverageBytes, fixtureStores }));
206
295
  }
207
296
  if (failures.length) fail(`repo-recall report invalid: ${failures.join('; ')}`);
208
297
  // NEVER trust the report's own `floor.value`. A report that declares its own bar could declare
@@ -231,14 +320,16 @@ export function validateRecallReport({ report, archive, expectedFixtureSha256 =
231
320
  // quality on the release path is gated by scripts/retrieval-canary.mjs through a real
232
321
  // installed host, which is the instrument that belongs in that role.
233
322
  const gate = evaluateGate({ totals: report.totals, floorValue: floor, fixtureCount: report.fixture.questionCount });
234
- const blocking = gate.failures.filter((f) => !/below the accepted floor/.test(f));
323
+ const blocking = blockingFailures(gate.failures);
235
324
  report.gate = { blocking: blocking.length > 0, verdict: gate.verdict, failures: gate.failures, enforced: blocking };
236
325
  if (blocking.length) fail(`repo-recall integrity FAILED: ${blocking.join('; ')}`);
237
326
  return report;
238
327
  }
239
328
 
240
329
  /** Read a detached report beside an archive and enforce the gate. Mirrors readAccuracyReport. */
241
- export function readRecallReport({ reportFile, archive, expectedFixtureSha256 = null, floorValue = null, floorFile = null } = {}) {
330
+ export function readRecallReport({
331
+ reportFile, archive, expectedFixtureSha256 = null, floorValue = null, floorFile = null, coverageBytes = null, fixtureStores = null,
332
+ } = {}) {
242
333
  const resolved = path.resolve(reportFile || '');
243
334
  if (!resolved || !fs.existsSync(resolved)) fail(`detached repo-recall report missing (${resolved || 'no path supplied'})`);
244
335
  const stat = fs.lstatSync(resolved);
@@ -246,7 +337,7 @@ export function readRecallReport({ reportFile, archive, expectedFixtureSha256 =
246
337
  let parsed;
247
338
  try { parsed = JSON.parse(fs.readFileSync(resolved, 'utf8')); }
248
339
  catch (error) { fail(`detached repo-recall report unreadable/corrupt (${error.message})`); }
249
- const report = validateRecallReport({ report: parsed, archive, expectedFixtureSha256, floorValue, floorFile });
340
+ const report = validateRecallReport({ report: parsed, archive, expectedFixtureSha256, floorValue, floorFile, coverageBytes, fixtureStores });
250
341
  return { identity: { file: path.basename(resolved), sha256: sha256File(resolved), bytes: stat.size }, report };
251
342
  }
252
343
 
@@ -256,9 +347,22 @@ export function readRecallReport({ reportFile, archive, expectedFixtureSha256 =
256
347
  * forge-ask-all.mjs — the exact bytes a customer installs — not the checkout's copy.
257
348
  */
258
349
  export async function runRepoRecall({
259
- kbDir, fixtureFile, floorFile, archive = null, k = DEFAULT_K, searchAll = null, now = () => new Date(),
350
+ kbDir, fixtureFile, floorFile, archive = null, k = DEFAULT_K, searchAll = null, now = () => new Date(), coverageFile = null,
260
351
  } = {}) {
261
352
  const fixture = loadFixture(fixtureFile);
353
+ // ADR-0091 D7.2: a fixture repository with NO row in a complete sealed coverage observation is
354
+ // retired -- its question is not asked (there is no store to ask) and it leaves the denominator.
355
+ // The fixture itself is never edited (D7.5): its digest is what seeds and the canary re-verify.
356
+ let retirement = null;
357
+ if (coverageFile) {
358
+ const resolved = path.resolve(coverageFile);
359
+ if (!fs.existsSync(resolved)) fail(`retirement coverage missing (${resolved})`);
360
+ const bytes = fs.readFileSync(resolved);
361
+ const coverage = parseSealedCoverage(bytes, 'retirement coverage');
362
+ const stores = retiredFixtureStores({ coverage, fixtureStores: fixture.questions.map((q) => q.store) });
363
+ if (stores.length) retirement = { coverageSha256: sha256(bytes), stores };
364
+ }
365
+ const retiredSet = new Set(retirement?.stores || []);
262
366
  // Fail-soft, same reason as the reader: a floor recorded against another fixture is a note, not a
263
367
  // reason to stop. Nothing here refuses a candidate any more.
264
368
  let floor = null; let floorValue = ABSOLUTE_FLOOR;
@@ -291,6 +395,11 @@ export async function runRepoRecall({
291
395
 
292
396
  const rows = [];
293
397
  for (const question of fixture.questions) {
398
+ if (retiredSet.has(question.store.toLowerCase())) {
399
+ rows.push({ store: question.store, expectedPath: question.expectedPath, retired: true,
400
+ repoCovered: false, exactFileRank: null, returnedPaths: [] });
401
+ continue;
402
+ }
294
403
  try {
295
404
  const out = await search({ dir: kbDir, query: question.query, k, repos: [question.store] });
296
405
  // searchAll reports a store that could not be OPENED as an "ERR: ..." string in perRepo and
@@ -346,6 +455,8 @@ export async function runRepoRecall({
346
455
  questionCount: fixture.questions.length,
347
456
  shape: 'exactly one human-written question per repository',
348
457
  },
458
+ // Present ONLY when something retired, so a report with none is the pre-D7 shape byte for byte.
459
+ ...(retirement ? { retirement } : {}),
349
460
  protocol: { entryPoint, k, repositoryScope: 'explicit', scoring: 'exact labeled file path within top-k of results from the requested repository' },
350
461
  floor: { value: floorValue, committed: floor?.hitTop5Floor ?? null, absolute: ABSOLUTE_FLOOR, acceptedForRelease: floor?.acceptedForRelease ?? null },
351
462
  totals,
@@ -370,7 +481,7 @@ const arg = (argv, name, fallback = null) => {
370
481
  * Measure a SEALED archive: extract it, find the store root, and grade that — never the build
371
482
  * directory the archive was assembled from. Returns the report bound to the archive's own digest.
372
483
  */
373
- export async function runRepoRecallOnBundle({ bundleFile, fixtureFile, floorFile, k = DEFAULT_K } = {}) {
484
+ export async function runRepoRecallOnBundle({ bundleFile, fixtureFile, floorFile, k = DEFAULT_K, coverageFile = null } = {}) {
374
485
  const bundle = path.resolve(bundleFile || '');
375
486
  if (!bundle || !fs.existsSync(bundle) || !fs.statSync(bundle).isFile()) {
376
487
  fail(`archive missing (${bundle || 'no path supplied'})`);
@@ -389,17 +500,19 @@ export async function runRepoRecallOnBundle({ bundleFile, fixtureFile, floorFile
389
500
  };
390
501
  walk(tmp);
391
502
  if (roots.length !== 1) fail(`expected exactly one ARCHIVE-MANIFEST.json in the archive, found ${roots.length}`);
392
- return await runRepoRecall({ kbDir: roots[0], fixtureFile, floorFile, archive, k });
503
+ return await runRepoRecall({ kbDir: roots[0], fixtureFile, floorFile, archive, k, coverageFile });
393
504
  } finally {
394
505
  fs.rmSync(tmp, { recursive: true, force: true });
395
506
  }
396
507
  }
397
508
 
398
- export async function main(argv = process.argv.slice(2)) {
509
+ // `searchAll` is a test seam only (the CLI never passes it): it lets the exit-code contract be proven
510
+ // without an embedded corpus. Unset, --kb grades the archive's own shipped entry point as always.
511
+ export async function main(argv = process.argv.slice(2), { searchAll = null } = {}) {
399
512
  const kbDir = arg(argv, '--kb');
400
513
  const bundleFile = arg(argv, '--bundle');
401
514
  if (!kbDir && !bundleFile) {
402
- process.stderr.write('usage: repo-recall.mjs (--bundle <archive.zip> | --kb <extracted root>) [--out <report.json>] [--fixture <file>] [--floor <file>]\n');
515
+ process.stderr.write('usage: repo-recall.mjs (--bundle <archive.zip> | --kb <extracted root>) [--out <report.json>] [--fixture <file>] [--floor <file>] [--coverage <sealed coverage>]\n');
403
516
  return 64;
404
517
  }
405
518
  const { report, gate } = bundleFile
@@ -407,11 +520,14 @@ export async function main(argv = process.argv.slice(2)) {
407
520
  bundleFile: path.resolve(bundleFile),
408
521
  fixtureFile: arg(argv, '--fixture'),
409
522
  floorFile: arg(argv, '--floor'),
523
+ coverageFile: arg(argv, '--coverage'),
410
524
  })
411
525
  : await runRepoRecall({
412
526
  kbDir: path.resolve(kbDir),
527
+ searchAll,
413
528
  fixtureFile: arg(argv, '--fixture'),
414
529
  floorFile: arg(argv, '--floor'),
530
+ coverageFile: arg(argv, '--coverage'),
415
531
  });
416
532
  const out = arg(argv, '--out') || (bundleFile ? `${path.resolve(bundleFile)}.recall.json` : null);
417
533
  if (out) fs.writeFileSync(path.resolve(out), `${JSON.stringify(report, null, 2)}\n`);
@@ -423,10 +539,15 @@ export async function main(argv = process.argv.slice(2)) {
423
539
  repositoriesAnswering: `${t.repoCoverage}/${t.questions}`,
424
540
  exactFileTop1: `${t.hitTop1}/${t.questions}`,
425
541
  exactFileTop5: `${t.hitTop5}/${t.questions}`,
542
+ ...(t.retired ? { retired: t.retired } : {}),
426
543
  floor: report.floor.value,
427
544
  failures: gate.failures,
545
+ enforced: blockingFailures(gate.failures),
428
546
  }, null, 2)}\n`);
429
- return gate.verdict === 'PASS' ? 0 : 1;
547
+ // ADR-0091 D7.6: the exit code derives from the INTEGRITY failures only. A floor miss stays in the
548
+ // report (state/failures) and is never fatal -- exiting 1 on it would silently restore the blocking
549
+ // ratchet ADR-086's 2026-09-15 amendment removed, the moment anyone re-accepts the floor.
550
+ return blockingFailures(gate.failures).length ? 1 : 0;
430
551
  }
431
552
 
432
553
  // Realpath both sides: argv[1] is whatever the caller typed, while Node resolves import.meta.url
@@ -34,6 +34,14 @@
34
34
  // complete. A bounded run can therefore be read, reported and compared, but it can never seal a
35
35
  // publishable corpus receipt.
36
36
  //
37
+ // QUESTION SAMPLING (ADR-0091 D2). `--sample-questions <n>` measures n oracle questions in total,
38
+ // chosen by selectQuestionSample(): stratified across partitions (one question from each of n
39
+ // partitions before any partition gets a second) and ordered by sha256(seed, id), so the same
40
+ // seed, oracle and n always select the same questions. It exists because the full diagnostic is
41
+ // ~1,164 queries at ~4.2 s each on a hosted runner (82 minutes), and C3 no longer blocks anything.
42
+ // A sampled report is bounded like any other bounded run: `coverage.complete: false`, the seed and
43
+ // the exact selected label ids recorded in `coverage.bounded.sample`. Omit the flag for the full audit.
44
+ //
37
45
  // TIMEOUTS. The threshold text says errors and timeouts count as failures (they are never excluded
38
46
  // from the denominator), and the Step 15 proof text additionally names "timeout" as a standalone
39
47
  // blocker. Both readings are honoured, strictly: a timeout is counted as a failure in the
@@ -74,6 +82,12 @@ export const THRESHOLD_DENOMINATOR = 20;
74
82
  export const QUERY_MODES = Object.freeze(['explicit-repository', 'full-corpus']);
75
83
  export const DEFAULT_QUERY_TIMEOUT_MS = 120_000;
76
84
  export const DEFAULT_ORACLE_FILE = 'data/retrieval-accuracy-oracle.json';
85
+ // ADR-0091 D2 — the sample the corpus pipeline measures. 80 questions x 2 modes = 160 queries. At the
86
+ // hosted-runner cost of 4.23 s/query (82 min / 1,164 queries, the ADR's measured C3 run) that is
87
+ // 677 s of querying, ~11.3 min, leaving ~3.7 min of the 15-minute bound for extraction and model
88
+ // load. corpus-seed.yml passes this same number; a unit test pins the two together.
89
+ export const C3_DIAGNOSTIC_SAMPLE_QUESTIONS = 80;
90
+ export const DEFAULT_SAMPLE_SEED = 'c3-diagnostic-sample/1';
77
91
 
78
92
  const HEX64 = /^[a-f0-9]{64}$/;
79
93
  const HEX40 = /^[a-f0-9]{40}$/;
@@ -357,6 +371,37 @@ export function archiveStores(root) {
357
371
  .sort();
358
372
  }
359
373
 
374
+ /**
375
+ * Deterministic, stratified question sample (ADR-0091 D2). Partitions are ranked by
376
+ * sha256(seed, partition) and each partition's labels by sha256(seed, label id); the sample takes one
377
+ * label from every partition in rank order, then a second from every partition that has one, and so
378
+ * on until `size` labels are chosen. Same seed + same oracle + same size = same labels, always.
379
+ * Returns the chosen labels sorted by id.
380
+ */
381
+ export function selectQuestionSample({ labels, size, seed = DEFAULT_SAMPLE_SEED }) {
382
+ if (!Number.isSafeInteger(size) || size <= 0) fail('question sample size must be a positive integer');
383
+ if (typeof seed !== 'string' || !seed) fail('question sample seed must be a non-empty string');
384
+ const rank = (value) => sha256Of(`${seed}\u0000${value}`);
385
+ const byPartition = new Map();
386
+ for (const label of labels) {
387
+ if (!byPartition.has(label.partition)) byPartition.set(label.partition, []);
388
+ byPartition.get(label.partition).push({ label, key: rank(`label\u0000${label.id}`) });
389
+ }
390
+ const queues = [...byPartition.entries()]
391
+ .map(([partition, rows]) => ({ key: rank(`partition\u0000${partition}`), rows: rows.sort((a, b) => a.key.localeCompare(b.key)) }))
392
+ .sort((a, b) => a.key.localeCompare(b.key));
393
+ const chosen = [];
394
+ for (let depth = 0; chosen.length < size; depth += 1) {
395
+ let took = false;
396
+ for (const queue of queues) {
397
+ if (chosen.length >= size) break;
398
+ if (depth < queue.rows.length) { chosen.push(queue.rows[depth].label); took = true; }
399
+ }
400
+ if (!took) break; // the oracle has fewer labels than requested: the sample is every label
401
+ }
402
+ return chosen.sort((a, b) => a.id.localeCompare(b.id));
403
+ }
404
+
360
405
  async function defaultSearch({ dir, query, repos, timeoutMs }) {
361
406
  const module = await import('../../kb/forge-ask-all.mjs');
362
407
  let timer = null;
@@ -393,6 +438,8 @@ export async function runRetrievalAccuracy({
393
438
  outFile,
394
439
  storeLimit = null,
395
440
  sampleLimit = null,
441
+ sampleQuestions = null,
442
+ sampleSeed = DEFAULT_SAMPLE_SEED,
396
443
  modes = QUERY_MODES,
397
444
  timeoutMs = DEFAULT_QUERY_TIMEOUT_MS,
398
445
  search = defaultSearch,
@@ -407,6 +454,14 @@ export async function runRetrievalAccuracy({
407
454
  const oracle = readAccuracyOracle(oracleFile);
408
455
  const selectedModes = QUERY_MODES.filter((mode) => modes.includes(mode));
409
456
  if (!selectedModes.length) fail('no supported query mode selected');
457
+ if (sampleQuestions != null && (storeLimit != null || sampleLimit != null)) {
458
+ fail('--sample-questions is a whole-oracle sample; it cannot be combined with --stores or --sample');
459
+ }
460
+ const questionSample = sampleQuestions == null ? null
461
+ : selectQuestionSample({ labels: oracle.labels, size: sampleQuestions, seed: sampleSeed });
462
+ const sampledIds = questionSample ? new Set(questionSample.map((label) => label.id)) : null;
463
+ // Any sampling at all measures only what it sampled: unproduced slots are not charged and n is not N.
464
+ const sampling = sampleLimit != null || questionSample != null;
410
465
 
411
466
  const tmp = fs.mkdtempSync(path.join(os.tmpdir(), 'retrieval-accuracy-'));
412
467
  try {
@@ -437,7 +492,9 @@ export async function runRetrievalAccuracy({
437
492
  labelsByPartition.get(label.partition).push(label);
438
493
  }
439
494
  const orderedPartitions = [...oracle.partitions.values()].sort((a, b) => a.partition.localeCompare(b.partition));
440
- const measuredPartitions = storeLimit == null ? orderedPartitions : orderedPartitions.slice(0, storeLimit);
495
+ const measuredPartitions = sampledIds
496
+ ? orderedPartitions.filter((row) => (labelsByPartition.get(row.partition) || []).some((label) => sampledIds.has(label.id)))
497
+ : storeLimit == null ? orderedPartitions : orderedPartitions.slice(0, storeLimit);
441
498
 
442
499
  const partitions = [];
443
500
  let timeouts = 0;
@@ -446,12 +503,13 @@ export async function runRetrievalAccuracy({
446
503
  const all = (labelsByPartition.get(partition.partition) || [])
447
504
  .slice()
448
505
  .sort((a, b) => a.id.localeCompare(b.id));
449
- const selected = sampleLimit == null ? all : all.slice(0, sampleLimit);
506
+ const selected = sampledIds ? all.filter((label) => sampledIds.has(label.id))
507
+ : sampleLimit == null ? all : all.slice(0, sampleLimit);
450
508
  // THE DENOMINATOR. For a compliant oracle N comes from the unit inventory — 2 x min(100, U) —
451
509
  // never from how many labels happened to survive production. Every unproduced unit keeps its two
452
510
  // slots and scores them as misses below. A bounded --sample run is incomplete and unacceptable
453
511
  // regardless, so it measures only what it sampled.
454
- const unproducedSlots = oracle.c3Eligible && sampleLimit == null ? partition.unproduced : [];
512
+ const unproducedSlots = oracle.c3Eligible && !sampling ? partition.unproduced : [];
455
513
  for (const mode of selectedModes) {
456
514
  const row = {
457
515
  partition: partition.partition,
@@ -466,7 +524,7 @@ export async function runRetrievalAccuracy({
466
524
  failures: 0,
467
525
  errors: 0,
468
526
  timeouts: 0,
469
- sampled: sampleLimit != null && selected.length < all.length,
527
+ sampled: sampling && selected.length < all.length,
470
528
  oracleRows: all.length,
471
529
  failedLabels: [],
472
530
  };
@@ -514,7 +572,7 @@ export async function runRetrievalAccuracy({
514
572
  if (row.successes + row.failures !== row.n) {
515
573
  fail(`internal: partition ${row.partition} (${mode}) scored ${row.successes + row.failures} outcomes for n=${row.n}`);
516
574
  }
517
- if (oracle.c3Eligible && sampleLimit == null && row.n !== row.N) {
575
+ if (oracle.c3Eligible && !sampling && row.n !== row.N) {
518
576
  fail(`internal: partition ${row.partition} (${mode}) measured n=${row.n} but its inventory fixes N=${row.N}`);
519
577
  }
520
578
  row.state = meetsThreshold(row.successes, row.n) && row.timeouts === 0 ? 'PASS' : 'FAIL';
@@ -533,6 +591,7 @@ export async function runRetrievalAccuracy({
533
591
  const boundedReasons = [];
534
592
  if (storeLimit != null) boundedReasons.push(`--stores ${storeLimit}`);
535
593
  if (sampleLimit != null) boundedReasons.push(`--sample ${sampleLimit}`);
594
+ if (questionSample) boundedReasons.push(`--sample-questions ${sampleQuestions} (seed ${sampleSeed}): ${questionSample.length} of ${oracle.labels.length} oracle questions`);
536
595
  if (selectedModes.length !== QUERY_MODES.length) boundedReasons.push(`--modes ${selectedModes.join(',')}`);
537
596
  if (unmeasuredPartitions.length) boundedReasons.push(`${unmeasuredPartitions.length} oracle partition(s) not measured`);
538
597
  if (uncoveredArchiveStores.length) boundedReasons.push(`${uncoveredArchiveStores.length} shipped store(s) with no oracle coverage`);
@@ -568,7 +627,14 @@ export async function runRetrievalAccuracy({
568
627
  queryTimeoutMs: timeoutMs,
569
628
  coverage: {
570
629
  complete,
571
- bounded: complete ? null : { reasons: boundedReasons, storeLimit, sampleLimit, modes: selectedModes },
630
+ bounded: complete ? null : {
631
+ reasons: boundedReasons, storeLimit, sampleLimit, modes: selectedModes,
632
+ ...(questionSample ? { sample: {
633
+ method: 'stratified-by-partition/sha256-rank', seed: sampleSeed, requested: sampleQuestions,
634
+ questions: questionSample.length, oracleQuestions: oracle.labels.length,
635
+ labelIds: questionSample.map((label) => label.id),
636
+ } } : {}),
637
+ },
572
638
  archiveStores: shipped,
573
639
  oraclePartitions: orderedPartitions.length,
574
640
  measuredPartitions: measuredPartitions.length,
@@ -778,6 +844,8 @@ export async function main(argv = process.argv.slice(2)) {
778
844
  outFile: arg(argv, '--out'),
779
845
  storeLimit: positiveInt(arg(argv, '--stores'), '--stores'),
780
846
  sampleLimit: positiveInt(arg(argv, '--sample'), '--sample'),
847
+ sampleQuestions: positiveInt(arg(argv, '--sample-questions'), '--sample-questions'),
848
+ sampleSeed: arg(argv, '--sample-seed', DEFAULT_SAMPLE_SEED),
781
849
  modes: arg(argv, '--modes') ? String(arg(argv, '--modes')).split(',').map((mode) => mode.trim()) : QUERY_MODES,
782
850
  timeoutMs: positiveInt(arg(argv, '--timeout-ms'), '--timeout-ms') || DEFAULT_QUERY_TIMEOUT_MS,
783
851
  });
@@ -250,6 +250,19 @@ function retrospectiveBaselineFromTree({ extractedRoot, bundleFile, expectedTag,
250
250
  return { receipt, bytes, fileSha256: crypto.createHash('sha256').update(bytes).digest('hex'), root, archiveManifest };
251
251
  }
252
252
 
253
+ /**
254
+ * ADR-0091 D6.3: does the baseline archive carry the seed's own published tag? A code-release seed
255
+ * (vX.Y.Z) records that tag in its generation ledger. A corpus-generation seed CANNOT: its tag is
256
+ * corpus-sha256-<the archive's own digest>, which no file inside the archive can contain, and its
257
+ * ledger names the runtime that built it. So a content-addressed tag is proven by the archive digest
258
+ * it names; every other tag must equal the ledger's releaseTag, exactly as before.
259
+ */
260
+ export function baselineTagMatches({ publishedTag, ledgerReleaseTag, archiveSha256 }) {
261
+ const contentAddressed = /^corpus-sha256-([0-9a-f]{64})$/.exec(String(publishedTag || ''));
262
+ if (contentAddressed) return contentAddressed[1] === archiveSha256;
263
+ return typeof publishedTag === 'string' && publishedTag.length > 0 && ledgerReleaseTag === publishedTag;
264
+ }
265
+
253
266
  function observedBaselineFromTree({ extractedRoot, bundleFile, expectedTag, expectedSha256, expectedBytes }) {
254
267
  const ledgerFile = findNamed(extractedRoot, 'RVF-GENERATIONS.json');
255
268
  const root = path.dirname(ledgerFile);
@@ -259,7 +272,9 @@ function observedBaselineFromTree({ extractedRoot, bundleFile, expectedTag, expe
259
272
  fail('historical baseline generation ledger is malformed');
260
273
  }
261
274
  const archive = namedIdentity(bundleFile);
262
- if (ledger.releaseTag !== expectedTag) fail('historical baseline differs from the expected public release tag');
275
+ if (!baselineTagMatches({ publishedTag: expectedTag, ledgerReleaseTag: ledger.releaseTag, archiveSha256: archive.sha256 })) {
276
+ fail('historical baseline differs from the expected public release tag');
277
+ }
263
278
  if (!HEX64.test(String(expectedSha256 || '')) || archive.sha256 !== expectedSha256) {
264
279
  fail('historical baseline differs from the expected public archive SHA-256');
265
280
  }
@@ -476,7 +491,8 @@ export async function createPublicVerificationInputs({ baselineBundle, candidate
476
491
  fail('baseline archive bytes differ from release coverage');
477
492
  }
478
493
  if (seed.receiptSha256 !== baselineProof.fileSha256) fail('baseline receipt differs from release coverage');
479
- if (seed.tag !== baselineProof.receipt.releaseTag) {
494
+ if (!baselineTagMatches({ publishedTag: seed.tag, ledgerReleaseTag: baselineProof.receipt.releaseTag,
495
+ archiveSha256: baselineProof.receipt.archive.sha256 })) {
480
496
  fail('baseline release tag differs from release coverage');
481
497
  }
482
498
  const baselineStores = baselineProof.receipt.stores.map(({ name }) => name);