ruvnet-brain 4.3.34 → 4.3.36
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/package.json +1 -1
- package/plugin/.claude-plugin/plugin.json +1 -1
- package/plugin/.codex-plugin/plugin.json +1 -1
- package/plugin/scripts/coverage-integrity.mjs +58 -2
- package/scripts/approved-runtime.mjs +247 -26
- package/scripts/build-bundle.mjs +27 -1
- package/scripts/code-release-corpus.mjs +276 -0
- package/scripts/corpus-candidate.mjs +19 -3
- package/scripts/corpus-coverage-sidecar.mjs +202 -0
- package/scripts/corpus-currency.mjs +71 -0
- package/scripts/corpus-dispatch-decision.mjs +138 -0
- package/scripts/corpus-next-seed.mjs +250 -72
- package/scripts/corpus-reconcile.mjs +448 -75
- package/scripts/corpus-store-failure.mjs +82 -0
- package/scripts/corpus-watchdog.mjs +334 -0
- package/scripts/fixture-denominator.mjs +72 -0
- package/scripts/knowledge-input-digest.mjs +163 -0
- package/scripts/oracle/repo-recall.mjs +142 -21
- package/scripts/oracle/retrieval-accuracy.mjs +74 -6
- package/scripts/public-verification-inputs.mjs +18 -2
- package/scripts/rehearse-corpus-pipeline.mjs +651 -31
- package/scripts/rehearse-seed-selection.mjs +234 -0
- package/scripts/release.mjs +163 -18
- package/scripts/retrieval-canary.mjs +42 -20
- package/scripts/single-source-check.mjs +15 -2
- package/scripts/source-coverage.mjs +19 -3
|
@@ -34,6 +34,8 @@ import os from 'node:os';
|
|
|
34
34
|
import path from 'node:path';
|
|
35
35
|
import { fileURLToPath, pathToFileURL } from 'node:url';
|
|
36
36
|
import { extractZip } from '../../kb/zip-extract.mjs';
|
|
37
|
+
import { validateCoverageLedger } from '../coverage-integrity.mjs';
|
|
38
|
+
import { retiredFixtureStores } from '../fixture-denominator.mjs';
|
|
37
39
|
|
|
38
40
|
const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '../..');
|
|
39
41
|
|
|
@@ -153,30 +155,50 @@ export function scoreQuestion({ store, expectedPath, results }) {
|
|
|
153
155
|
};
|
|
154
156
|
}
|
|
155
157
|
|
|
156
|
-
/**
|
|
158
|
+
/**
|
|
159
|
+
* Roll per-question rows into the counts the predicates are evaluated on.
|
|
160
|
+
*
|
|
161
|
+
* ADR-0091 D7.4: a row marked `retired: true` (its repository has no row at all in a complete sealed
|
|
162
|
+
* coverage observation) is excluded from EVERY count, and `retired` is added -- but only when it is
|
|
163
|
+
* non-zero, so a report with no retirements is byte-for-byte the pre-D7 shape and readable in both
|
|
164
|
+
* directions. An older reader of a report WITH retirements re-derives different totals and fails
|
|
165
|
+
* closed, which is the intended behavior (D4's pre-download check turns that into a seed fallback).
|
|
166
|
+
*/
|
|
157
167
|
export function tally(rows) {
|
|
158
|
-
const
|
|
168
|
+
const live = rows.filter((r) => r.retired !== true);
|
|
169
|
+
const retired = rows.length - live.length;
|
|
170
|
+
const completed = live.filter((r) => !r.error).length;
|
|
159
171
|
return {
|
|
160
|
-
questions:
|
|
172
|
+
questions: live.length,
|
|
161
173
|
completed,
|
|
162
|
-
errors:
|
|
163
|
-
repoCoverage:
|
|
164
|
-
hitTop1:
|
|
165
|
-
hitTop5:
|
|
174
|
+
errors: live.length - completed,
|
|
175
|
+
repoCoverage: live.filter((r) => r.repoCovered).length,
|
|
176
|
+
hitTop1: live.filter((r) => r.exactFileRank === 1).length,
|
|
177
|
+
hitTop5: live.filter((r) => r.exactFileRank != null && r.exactFileRank <= DEFAULT_K).length,
|
|
178
|
+
...(retired > 0 ? { retired } : {}),
|
|
166
179
|
};
|
|
167
180
|
}
|
|
168
181
|
|
|
182
|
+
const FLOOR_FAILURE = /below the accepted floor/;
|
|
183
|
+
/** ADR-0091 D7.6: the integrity failures. A floor miss is recorded in the report, never enforced. */
|
|
184
|
+
export const blockingFailures = (failures) => failures.filter((failure) => !FLOOR_FAILURE.test(failure));
|
|
185
|
+
|
|
169
186
|
/**
|
|
170
187
|
* EVERY blocking predicate, in one place, evaluated over counts alone. Returns the full failure
|
|
171
188
|
* list rather than the first failure, so one run tells an operator everything that is wrong.
|
|
189
|
+
* The denominator is the frozen fixture minus retired questions (ADR-0091 D7.4) -- the SAME exclusion
|
|
190
|
+
* tally applies, or the totals re-derivation and the gate would disagree.
|
|
172
191
|
*/
|
|
173
192
|
export function evaluateGate({ totals, floorValue, fixtureCount }) {
|
|
174
193
|
const failures = [];
|
|
175
|
-
|
|
194
|
+
const retired = Number.isSafeInteger(totals.retired) && totals.retired > 0 ? totals.retired : 0;
|
|
195
|
+
const expected = fixtureCount - retired;
|
|
196
|
+
const of = retired ? `${expected} (${fixtureCount} frozen, ${retired} retired)` : `${fixtureCount}`;
|
|
197
|
+
if (totals.questions !== expected) failures.push(`asked ${totals.questions} of ${of} frozen questions`);
|
|
176
198
|
if (totals.errors !== 0) failures.push(`${totals.errors} question(s) failed to complete`);
|
|
177
|
-
if (totals.completed !==
|
|
178
|
-
if (totals.repoCoverage !==
|
|
179
|
-
failures.push(`${
|
|
199
|
+
if (totals.completed !== expected) failures.push(`${totals.completed} of ${of} questions completed`);
|
|
200
|
+
if (totals.repoCoverage !== expected) {
|
|
201
|
+
failures.push(`${expected - totals.repoCoverage} repository(ies) returned nothing of their own`);
|
|
180
202
|
}
|
|
181
203
|
if (totals.hitTop5 < floorValue) {
|
|
182
204
|
failures.push(`exact-file Hit@5 regressed to ${totals.hitTop5}, below the accepted floor of ${floorValue}`);
|
|
@@ -184,7 +206,73 @@ export function evaluateGate({ totals, floorValue, fixtureCount }) {
|
|
|
184
206
|
return { verdict: failures.length === 0 ? 'PASS' : 'FAIL', failures };
|
|
185
207
|
}
|
|
186
208
|
|
|
187
|
-
|
|
209
|
+
function parseSealedCoverage(bytes, label) {
|
|
210
|
+
let coverage;
|
|
211
|
+
try { coverage = JSON.parse(Buffer.from(bytes).toString('utf8')); }
|
|
212
|
+
catch (error) { fail(`${label} is unreadable (${error.message})`); }
|
|
213
|
+
const checked = validateCoverageLedger(coverage);
|
|
214
|
+
if (coverage?.kind !== 'ruvnet-brain-corpus-coverage' || !checked.valid) {
|
|
215
|
+
fail(`${label} is not a valid sealed ruvnet-brain-corpus-coverage ledger (${checked.failures.join('; ') || coverage?.kind})`);
|
|
216
|
+
}
|
|
217
|
+
return coverage;
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
const sortedLower = (values) => [...new Set(values.map((value) => String(value).toLowerCase()))].sort();
|
|
221
|
+
|
|
222
|
+
/**
|
|
223
|
+
* ADR-0091 D7.3 -- readers verify, never trust. The retired set a report CLAIMS (its `retirement`
|
|
224
|
+
* block and the rows it marks `retired: true`) is recomputed here, independently, from coverage bytes
|
|
225
|
+
* the reader obtained itself, through the one shared retirement function (scripts/fixture-
|
|
226
|
+
* denominator.mjs, also used by the release canary). Returns the failures; empty means consistent.
|
|
227
|
+
*
|
|
228
|
+
* A reader with no verifiable coverage treats ANY claimed retirement as invalid, and every claimed
|
|
229
|
+
* store must be retired by the coverage the claim names. The check is claimed ⊆ recomputed, not
|
|
230
|
+
* equality: retirement only ever REMOVES a question from the denominator, so the unsafe direction is
|
|
231
|
+
* an over-claim (hiding an unanswered question). An under-claim keeps a question in the denominator,
|
|
232
|
+
* which can only make the gate stricter. A report that claims nothing needs no coverage at all, so
|
|
233
|
+
* the pre-D7 path (every report today) reads exactly as it did.
|
|
234
|
+
*/
|
|
235
|
+
export function retirementFailures({ report, coverageBytes = null, fixtureStores = null }) {
|
|
236
|
+
const rows = Array.isArray(report?.rows) ? report.rows : [];
|
|
237
|
+
const markedRows = rows.filter((row) => row?.retired === true);
|
|
238
|
+
const claim = report?.retirement;
|
|
239
|
+
const claims = claim !== undefined || markedRows.length > 0 || report?.totals?.retired !== undefined;
|
|
240
|
+
if (!claims) return [];
|
|
241
|
+
if (!claim || typeof claim !== 'object' || !HEX64.test(String(claim.coverageSha256 || ''))
|
|
242
|
+
|| !Array.isArray(claim.stores) || !claim.stores.length) {
|
|
243
|
+
return ['repo-recall report marks questions retired without a well-formed retirement block'];
|
|
244
|
+
}
|
|
245
|
+
const failures = [];
|
|
246
|
+
const claimed = sortedLower(claim.stores);
|
|
247
|
+
if (JSON.stringify(claimed) !== JSON.stringify(sortedLower(markedRows.map((row) => row.store)))) {
|
|
248
|
+
failures.push('repo-recall report\'s retirement block does not name exactly the rows it marks retired');
|
|
249
|
+
}
|
|
250
|
+
// A retired question was never asked, so it can carry no evidence of an answer.
|
|
251
|
+
if (markedRows.some((row) => row.repoCovered !== false || row.exactFileRank !== null || row.error !== undefined)) {
|
|
252
|
+
failures.push('repo-recall report credits a retired question with an answer');
|
|
253
|
+
}
|
|
254
|
+
if (coverageBytes == null) {
|
|
255
|
+
return [...failures, 'repo-recall report claims retired question(s) but no coverage was supplied to verify them against'];
|
|
256
|
+
}
|
|
257
|
+
if (sha256(Buffer.from(coverageBytes)) !== claim.coverageSha256) {
|
|
258
|
+
return [...failures, 'repo-recall report\'s retirement was measured against different coverage bytes than the ones supplied'];
|
|
259
|
+
}
|
|
260
|
+
let coverage;
|
|
261
|
+
try { coverage = parseSealedCoverage(coverageBytes, 'retirement coverage'); }
|
|
262
|
+
catch (error) { return [...failures, error.message]; }
|
|
263
|
+
const recomputed = new Set(retiredFixtureStores({ coverage, fixtureStores: fixtureStores ?? rows.map((row) => row.store) }));
|
|
264
|
+
const unsupported = claimed.filter((store) => !recomputed.has(store));
|
|
265
|
+
if (unsupported.length) {
|
|
266
|
+
failures.push(`repo-recall report claims [${unsupported.join(', ')}] retired, but the coverage it names does not `
|
|
267
|
+
+ 'retire them (they have a repository row, or the enumeration is not complete)');
|
|
268
|
+
}
|
|
269
|
+
return failures;
|
|
270
|
+
}
|
|
271
|
+
|
|
272
|
+
export function validateRecallReport({
|
|
273
|
+
report, archive, expectedFixtureSha256 = null, floorValue = null, floorFile = null,
|
|
274
|
+
coverageBytes = null, fixtureStores = null,
|
|
275
|
+
} = {}) {
|
|
188
276
|
const failures = [];
|
|
189
277
|
if (!report || typeof report !== 'object') fail('repo-recall report is not an object');
|
|
190
278
|
if (report.schemaVersion !== RECALL_SCHEMA_VERSION || report.kind !== RECALL_KIND) {
|
|
@@ -203,6 +291,7 @@ export function validateRecallReport({ report, archive, expectedFixtureSha256 =
|
|
|
203
291
|
if (canonical(recomputed) !== canonical(report.totals)) {
|
|
204
292
|
failures.push('repo-recall report totals do not re-derive from its own rows');
|
|
205
293
|
}
|
|
294
|
+
failures.push(...retirementFailures({ report, coverageBytes, fixtureStores }));
|
|
206
295
|
}
|
|
207
296
|
if (failures.length) fail(`repo-recall report invalid: ${failures.join('; ')}`);
|
|
208
297
|
// NEVER trust the report's own `floor.value`. A report that declares its own bar could declare
|
|
@@ -231,14 +320,16 @@ export function validateRecallReport({ report, archive, expectedFixtureSha256 =
|
|
|
231
320
|
// quality on the release path is gated by scripts/retrieval-canary.mjs through a real
|
|
232
321
|
// installed host, which is the instrument that belongs in that role.
|
|
233
322
|
const gate = evaluateGate({ totals: report.totals, floorValue: floor, fixtureCount: report.fixture.questionCount });
|
|
234
|
-
const blocking = gate.failures
|
|
323
|
+
const blocking = blockingFailures(gate.failures);
|
|
235
324
|
report.gate = { blocking: blocking.length > 0, verdict: gate.verdict, failures: gate.failures, enforced: blocking };
|
|
236
325
|
if (blocking.length) fail(`repo-recall integrity FAILED: ${blocking.join('; ')}`);
|
|
237
326
|
return report;
|
|
238
327
|
}
|
|
239
328
|
|
|
240
329
|
/** Read a detached report beside an archive and enforce the gate. Mirrors readAccuracyReport. */
|
|
241
|
-
export function readRecallReport({
|
|
330
|
+
export function readRecallReport({
|
|
331
|
+
reportFile, archive, expectedFixtureSha256 = null, floorValue = null, floorFile = null, coverageBytes = null, fixtureStores = null,
|
|
332
|
+
} = {}) {
|
|
242
333
|
const resolved = path.resolve(reportFile || '');
|
|
243
334
|
if (!resolved || !fs.existsSync(resolved)) fail(`detached repo-recall report missing (${resolved || 'no path supplied'})`);
|
|
244
335
|
const stat = fs.lstatSync(resolved);
|
|
@@ -246,7 +337,7 @@ export function readRecallReport({ reportFile, archive, expectedFixtureSha256 =
|
|
|
246
337
|
let parsed;
|
|
247
338
|
try { parsed = JSON.parse(fs.readFileSync(resolved, 'utf8')); }
|
|
248
339
|
catch (error) { fail(`detached repo-recall report unreadable/corrupt (${error.message})`); }
|
|
249
|
-
const report = validateRecallReport({ report: parsed, archive, expectedFixtureSha256, floorValue, floorFile });
|
|
340
|
+
const report = validateRecallReport({ report: parsed, archive, expectedFixtureSha256, floorValue, floorFile, coverageBytes, fixtureStores });
|
|
250
341
|
return { identity: { file: path.basename(resolved), sha256: sha256File(resolved), bytes: stat.size }, report };
|
|
251
342
|
}
|
|
252
343
|
|
|
@@ -256,9 +347,22 @@ export function readRecallReport({ reportFile, archive, expectedFixtureSha256 =
|
|
|
256
347
|
* forge-ask-all.mjs — the exact bytes a customer installs — not the checkout's copy.
|
|
257
348
|
*/
|
|
258
349
|
export async function runRepoRecall({
|
|
259
|
-
kbDir, fixtureFile, floorFile, archive = null, k = DEFAULT_K, searchAll = null, now = () => new Date(),
|
|
350
|
+
kbDir, fixtureFile, floorFile, archive = null, k = DEFAULT_K, searchAll = null, now = () => new Date(), coverageFile = null,
|
|
260
351
|
} = {}) {
|
|
261
352
|
const fixture = loadFixture(fixtureFile);
|
|
353
|
+
// ADR-0091 D7.2: a fixture repository with NO row in a complete sealed coverage observation is
|
|
354
|
+
// retired -- its question is not asked (there is no store to ask) and it leaves the denominator.
|
|
355
|
+
// The fixture itself is never edited (D7.5): its digest is what seeds and the canary re-verify.
|
|
356
|
+
let retirement = null;
|
|
357
|
+
if (coverageFile) {
|
|
358
|
+
const resolved = path.resolve(coverageFile);
|
|
359
|
+
if (!fs.existsSync(resolved)) fail(`retirement coverage missing (${resolved})`);
|
|
360
|
+
const bytes = fs.readFileSync(resolved);
|
|
361
|
+
const coverage = parseSealedCoverage(bytes, 'retirement coverage');
|
|
362
|
+
const stores = retiredFixtureStores({ coverage, fixtureStores: fixture.questions.map((q) => q.store) });
|
|
363
|
+
if (stores.length) retirement = { coverageSha256: sha256(bytes), stores };
|
|
364
|
+
}
|
|
365
|
+
const retiredSet = new Set(retirement?.stores || []);
|
|
262
366
|
// Fail-soft, same reason as the reader: a floor recorded against another fixture is a note, not a
|
|
263
367
|
// reason to stop. Nothing here refuses a candidate any more.
|
|
264
368
|
let floor = null; let floorValue = ABSOLUTE_FLOOR;
|
|
@@ -291,6 +395,11 @@ export async function runRepoRecall({
|
|
|
291
395
|
|
|
292
396
|
const rows = [];
|
|
293
397
|
for (const question of fixture.questions) {
|
|
398
|
+
if (retiredSet.has(question.store.toLowerCase())) {
|
|
399
|
+
rows.push({ store: question.store, expectedPath: question.expectedPath, retired: true,
|
|
400
|
+
repoCovered: false, exactFileRank: null, returnedPaths: [] });
|
|
401
|
+
continue;
|
|
402
|
+
}
|
|
294
403
|
try {
|
|
295
404
|
const out = await search({ dir: kbDir, query: question.query, k, repos: [question.store] });
|
|
296
405
|
// searchAll reports a store that could not be OPENED as an "ERR: ..." string in perRepo and
|
|
@@ -346,6 +455,8 @@ export async function runRepoRecall({
|
|
|
346
455
|
questionCount: fixture.questions.length,
|
|
347
456
|
shape: 'exactly one human-written question per repository',
|
|
348
457
|
},
|
|
458
|
+
// Present ONLY when something retired, so a report with none is the pre-D7 shape byte for byte.
|
|
459
|
+
...(retirement ? { retirement } : {}),
|
|
349
460
|
protocol: { entryPoint, k, repositoryScope: 'explicit', scoring: 'exact labeled file path within top-k of results from the requested repository' },
|
|
350
461
|
floor: { value: floorValue, committed: floor?.hitTop5Floor ?? null, absolute: ABSOLUTE_FLOOR, acceptedForRelease: floor?.acceptedForRelease ?? null },
|
|
351
462
|
totals,
|
|
@@ -370,7 +481,7 @@ const arg = (argv, name, fallback = null) => {
|
|
|
370
481
|
* Measure a SEALED archive: extract it, find the store root, and grade that — never the build
|
|
371
482
|
* directory the archive was assembled from. Returns the report bound to the archive's own digest.
|
|
372
483
|
*/
|
|
373
|
-
export async function runRepoRecallOnBundle({ bundleFile, fixtureFile, floorFile, k = DEFAULT_K } = {}) {
|
|
484
|
+
export async function runRepoRecallOnBundle({ bundleFile, fixtureFile, floorFile, k = DEFAULT_K, coverageFile = null } = {}) {
|
|
374
485
|
const bundle = path.resolve(bundleFile || '');
|
|
375
486
|
if (!bundle || !fs.existsSync(bundle) || !fs.statSync(bundle).isFile()) {
|
|
376
487
|
fail(`archive missing (${bundle || 'no path supplied'})`);
|
|
@@ -389,17 +500,19 @@ export async function runRepoRecallOnBundle({ bundleFile, fixtureFile, floorFile
|
|
|
389
500
|
};
|
|
390
501
|
walk(tmp);
|
|
391
502
|
if (roots.length !== 1) fail(`expected exactly one ARCHIVE-MANIFEST.json in the archive, found ${roots.length}`);
|
|
392
|
-
return await runRepoRecall({ kbDir: roots[0], fixtureFile, floorFile, archive, k });
|
|
503
|
+
return await runRepoRecall({ kbDir: roots[0], fixtureFile, floorFile, archive, k, coverageFile });
|
|
393
504
|
} finally {
|
|
394
505
|
fs.rmSync(tmp, { recursive: true, force: true });
|
|
395
506
|
}
|
|
396
507
|
}
|
|
397
508
|
|
|
398
|
-
|
|
509
|
+
// `searchAll` is a test seam only (the CLI never passes it): it lets the exit-code contract be proven
|
|
510
|
+
// without an embedded corpus. Unset, --kb grades the archive's own shipped entry point as always.
|
|
511
|
+
export async function main(argv = process.argv.slice(2), { searchAll = null } = {}) {
|
|
399
512
|
const kbDir = arg(argv, '--kb');
|
|
400
513
|
const bundleFile = arg(argv, '--bundle');
|
|
401
514
|
if (!kbDir && !bundleFile) {
|
|
402
|
-
process.stderr.write('usage: repo-recall.mjs (--bundle <archive.zip> | --kb <extracted root>) [--out <report.json>] [--fixture <file>] [--floor <file>]\n');
|
|
515
|
+
process.stderr.write('usage: repo-recall.mjs (--bundle <archive.zip> | --kb <extracted root>) [--out <report.json>] [--fixture <file>] [--floor <file>] [--coverage <sealed coverage>]\n');
|
|
403
516
|
return 64;
|
|
404
517
|
}
|
|
405
518
|
const { report, gate } = bundleFile
|
|
@@ -407,11 +520,14 @@ export async function main(argv = process.argv.slice(2)) {
|
|
|
407
520
|
bundleFile: path.resolve(bundleFile),
|
|
408
521
|
fixtureFile: arg(argv, '--fixture'),
|
|
409
522
|
floorFile: arg(argv, '--floor'),
|
|
523
|
+
coverageFile: arg(argv, '--coverage'),
|
|
410
524
|
})
|
|
411
525
|
: await runRepoRecall({
|
|
412
526
|
kbDir: path.resolve(kbDir),
|
|
527
|
+
searchAll,
|
|
413
528
|
fixtureFile: arg(argv, '--fixture'),
|
|
414
529
|
floorFile: arg(argv, '--floor'),
|
|
530
|
+
coverageFile: arg(argv, '--coverage'),
|
|
415
531
|
});
|
|
416
532
|
const out = arg(argv, '--out') || (bundleFile ? `${path.resolve(bundleFile)}.recall.json` : null);
|
|
417
533
|
if (out) fs.writeFileSync(path.resolve(out), `${JSON.stringify(report, null, 2)}\n`);
|
|
@@ -423,10 +539,15 @@ export async function main(argv = process.argv.slice(2)) {
|
|
|
423
539
|
repositoriesAnswering: `${t.repoCoverage}/${t.questions}`,
|
|
424
540
|
exactFileTop1: `${t.hitTop1}/${t.questions}`,
|
|
425
541
|
exactFileTop5: `${t.hitTop5}/${t.questions}`,
|
|
542
|
+
...(t.retired ? { retired: t.retired } : {}),
|
|
426
543
|
floor: report.floor.value,
|
|
427
544
|
failures: gate.failures,
|
|
545
|
+
enforced: blockingFailures(gate.failures),
|
|
428
546
|
}, null, 2)}\n`);
|
|
429
|
-
|
|
547
|
+
// ADR-0091 D7.6: the exit code derives from the INTEGRITY failures only. A floor miss stays in the
|
|
548
|
+
// report (state/failures) and is never fatal -- exiting 1 on it would silently restore the blocking
|
|
549
|
+
// ratchet ADR-086's 2026-09-15 amendment removed, the moment anyone re-accepts the floor.
|
|
550
|
+
return blockingFailures(gate.failures).length ? 1 : 0;
|
|
430
551
|
}
|
|
431
552
|
|
|
432
553
|
// Realpath both sides: argv[1] is whatever the caller typed, while Node resolves import.meta.url
|
|
@@ -34,6 +34,14 @@
|
|
|
34
34
|
// complete. A bounded run can therefore be read, reported and compared, but it can never seal a
|
|
35
35
|
// publishable corpus receipt.
|
|
36
36
|
//
|
|
37
|
+
// QUESTION SAMPLING (ADR-0091 D2). `--sample-questions <n>` measures n oracle questions in total,
|
|
38
|
+
// chosen by selectQuestionSample(): stratified across partitions (one question from each of n
|
|
39
|
+
// partitions before any partition gets a second) and ordered by sha256(seed, id), so the same
|
|
40
|
+
// seed, oracle and n always select the same questions. It exists because the full diagnostic is
|
|
41
|
+
// ~1,164 queries at ~4.2 s each on a hosted runner (82 minutes), and C3 no longer blocks anything.
|
|
42
|
+
// A sampled report is bounded like any other bounded run: `coverage.complete: false`, the seed and
|
|
43
|
+
// the exact selected label ids recorded in `coverage.bounded.sample`. Omit the flag for the full audit.
|
|
44
|
+
//
|
|
37
45
|
// TIMEOUTS. The threshold text says errors and timeouts count as failures (they are never excluded
|
|
38
46
|
// from the denominator), and the Step 15 proof text additionally names "timeout" as a standalone
|
|
39
47
|
// blocker. Both readings are honoured, strictly: a timeout is counted as a failure in the
|
|
@@ -74,6 +82,12 @@ export const THRESHOLD_DENOMINATOR = 20;
|
|
|
74
82
|
export const QUERY_MODES = Object.freeze(['explicit-repository', 'full-corpus']);
|
|
75
83
|
export const DEFAULT_QUERY_TIMEOUT_MS = 120_000;
|
|
76
84
|
export const DEFAULT_ORACLE_FILE = 'data/retrieval-accuracy-oracle.json';
|
|
85
|
+
// ADR-0091 D2 — the sample the corpus pipeline measures. 80 questions x 2 modes = 160 queries. At the
|
|
86
|
+
// hosted-runner cost of 4.23 s/query (82 min / 1,164 queries, the ADR's measured C3 run) that is
|
|
87
|
+
// 677 s of querying, ~11.3 min, leaving ~3.7 min of the 15-minute bound for extraction and model
|
|
88
|
+
// load. corpus-seed.yml passes this same number; a unit test pins the two together.
|
|
89
|
+
export const C3_DIAGNOSTIC_SAMPLE_QUESTIONS = 80;
|
|
90
|
+
export const DEFAULT_SAMPLE_SEED = 'c3-diagnostic-sample/1';
|
|
77
91
|
|
|
78
92
|
const HEX64 = /^[a-f0-9]{64}$/;
|
|
79
93
|
const HEX40 = /^[a-f0-9]{40}$/;
|
|
@@ -357,6 +371,37 @@ export function archiveStores(root) {
|
|
|
357
371
|
.sort();
|
|
358
372
|
}
|
|
359
373
|
|
|
374
|
+
/**
|
|
375
|
+
* Deterministic, stratified question sample (ADR-0091 D2). Partitions are ranked by
|
|
376
|
+
* sha256(seed, partition) and each partition's labels by sha256(seed, label id); the sample takes one
|
|
377
|
+
* label from every partition in rank order, then a second from every partition that has one, and so
|
|
378
|
+
* on until `size` labels are chosen. Same seed + same oracle + same size = same labels, always.
|
|
379
|
+
* Returns the chosen labels sorted by id.
|
|
380
|
+
*/
|
|
381
|
+
export function selectQuestionSample({ labels, size, seed = DEFAULT_SAMPLE_SEED }) {
|
|
382
|
+
if (!Number.isSafeInteger(size) || size <= 0) fail('question sample size must be a positive integer');
|
|
383
|
+
if (typeof seed !== 'string' || !seed) fail('question sample seed must be a non-empty string');
|
|
384
|
+
const rank = (value) => sha256Of(`${seed}\u0000${value}`);
|
|
385
|
+
const byPartition = new Map();
|
|
386
|
+
for (const label of labels) {
|
|
387
|
+
if (!byPartition.has(label.partition)) byPartition.set(label.partition, []);
|
|
388
|
+
byPartition.get(label.partition).push({ label, key: rank(`label\u0000${label.id}`) });
|
|
389
|
+
}
|
|
390
|
+
const queues = [...byPartition.entries()]
|
|
391
|
+
.map(([partition, rows]) => ({ key: rank(`partition\u0000${partition}`), rows: rows.sort((a, b) => a.key.localeCompare(b.key)) }))
|
|
392
|
+
.sort((a, b) => a.key.localeCompare(b.key));
|
|
393
|
+
const chosen = [];
|
|
394
|
+
for (let depth = 0; chosen.length < size; depth += 1) {
|
|
395
|
+
let took = false;
|
|
396
|
+
for (const queue of queues) {
|
|
397
|
+
if (chosen.length >= size) break;
|
|
398
|
+
if (depth < queue.rows.length) { chosen.push(queue.rows[depth].label); took = true; }
|
|
399
|
+
}
|
|
400
|
+
if (!took) break; // the oracle has fewer labels than requested: the sample is every label
|
|
401
|
+
}
|
|
402
|
+
return chosen.sort((a, b) => a.id.localeCompare(b.id));
|
|
403
|
+
}
|
|
404
|
+
|
|
360
405
|
async function defaultSearch({ dir, query, repos, timeoutMs }) {
|
|
361
406
|
const module = await import('../../kb/forge-ask-all.mjs');
|
|
362
407
|
let timer = null;
|
|
@@ -393,6 +438,8 @@ export async function runRetrievalAccuracy({
|
|
|
393
438
|
outFile,
|
|
394
439
|
storeLimit = null,
|
|
395
440
|
sampleLimit = null,
|
|
441
|
+
sampleQuestions = null,
|
|
442
|
+
sampleSeed = DEFAULT_SAMPLE_SEED,
|
|
396
443
|
modes = QUERY_MODES,
|
|
397
444
|
timeoutMs = DEFAULT_QUERY_TIMEOUT_MS,
|
|
398
445
|
search = defaultSearch,
|
|
@@ -407,6 +454,14 @@ export async function runRetrievalAccuracy({
|
|
|
407
454
|
const oracle = readAccuracyOracle(oracleFile);
|
|
408
455
|
const selectedModes = QUERY_MODES.filter((mode) => modes.includes(mode));
|
|
409
456
|
if (!selectedModes.length) fail('no supported query mode selected');
|
|
457
|
+
if (sampleQuestions != null && (storeLimit != null || sampleLimit != null)) {
|
|
458
|
+
fail('--sample-questions is a whole-oracle sample; it cannot be combined with --stores or --sample');
|
|
459
|
+
}
|
|
460
|
+
const questionSample = sampleQuestions == null ? null
|
|
461
|
+
: selectQuestionSample({ labels: oracle.labels, size: sampleQuestions, seed: sampleSeed });
|
|
462
|
+
const sampledIds = questionSample ? new Set(questionSample.map((label) => label.id)) : null;
|
|
463
|
+
// Any sampling at all measures only what it sampled: unproduced slots are not charged and n is not N.
|
|
464
|
+
const sampling = sampleLimit != null || questionSample != null;
|
|
410
465
|
|
|
411
466
|
const tmp = fs.mkdtempSync(path.join(os.tmpdir(), 'retrieval-accuracy-'));
|
|
412
467
|
try {
|
|
@@ -437,7 +492,9 @@ export async function runRetrievalAccuracy({
|
|
|
437
492
|
labelsByPartition.get(label.partition).push(label);
|
|
438
493
|
}
|
|
439
494
|
const orderedPartitions = [...oracle.partitions.values()].sort((a, b) => a.partition.localeCompare(b.partition));
|
|
440
|
-
const measuredPartitions =
|
|
495
|
+
const measuredPartitions = sampledIds
|
|
496
|
+
? orderedPartitions.filter((row) => (labelsByPartition.get(row.partition) || []).some((label) => sampledIds.has(label.id)))
|
|
497
|
+
: storeLimit == null ? orderedPartitions : orderedPartitions.slice(0, storeLimit);
|
|
441
498
|
|
|
442
499
|
const partitions = [];
|
|
443
500
|
let timeouts = 0;
|
|
@@ -446,12 +503,13 @@ export async function runRetrievalAccuracy({
|
|
|
446
503
|
const all = (labelsByPartition.get(partition.partition) || [])
|
|
447
504
|
.slice()
|
|
448
505
|
.sort((a, b) => a.id.localeCompare(b.id));
|
|
449
|
-
const selected =
|
|
506
|
+
const selected = sampledIds ? all.filter((label) => sampledIds.has(label.id))
|
|
507
|
+
: sampleLimit == null ? all : all.slice(0, sampleLimit);
|
|
450
508
|
// THE DENOMINATOR. For a compliant oracle N comes from the unit inventory — 2 x min(100, U) —
|
|
451
509
|
// never from how many labels happened to survive production. Every unproduced unit keeps its two
|
|
452
510
|
// slots and scores them as misses below. A bounded --sample run is incomplete and unacceptable
|
|
453
511
|
// regardless, so it measures only what it sampled.
|
|
454
|
-
const unproducedSlots = oracle.c3Eligible &&
|
|
512
|
+
const unproducedSlots = oracle.c3Eligible && !sampling ? partition.unproduced : [];
|
|
455
513
|
for (const mode of selectedModes) {
|
|
456
514
|
const row = {
|
|
457
515
|
partition: partition.partition,
|
|
@@ -466,7 +524,7 @@ export async function runRetrievalAccuracy({
|
|
|
466
524
|
failures: 0,
|
|
467
525
|
errors: 0,
|
|
468
526
|
timeouts: 0,
|
|
469
|
-
sampled:
|
|
527
|
+
sampled: sampling && selected.length < all.length,
|
|
470
528
|
oracleRows: all.length,
|
|
471
529
|
failedLabels: [],
|
|
472
530
|
};
|
|
@@ -514,7 +572,7 @@ export async function runRetrievalAccuracy({
|
|
|
514
572
|
if (row.successes + row.failures !== row.n) {
|
|
515
573
|
fail(`internal: partition ${row.partition} (${mode}) scored ${row.successes + row.failures} outcomes for n=${row.n}`);
|
|
516
574
|
}
|
|
517
|
-
if (oracle.c3Eligible &&
|
|
575
|
+
if (oracle.c3Eligible && !sampling && row.n !== row.N) {
|
|
518
576
|
fail(`internal: partition ${row.partition} (${mode}) measured n=${row.n} but its inventory fixes N=${row.N}`);
|
|
519
577
|
}
|
|
520
578
|
row.state = meetsThreshold(row.successes, row.n) && row.timeouts === 0 ? 'PASS' : 'FAIL';
|
|
@@ -533,6 +591,7 @@ export async function runRetrievalAccuracy({
|
|
|
533
591
|
const boundedReasons = [];
|
|
534
592
|
if (storeLimit != null) boundedReasons.push(`--stores ${storeLimit}`);
|
|
535
593
|
if (sampleLimit != null) boundedReasons.push(`--sample ${sampleLimit}`);
|
|
594
|
+
if (questionSample) boundedReasons.push(`--sample-questions ${sampleQuestions} (seed ${sampleSeed}): ${questionSample.length} of ${oracle.labels.length} oracle questions`);
|
|
536
595
|
if (selectedModes.length !== QUERY_MODES.length) boundedReasons.push(`--modes ${selectedModes.join(',')}`);
|
|
537
596
|
if (unmeasuredPartitions.length) boundedReasons.push(`${unmeasuredPartitions.length} oracle partition(s) not measured`);
|
|
538
597
|
if (uncoveredArchiveStores.length) boundedReasons.push(`${uncoveredArchiveStores.length} shipped store(s) with no oracle coverage`);
|
|
@@ -568,7 +627,14 @@ export async function runRetrievalAccuracy({
|
|
|
568
627
|
queryTimeoutMs: timeoutMs,
|
|
569
628
|
coverage: {
|
|
570
629
|
complete,
|
|
571
|
-
bounded: complete ? null : {
|
|
630
|
+
bounded: complete ? null : {
|
|
631
|
+
reasons: boundedReasons, storeLimit, sampleLimit, modes: selectedModes,
|
|
632
|
+
...(questionSample ? { sample: {
|
|
633
|
+
method: 'stratified-by-partition/sha256-rank', seed: sampleSeed, requested: sampleQuestions,
|
|
634
|
+
questions: questionSample.length, oracleQuestions: oracle.labels.length,
|
|
635
|
+
labelIds: questionSample.map((label) => label.id),
|
|
636
|
+
} } : {}),
|
|
637
|
+
},
|
|
572
638
|
archiveStores: shipped,
|
|
573
639
|
oraclePartitions: orderedPartitions.length,
|
|
574
640
|
measuredPartitions: measuredPartitions.length,
|
|
@@ -778,6 +844,8 @@ export async function main(argv = process.argv.slice(2)) {
|
|
|
778
844
|
outFile: arg(argv, '--out'),
|
|
779
845
|
storeLimit: positiveInt(arg(argv, '--stores'), '--stores'),
|
|
780
846
|
sampleLimit: positiveInt(arg(argv, '--sample'), '--sample'),
|
|
847
|
+
sampleQuestions: positiveInt(arg(argv, '--sample-questions'), '--sample-questions'),
|
|
848
|
+
sampleSeed: arg(argv, '--sample-seed', DEFAULT_SAMPLE_SEED),
|
|
781
849
|
modes: arg(argv, '--modes') ? String(arg(argv, '--modes')).split(',').map((mode) => mode.trim()) : QUERY_MODES,
|
|
782
850
|
timeoutMs: positiveInt(arg(argv, '--timeout-ms'), '--timeout-ms') || DEFAULT_QUERY_TIMEOUT_MS,
|
|
783
851
|
});
|
|
@@ -250,6 +250,19 @@ function retrospectiveBaselineFromTree({ extractedRoot, bundleFile, expectedTag,
|
|
|
250
250
|
return { receipt, bytes, fileSha256: crypto.createHash('sha256').update(bytes).digest('hex'), root, archiveManifest };
|
|
251
251
|
}
|
|
252
252
|
|
|
253
|
+
/**
|
|
254
|
+
* ADR-0091 D6.3: does the baseline archive carry the seed's own published tag? A code-release seed
|
|
255
|
+
* (vX.Y.Z) records that tag in its generation ledger. A corpus-generation seed CANNOT: its tag is
|
|
256
|
+
* corpus-sha256-<the archive's own digest>, which no file inside the archive can contain, and its
|
|
257
|
+
* ledger names the runtime that built it. So a content-addressed tag is proven by the archive digest
|
|
258
|
+
* it names; every other tag must equal the ledger's releaseTag, exactly as before.
|
|
259
|
+
*/
|
|
260
|
+
export function baselineTagMatches({ publishedTag, ledgerReleaseTag, archiveSha256 }) {
|
|
261
|
+
const contentAddressed = /^corpus-sha256-([0-9a-f]{64})$/.exec(String(publishedTag || ''));
|
|
262
|
+
if (contentAddressed) return contentAddressed[1] === archiveSha256;
|
|
263
|
+
return typeof publishedTag === 'string' && publishedTag.length > 0 && ledgerReleaseTag === publishedTag;
|
|
264
|
+
}
|
|
265
|
+
|
|
253
266
|
function observedBaselineFromTree({ extractedRoot, bundleFile, expectedTag, expectedSha256, expectedBytes }) {
|
|
254
267
|
const ledgerFile = findNamed(extractedRoot, 'RVF-GENERATIONS.json');
|
|
255
268
|
const root = path.dirname(ledgerFile);
|
|
@@ -259,7 +272,9 @@ function observedBaselineFromTree({ extractedRoot, bundleFile, expectedTag, expe
|
|
|
259
272
|
fail('historical baseline generation ledger is malformed');
|
|
260
273
|
}
|
|
261
274
|
const archive = namedIdentity(bundleFile);
|
|
262
|
-
if (
|
|
275
|
+
if (!baselineTagMatches({ publishedTag: expectedTag, ledgerReleaseTag: ledger.releaseTag, archiveSha256: archive.sha256 })) {
|
|
276
|
+
fail('historical baseline differs from the expected public release tag');
|
|
277
|
+
}
|
|
263
278
|
if (!HEX64.test(String(expectedSha256 || '')) || archive.sha256 !== expectedSha256) {
|
|
264
279
|
fail('historical baseline differs from the expected public archive SHA-256');
|
|
265
280
|
}
|
|
@@ -476,7 +491,8 @@ export async function createPublicVerificationInputs({ baselineBundle, candidate
|
|
|
476
491
|
fail('baseline archive bytes differ from release coverage');
|
|
477
492
|
}
|
|
478
493
|
if (seed.receiptSha256 !== baselineProof.fileSha256) fail('baseline receipt differs from release coverage');
|
|
479
|
-
if (seed.tag
|
|
494
|
+
if (!baselineTagMatches({ publishedTag: seed.tag, ledgerReleaseTag: baselineProof.receipt.releaseTag,
|
|
495
|
+
archiveSha256: baselineProof.receipt.archive.sha256 })) {
|
|
480
496
|
fail('baseline release tag differs from release coverage');
|
|
481
497
|
}
|
|
482
498
|
const baselineStores = baselineProof.receipt.stores.map(({ name }) => name);
|