ruvnet-brain 4.3.33 → 4.3.35
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/package.json +1 -1
- package/plugin/.claude-plugin/plugin.json +1 -1
- package/plugin/.codex-plugin/plugin.json +1 -1
- package/scripts/agentdb-fleet-doctor.mjs +7 -3
- package/scripts/approved-runtime.mjs +247 -26
- package/scripts/corpus-next-seed.mjs +191 -70
- package/scripts/corpus-reconcile.mjs +110 -16
- package/scripts/oracle/retrieval-accuracy.mjs +74 -6
- package/scripts/rehearse-corpus-pipeline.mjs +315 -25
- package/scripts/rehearse-seed-selection.mjs +234 -0
- package/scripts/release.mjs +13 -6
- package/scripts/single-source-check.mjs +15 -2
|
@@ -34,6 +34,14 @@
|
|
|
34
34
|
// complete. A bounded run can therefore be read, reported and compared, but it can never seal a
|
|
35
35
|
// publishable corpus receipt.
|
|
36
36
|
//
|
|
37
|
+
// QUESTION SAMPLING (ADR-0091 D2). `--sample-questions <n>` measures n oracle questions in total,
|
|
38
|
+
// chosen by selectQuestionSample(): stratified across partitions (one question from each of n
|
|
39
|
+
// partitions before any partition gets a second) and ordered by sha256(seed, id), so the same
|
|
40
|
+
// seed, oracle and n always select the same questions. It exists because the full diagnostic is
|
|
41
|
+
// ~1,164 queries at ~4.2 s each on a hosted runner (82 minutes), and C3 no longer blocks anything.
|
|
42
|
+
// A sampled report is bounded like any other bounded run: `coverage.complete: false`, the seed and
|
|
43
|
+
// the exact selected label ids recorded in `coverage.bounded.sample`. Omit the flag for the full audit.
|
|
44
|
+
//
|
|
37
45
|
// TIMEOUTS. The threshold text says errors and timeouts count as failures (they are never excluded
|
|
38
46
|
// from the denominator), and the Step 15 proof text additionally names "timeout" as a standalone
|
|
39
47
|
// blocker. Both readings are honoured, strictly: a timeout is counted as a failure in the
|
|
@@ -74,6 +82,12 @@ export const THRESHOLD_DENOMINATOR = 20;
|
|
|
74
82
|
export const QUERY_MODES = Object.freeze(['explicit-repository', 'full-corpus']);
|
|
75
83
|
export const DEFAULT_QUERY_TIMEOUT_MS = 120_000;
|
|
76
84
|
export const DEFAULT_ORACLE_FILE = 'data/retrieval-accuracy-oracle.json';
|
|
85
|
+
// ADR-0091 D2 — the sample the corpus pipeline measures. 80 questions x 2 modes = 160 queries. At the
|
|
86
|
+
// hosted-runner cost of 4.23 s/query (82 min / 1,164 queries, the ADR's measured C3 run) that is
|
|
87
|
+
// 677 s of querying, ~11.3 min, leaving ~3.7 min of the 15-minute bound for extraction and model
|
|
88
|
+
// load. corpus-seed.yml passes this same number; a unit test pins the two together.
|
|
89
|
+
export const C3_DIAGNOSTIC_SAMPLE_QUESTIONS = 80;
|
|
90
|
+
export const DEFAULT_SAMPLE_SEED = 'c3-diagnostic-sample/1';
|
|
77
91
|
|
|
78
92
|
const HEX64 = /^[a-f0-9]{64}$/;
|
|
79
93
|
const HEX40 = /^[a-f0-9]{40}$/;
|
|
@@ -357,6 +371,37 @@ export function archiveStores(root) {
|
|
|
357
371
|
.sort();
|
|
358
372
|
}
|
|
359
373
|
|
|
374
|
+
/**
|
|
375
|
+
* Deterministic, stratified question sample (ADR-0091 D2). Partitions are ranked by
|
|
376
|
+
* sha256(seed, partition) and each partition's labels by sha256(seed, label id); the sample takes one
|
|
377
|
+
* label from every partition in rank order, then a second from every partition that has one, and so
|
|
378
|
+
* on until `size` labels are chosen. Same seed + same oracle + same size = same labels, always.
|
|
379
|
+
* Returns the chosen labels sorted by id.
|
|
380
|
+
*/
|
|
381
|
+
export function selectQuestionSample({ labels, size, seed = DEFAULT_SAMPLE_SEED }) {
|
|
382
|
+
if (!Number.isSafeInteger(size) || size <= 0) fail('question sample size must be a positive integer');
|
|
383
|
+
if (typeof seed !== 'string' || !seed) fail('question sample seed must be a non-empty string');
|
|
384
|
+
const rank = (value) => sha256Of(`${seed}\u0000${value}`);
|
|
385
|
+
const byPartition = new Map();
|
|
386
|
+
for (const label of labels) {
|
|
387
|
+
if (!byPartition.has(label.partition)) byPartition.set(label.partition, []);
|
|
388
|
+
byPartition.get(label.partition).push({ label, key: rank(`label\u0000${label.id}`) });
|
|
389
|
+
}
|
|
390
|
+
const queues = [...byPartition.entries()]
|
|
391
|
+
.map(([partition, rows]) => ({ key: rank(`partition\u0000${partition}`), rows: rows.sort((a, b) => a.key.localeCompare(b.key)) }))
|
|
392
|
+
.sort((a, b) => a.key.localeCompare(b.key));
|
|
393
|
+
const chosen = [];
|
|
394
|
+
for (let depth = 0; chosen.length < size; depth += 1) {
|
|
395
|
+
let took = false;
|
|
396
|
+
for (const queue of queues) {
|
|
397
|
+
if (chosen.length >= size) break;
|
|
398
|
+
if (depth < queue.rows.length) { chosen.push(queue.rows[depth].label); took = true; }
|
|
399
|
+
}
|
|
400
|
+
if (!took) break; // the oracle has fewer labels than requested: the sample is every label
|
|
401
|
+
}
|
|
402
|
+
return chosen.sort((a, b) => a.id.localeCompare(b.id));
|
|
403
|
+
}
|
|
404
|
+
|
|
360
405
|
async function defaultSearch({ dir, query, repos, timeoutMs }) {
|
|
361
406
|
const module = await import('../../kb/forge-ask-all.mjs');
|
|
362
407
|
let timer = null;
|
|
@@ -393,6 +438,8 @@ export async function runRetrievalAccuracy({
|
|
|
393
438
|
outFile,
|
|
394
439
|
storeLimit = null,
|
|
395
440
|
sampleLimit = null,
|
|
441
|
+
sampleQuestions = null,
|
|
442
|
+
sampleSeed = DEFAULT_SAMPLE_SEED,
|
|
396
443
|
modes = QUERY_MODES,
|
|
397
444
|
timeoutMs = DEFAULT_QUERY_TIMEOUT_MS,
|
|
398
445
|
search = defaultSearch,
|
|
@@ -407,6 +454,14 @@ export async function runRetrievalAccuracy({
|
|
|
407
454
|
const oracle = readAccuracyOracle(oracleFile);
|
|
408
455
|
const selectedModes = QUERY_MODES.filter((mode) => modes.includes(mode));
|
|
409
456
|
if (!selectedModes.length) fail('no supported query mode selected');
|
|
457
|
+
if (sampleQuestions != null && (storeLimit != null || sampleLimit != null)) {
|
|
458
|
+
fail('--sample-questions is a whole-oracle sample; it cannot be combined with --stores or --sample');
|
|
459
|
+
}
|
|
460
|
+
const questionSample = sampleQuestions == null ? null
|
|
461
|
+
: selectQuestionSample({ labels: oracle.labels, size: sampleQuestions, seed: sampleSeed });
|
|
462
|
+
const sampledIds = questionSample ? new Set(questionSample.map((label) => label.id)) : null;
|
|
463
|
+
// Any sampling at all measures only what it sampled: unproduced slots are not charged and n is not N.
|
|
464
|
+
const sampling = sampleLimit != null || questionSample != null;
|
|
410
465
|
|
|
411
466
|
const tmp = fs.mkdtempSync(path.join(os.tmpdir(), 'retrieval-accuracy-'));
|
|
412
467
|
try {
|
|
@@ -437,7 +492,9 @@ export async function runRetrievalAccuracy({
|
|
|
437
492
|
labelsByPartition.get(label.partition).push(label);
|
|
438
493
|
}
|
|
439
494
|
const orderedPartitions = [...oracle.partitions.values()].sort((a, b) => a.partition.localeCompare(b.partition));
|
|
440
|
-
const measuredPartitions =
|
|
495
|
+
const measuredPartitions = sampledIds
|
|
496
|
+
? orderedPartitions.filter((row) => (labelsByPartition.get(row.partition) || []).some((label) => sampledIds.has(label.id)))
|
|
497
|
+
: storeLimit == null ? orderedPartitions : orderedPartitions.slice(0, storeLimit);
|
|
441
498
|
|
|
442
499
|
const partitions = [];
|
|
443
500
|
let timeouts = 0;
|
|
@@ -446,12 +503,13 @@ export async function runRetrievalAccuracy({
|
|
|
446
503
|
const all = (labelsByPartition.get(partition.partition) || [])
|
|
447
504
|
.slice()
|
|
448
505
|
.sort((a, b) => a.id.localeCompare(b.id));
|
|
449
|
-
const selected =
|
|
506
|
+
const selected = sampledIds ? all.filter((label) => sampledIds.has(label.id))
|
|
507
|
+
: sampleLimit == null ? all : all.slice(0, sampleLimit);
|
|
450
508
|
// THE DENOMINATOR. For a compliant oracle N comes from the unit inventory — 2 x min(100, U) —
|
|
451
509
|
// never from how many labels happened to survive production. Every unproduced unit keeps its two
|
|
452
510
|
// slots and scores them as misses below. A bounded --sample run is incomplete and unacceptable
|
|
453
511
|
// regardless, so it measures only what it sampled.
|
|
454
|
-
const unproducedSlots = oracle.c3Eligible &&
|
|
512
|
+
const unproducedSlots = oracle.c3Eligible && !sampling ? partition.unproduced : [];
|
|
455
513
|
for (const mode of selectedModes) {
|
|
456
514
|
const row = {
|
|
457
515
|
partition: partition.partition,
|
|
@@ -466,7 +524,7 @@ export async function runRetrievalAccuracy({
|
|
|
466
524
|
failures: 0,
|
|
467
525
|
errors: 0,
|
|
468
526
|
timeouts: 0,
|
|
469
|
-
sampled:
|
|
527
|
+
sampled: sampling && selected.length < all.length,
|
|
470
528
|
oracleRows: all.length,
|
|
471
529
|
failedLabels: [],
|
|
472
530
|
};
|
|
@@ -514,7 +572,7 @@ export async function runRetrievalAccuracy({
|
|
|
514
572
|
if (row.successes + row.failures !== row.n) {
|
|
515
573
|
fail(`internal: partition ${row.partition} (${mode}) scored ${row.successes + row.failures} outcomes for n=${row.n}`);
|
|
516
574
|
}
|
|
517
|
-
if (oracle.c3Eligible &&
|
|
575
|
+
if (oracle.c3Eligible && !sampling && row.n !== row.N) {
|
|
518
576
|
fail(`internal: partition ${row.partition} (${mode}) measured n=${row.n} but its inventory fixes N=${row.N}`);
|
|
519
577
|
}
|
|
520
578
|
row.state = meetsThreshold(row.successes, row.n) && row.timeouts === 0 ? 'PASS' : 'FAIL';
|
|
@@ -533,6 +591,7 @@ export async function runRetrievalAccuracy({
|
|
|
533
591
|
const boundedReasons = [];
|
|
534
592
|
if (storeLimit != null) boundedReasons.push(`--stores ${storeLimit}`);
|
|
535
593
|
if (sampleLimit != null) boundedReasons.push(`--sample ${sampleLimit}`);
|
|
594
|
+
if (questionSample) boundedReasons.push(`--sample-questions ${sampleQuestions} (seed ${sampleSeed}): ${questionSample.length} of ${oracle.labels.length} oracle questions`);
|
|
536
595
|
if (selectedModes.length !== QUERY_MODES.length) boundedReasons.push(`--modes ${selectedModes.join(',')}`);
|
|
537
596
|
if (unmeasuredPartitions.length) boundedReasons.push(`${unmeasuredPartitions.length} oracle partition(s) not measured`);
|
|
538
597
|
if (uncoveredArchiveStores.length) boundedReasons.push(`${uncoveredArchiveStores.length} shipped store(s) with no oracle coverage`);
|
|
@@ -568,7 +627,14 @@ export async function runRetrievalAccuracy({
|
|
|
568
627
|
queryTimeoutMs: timeoutMs,
|
|
569
628
|
coverage: {
|
|
570
629
|
complete,
|
|
571
|
-
bounded: complete ? null : {
|
|
630
|
+
bounded: complete ? null : {
|
|
631
|
+
reasons: boundedReasons, storeLimit, sampleLimit, modes: selectedModes,
|
|
632
|
+
...(questionSample ? { sample: {
|
|
633
|
+
method: 'stratified-by-partition/sha256-rank', seed: sampleSeed, requested: sampleQuestions,
|
|
634
|
+
questions: questionSample.length, oracleQuestions: oracle.labels.length,
|
|
635
|
+
labelIds: questionSample.map((label) => label.id),
|
|
636
|
+
} } : {}),
|
|
637
|
+
},
|
|
572
638
|
archiveStores: shipped,
|
|
573
639
|
oraclePartitions: orderedPartitions.length,
|
|
574
640
|
measuredPartitions: measuredPartitions.length,
|
|
@@ -778,6 +844,8 @@ export async function main(argv = process.argv.slice(2)) {
|
|
|
778
844
|
outFile: arg(argv, '--out'),
|
|
779
845
|
storeLimit: positiveInt(arg(argv, '--stores'), '--stores'),
|
|
780
846
|
sampleLimit: positiveInt(arg(argv, '--sample'), '--sample'),
|
|
847
|
+
sampleQuestions: positiveInt(arg(argv, '--sample-questions'), '--sample-questions'),
|
|
848
|
+
sampleSeed: arg(argv, '--sample-seed', DEFAULT_SAMPLE_SEED),
|
|
781
849
|
modes: arg(argv, '--modes') ? String(arg(argv, '--modes')).split(',').map((mode) => mode.trim()) : QUERY_MODES,
|
|
782
850
|
timeoutMs: positiveInt(arg(argv, '--timeout-ms'), '--timeout-ms') || DEFAULT_QUERY_TIMEOUT_MS,
|
|
783
851
|
});
|