ruvnet-brain 4.3.34 → 4.3.35

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -34,6 +34,14 @@
34
34
  // complete. A bounded run can therefore be read, reported and compared, but it can never seal a
35
35
  // publishable corpus receipt.
36
36
  //
37
+ // QUESTION SAMPLING (ADR-0091 D2). `--sample-questions <n>` measures n oracle questions in total,
38
+ // chosen by selectQuestionSample(): stratified across partitions (one question from each of n
39
+ // partitions before any partition gets a second) and ordered by sha256(seed, id), so the same
40
+ // seed, oracle and n always select the same questions. It exists because the full diagnostic is
41
+ // ~1,164 queries at ~4.2 s each on a hosted runner (82 minutes), and C3 no longer blocks anything.
42
+ // A sampled report is bounded like any other bounded run: `coverage.complete: false`, the seed and
43
+ // the exact selected label ids recorded in `coverage.bounded.sample`. Omit the flag for the full audit.
44
+ //
37
45
  // TIMEOUTS. The threshold text says errors and timeouts count as failures (they are never excluded
38
46
  // from the denominator), and the Step 15 proof text additionally names "timeout" as a standalone
39
47
  // blocker. Both readings are honoured, strictly: a timeout is counted as a failure in the
@@ -74,6 +82,12 @@ export const THRESHOLD_DENOMINATOR = 20;
74
82
  export const QUERY_MODES = Object.freeze(['explicit-repository', 'full-corpus']);
75
83
  export const DEFAULT_QUERY_TIMEOUT_MS = 120_000;
76
84
  export const DEFAULT_ORACLE_FILE = 'data/retrieval-accuracy-oracle.json';
85
+ // ADR-0091 D2 — the sample the corpus pipeline measures. 80 questions x 2 modes = 160 queries. At the
86
+ // hosted-runner cost of 4.23 s/query (82 min / 1,164 queries, the ADR's measured C3 run) that is
87
+ // 677 s of querying, ~11.3 min, leaving ~3.7 min of the 15-minute bound for extraction and model
88
+ // load. corpus-seed.yml passes this same number; a unit test pins the two together.
89
+ export const C3_DIAGNOSTIC_SAMPLE_QUESTIONS = 80;
90
+ export const DEFAULT_SAMPLE_SEED = 'c3-diagnostic-sample/1';
77
91
 
78
92
  const HEX64 = /^[a-f0-9]{64}$/;
79
93
  const HEX40 = /^[a-f0-9]{40}$/;
@@ -357,6 +371,37 @@ export function archiveStores(root) {
357
371
  .sort();
358
372
  }
359
373
 
374
+ /**
375
+ * Deterministic, stratified question sample (ADR-0091 D2). Partitions are ranked by
376
+ * sha256(seed, partition) and each partition's labels by sha256(seed, label id); the sample takes one
377
+ * label from every partition in rank order, then a second from every partition that has one, and so
378
+ * on until `size` labels are chosen. Same seed + same oracle + same size = same labels, always.
379
+ * Returns the chosen labels sorted by id.
380
+ */
381
+ export function selectQuestionSample({ labels, size, seed = DEFAULT_SAMPLE_SEED }) {
382
+ if (!Number.isSafeInteger(size) || size <= 0) fail('question sample size must be a positive integer');
383
+ if (typeof seed !== 'string' || !seed) fail('question sample seed must be a non-empty string');
384
+ const rank = (value) => sha256Of(`${seed}\u0000${value}`);
385
+ const byPartition = new Map();
386
+ for (const label of labels) {
387
+ if (!byPartition.has(label.partition)) byPartition.set(label.partition, []);
388
+ byPartition.get(label.partition).push({ label, key: rank(`label\u0000${label.id}`) });
389
+ }
390
+ const queues = [...byPartition.entries()]
391
+ .map(([partition, rows]) => ({ key: rank(`partition\u0000${partition}`), rows: rows.sort((a, b) => a.key.localeCompare(b.key)) }))
392
+ .sort((a, b) => a.key.localeCompare(b.key));
393
+ const chosen = [];
394
+ for (let depth = 0; chosen.length < size; depth += 1) {
395
+ let took = false;
396
+ for (const queue of queues) {
397
+ if (chosen.length >= size) break;
398
+ if (depth < queue.rows.length) { chosen.push(queue.rows[depth].label); took = true; }
399
+ }
400
+ if (!took) break; // the oracle has fewer labels than requested: the sample is every label
401
+ }
402
+ return chosen.sort((a, b) => a.id.localeCompare(b.id));
403
+ }
404
+
360
405
  async function defaultSearch({ dir, query, repos, timeoutMs }) {
361
406
  const module = await import('../../kb/forge-ask-all.mjs');
362
407
  let timer = null;
@@ -393,6 +438,8 @@ export async function runRetrievalAccuracy({
393
438
  outFile,
394
439
  storeLimit = null,
395
440
  sampleLimit = null,
441
+ sampleQuestions = null,
442
+ sampleSeed = DEFAULT_SAMPLE_SEED,
396
443
  modes = QUERY_MODES,
397
444
  timeoutMs = DEFAULT_QUERY_TIMEOUT_MS,
398
445
  search = defaultSearch,
@@ -407,6 +454,14 @@ export async function runRetrievalAccuracy({
407
454
  const oracle = readAccuracyOracle(oracleFile);
408
455
  const selectedModes = QUERY_MODES.filter((mode) => modes.includes(mode));
409
456
  if (!selectedModes.length) fail('no supported query mode selected');
457
+ if (sampleQuestions != null && (storeLimit != null || sampleLimit != null)) {
458
+ fail('--sample-questions is a whole-oracle sample; it cannot be combined with --stores or --sample');
459
+ }
460
+ const questionSample = sampleQuestions == null ? null
461
+ : selectQuestionSample({ labels: oracle.labels, size: sampleQuestions, seed: sampleSeed });
462
+ const sampledIds = questionSample ? new Set(questionSample.map((label) => label.id)) : null;
463
+ // Any sampling at all measures only what it sampled: unproduced slots are not charged and n is not N.
464
+ const sampling = sampleLimit != null || questionSample != null;
410
465
 
411
466
  const tmp = fs.mkdtempSync(path.join(os.tmpdir(), 'retrieval-accuracy-'));
412
467
  try {
@@ -437,7 +492,9 @@ export async function runRetrievalAccuracy({
437
492
  labelsByPartition.get(label.partition).push(label);
438
493
  }
439
494
  const orderedPartitions = [...oracle.partitions.values()].sort((a, b) => a.partition.localeCompare(b.partition));
440
- const measuredPartitions = storeLimit == null ? orderedPartitions : orderedPartitions.slice(0, storeLimit);
495
+ const measuredPartitions = sampledIds
496
+ ? orderedPartitions.filter((row) => (labelsByPartition.get(row.partition) || []).some((label) => sampledIds.has(label.id)))
497
+ : storeLimit == null ? orderedPartitions : orderedPartitions.slice(0, storeLimit);
441
498
 
442
499
  const partitions = [];
443
500
  let timeouts = 0;
@@ -446,12 +503,13 @@ export async function runRetrievalAccuracy({
446
503
  const all = (labelsByPartition.get(partition.partition) || [])
447
504
  .slice()
448
505
  .sort((a, b) => a.id.localeCompare(b.id));
449
- const selected = sampleLimit == null ? all : all.slice(0, sampleLimit);
506
+ const selected = sampledIds ? all.filter((label) => sampledIds.has(label.id))
507
+ : sampleLimit == null ? all : all.slice(0, sampleLimit);
450
508
  // THE DENOMINATOR. For a compliant oracle N comes from the unit inventory — 2 x min(100, U) —
451
509
  // never from how many labels happened to survive production. Every unproduced unit keeps its two
452
510
  // slots and scores them as misses below. A bounded --sample run is incomplete and unacceptable
453
511
  // regardless, so it measures only what it sampled.
454
- const unproducedSlots = oracle.c3Eligible && sampleLimit == null ? partition.unproduced : [];
512
+ const unproducedSlots = oracle.c3Eligible && !sampling ? partition.unproduced : [];
455
513
  for (const mode of selectedModes) {
456
514
  const row = {
457
515
  partition: partition.partition,
@@ -466,7 +524,7 @@ export async function runRetrievalAccuracy({
466
524
  failures: 0,
467
525
  errors: 0,
468
526
  timeouts: 0,
469
- sampled: sampleLimit != null && selected.length < all.length,
527
+ sampled: sampling && selected.length < all.length,
470
528
  oracleRows: all.length,
471
529
  failedLabels: [],
472
530
  };
@@ -514,7 +572,7 @@ export async function runRetrievalAccuracy({
514
572
  if (row.successes + row.failures !== row.n) {
515
573
  fail(`internal: partition ${row.partition} (${mode}) scored ${row.successes + row.failures} outcomes for n=${row.n}`);
516
574
  }
517
- if (oracle.c3Eligible && sampleLimit == null && row.n !== row.N) {
575
+ if (oracle.c3Eligible && !sampling && row.n !== row.N) {
518
576
  fail(`internal: partition ${row.partition} (${mode}) measured n=${row.n} but its inventory fixes N=${row.N}`);
519
577
  }
520
578
  row.state = meetsThreshold(row.successes, row.n) && row.timeouts === 0 ? 'PASS' : 'FAIL';
@@ -533,6 +591,7 @@ export async function runRetrievalAccuracy({
533
591
  const boundedReasons = [];
534
592
  if (storeLimit != null) boundedReasons.push(`--stores ${storeLimit}`);
535
593
  if (sampleLimit != null) boundedReasons.push(`--sample ${sampleLimit}`);
594
+ if (questionSample) boundedReasons.push(`--sample-questions ${sampleQuestions} (seed ${sampleSeed}): ${questionSample.length} of ${oracle.labels.length} oracle questions`);
536
595
  if (selectedModes.length !== QUERY_MODES.length) boundedReasons.push(`--modes ${selectedModes.join(',')}`);
537
596
  if (unmeasuredPartitions.length) boundedReasons.push(`${unmeasuredPartitions.length} oracle partition(s) not measured`);
538
597
  if (uncoveredArchiveStores.length) boundedReasons.push(`${uncoveredArchiveStores.length} shipped store(s) with no oracle coverage`);
@@ -568,7 +627,14 @@ export async function runRetrievalAccuracy({
568
627
  queryTimeoutMs: timeoutMs,
569
628
  coverage: {
570
629
  complete,
571
- bounded: complete ? null : { reasons: boundedReasons, storeLimit, sampleLimit, modes: selectedModes },
630
+ bounded: complete ? null : {
631
+ reasons: boundedReasons, storeLimit, sampleLimit, modes: selectedModes,
632
+ ...(questionSample ? { sample: {
633
+ method: 'stratified-by-partition/sha256-rank', seed: sampleSeed, requested: sampleQuestions,
634
+ questions: questionSample.length, oracleQuestions: oracle.labels.length,
635
+ labelIds: questionSample.map((label) => label.id),
636
+ } } : {}),
637
+ },
572
638
  archiveStores: shipped,
573
639
  oraclePartitions: orderedPartitions.length,
574
640
  measuredPartitions: measuredPartitions.length,
@@ -778,6 +844,8 @@ export async function main(argv = process.argv.slice(2)) {
778
844
  outFile: arg(argv, '--out'),
779
845
  storeLimit: positiveInt(arg(argv, '--stores'), '--stores'),
780
846
  sampleLimit: positiveInt(arg(argv, '--sample'), '--sample'),
847
+ sampleQuestions: positiveInt(arg(argv, '--sample-questions'), '--sample-questions'),
848
+ sampleSeed: arg(argv, '--sample-seed', DEFAULT_SAMPLE_SEED),
781
849
  modes: arg(argv, '--modes') ? String(arg(argv, '--modes')).split(',').map((mode) => mode.trim()) : QUERY_MODES,
782
850
  timeoutMs: positiveInt(arg(argv, '--timeout-ms'), '--timeout-ms') || DEFAULT_QUERY_TIMEOUT_MS,
783
851
  });