@orangepro/orangepro-mcp 0.2.43 → 0.2.45

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -13,7 +13,7 @@
13
13
  * No key ⇒ writes NO files, mints NO proof, returns explicit guidance.
14
14
  */
15
15
  import { existsSync, mkdirSync, readFileSync, writeFileSync } from "node:fs";
16
- import { basename, dirname, posix, resolve, sep } from "node:path";
16
+ import { basename, dirname, join, posix, relative, resolve, sep } from "node:path";
17
17
  import { generateTests, readDeclaredDeps, unresolvedLocalImports } from "./generate/generator.js";
18
18
  import { GENERATED_DIR, runHintsFor } from "./generate/runHints.js";
19
19
  import { rankRiskGaps } from "./score/risk.js";
@@ -21,12 +21,12 @@ import { resolveProviderConfig } from "./localConfig.js";
21
21
  import { buildProvider, DeterministicProvider } from "./generate/providers.js";
22
22
  import { resolveContained } from "./reprove/paths.js";
23
23
  import { buildRtm } from "./rtm.js";
24
- import { loadLedger } from "./ledger.js";
24
+ import { loadLedger, proofEdgesFor, targetLanguage } from "./ledger.js";
25
25
  import { reportProgress } from "./util/progress.js";
26
26
  import { loadGraph, workspacePaths } from "./workspace.js";
27
27
  import { systemClock } from "./util/time.js";
28
28
  import { redactSecrets } from "./util/redact.js";
29
- import { classifyBaselineFailure, EXPERIMENTAL_SQLITE_TEST_ENV, IMPORT_TIME_CATEGORIES, isNeedsSetupCategory, readEnginesNode, targetNeedsExperimentalSqlite } from "./proofRunnability.js";
29
+ import { classifyBaselineFailure, EXPERIMENTAL_SQLITE_TEST_ENV, isNeedsSetupCategory, readEnginesNode, targetNeedsExperimentalSqlite } from "./proofRunnability.js";
30
30
  /** Real dynamic proof is profile-gated; only wired runner targets are attemptable. */
31
31
  function isTsJsFile(file) {
32
32
  return /\.[cm]?[jt]sx?$/i.test(file);
@@ -82,7 +82,7 @@ function pytestNodeidsForFile(sourceRoot, testRel) {
82
82
  }
83
83
  if (currentClass && indent <= classIndent && line.trim() !== "" && !line.startsWith(" "))
84
84
  currentClass = null;
85
- const fnMatch = /^(\s*)def\s+(test_[A-Za-z0-9_]*)\s*\(/.exec(line);
85
+ const fnMatch = /^(\s*)(?:async\s+)?def\s+(test_[A-Za-z0-9_]*)\s*\(/.exec(line);
86
86
  if (!fnMatch)
87
87
  continue;
88
88
  const fnIndent = fnMatch[1].length;
@@ -358,13 +358,12 @@ export function javaTestForTarget(nodeById, graph, symId) {
358
358
  return selectors.size === 1 ? [...selectors][0] : null;
359
359
  }
360
360
  const GENERATED_HEADER = "// Generated by OrangePro — do not edit";
361
- // "Static map first, dynamically prove top 5": ONE unified dynamic-proof budget spans
362
- // BOTH lanes (existing-tests first, then generation for the remaining eligible gaps).
363
- // TOTAL attempts (existing + generation) are capped at this budget; `--auto-limit N`
364
- // overrides it (clamped to MAX_AUTO_LIMIT) for deeper runs (`opro start --auto-limit 25`,
365
- // `opro prove`, `opro prove-loop`). Static breadth (behaviors/flows/associated/risk) is
366
- // never gated on this budget — only the dynamic verification pass is.
361
+ // Released v0.2.44 generic dynamic-proof limit. Python's existing-test lane has its
362
+ // own retry limit below; do not widen this generic default or TS/JS, Go, Java, and
363
+ // generated-test scheduling changes before a caller opts in.
367
364
  const DEFAULT_AUTO_LIMIT = 5;
365
+ const DEFAULT_PROOF_EXISTING_LIMIT = 20;
366
+ const DEFAULT_PROOF_GEN_LIMIT = 5;
368
367
  const MAX_AUTO_LIMIT = 50;
369
368
  /** Generator caps a single call at 5 (MAX_LIMIT); page candidates in windows of that. */
370
369
  const GEN_WINDOW = 5;
@@ -372,7 +371,73 @@ const GEN_WINDOW = 5;
372
371
  // file (every eligible symbol × every importing test) cannot starve the shared budget before
373
372
  // a provable hard-edge symbol is tried. Small K; promote to a flag if a repo needs a wider sweep.
374
373
  const EXISTING_LANE_MAX_WEAK_PER_SYMBOL = 3;
374
+ const RUNNER_ROOT_MARKERS = {
375
+ typescript: ["package.json"],
376
+ javascript: ["package.json"],
377
+ python: ["pyproject.toml", "setup.cfg", "setup.py", "pytest.ini", ".pytest.ini", "tox.ini"],
378
+ go: ["go.mod"],
379
+ java: ["pom.xml", "build.gradle", "build.gradle.kts"],
380
+ rs: ["Cargo.toml"]
381
+ };
382
+ /**
383
+ * The nearest language runner root, bounded by the analyzed source root. Python is
384
+ * test-owned: pytest configuration/environment follows the selected test file even when
385
+ * that test exercises a target elsewhere in the same repository. Other languages keep
386
+ * their target-owned v0.2.44 behavior.
387
+ */
388
+ export function proofRunnerRoot(sourceRoot, targetRel, testRel) {
389
+ const stop = resolve(sourceRoot);
390
+ const targetAbs = resolve(stop, targetRel);
391
+ if (targetAbs !== stop && !targetAbs.startsWith(stop + sep))
392
+ return ".";
393
+ const language = targetLanguage(`sym:${targetRel}#target`);
394
+ const markers = RUNNER_ROOT_MARKERS[language] ?? [];
395
+ const testFile = testRel?.split("::", 1)[0] ?? testRel;
396
+ const testAbs = testFile ? resolve(stop, testFile) : undefined;
397
+ if (language === "python" && testAbs && testAbs !== stop && !testAbs.startsWith(stop + sep))
398
+ return ".";
399
+ let dir = dirname(language === "python" && testAbs ? testAbs : targetAbs);
400
+ let found = stop;
401
+ for (;;) {
402
+ if (markers.some((marker) => existsSync(join(dir, marker)))) {
403
+ found = dir;
404
+ break;
405
+ }
406
+ if (dir === stop)
407
+ break;
408
+ const parent = dirname(dir);
409
+ if (parent === dir || (parent !== stop && !parent.startsWith(stop + sep)))
410
+ break;
411
+ dir = parent;
412
+ }
413
+ const rel = relative(stop, found).split(sep).join("/");
414
+ return rel || ".";
415
+ }
416
+ /** Dominant-language order from all denominator-eligible code behaviors. */
417
+ export function proofLanguageOrder(graph) {
418
+ const counts = new Map();
419
+ const first = new Map();
420
+ for (const node of graph.nodes) {
421
+ if (node.kind !== "CodeSymbol" || node.denominator_eligible !== true)
422
+ continue;
423
+ const language = targetLanguage(node.external_id);
424
+ if (!first.has(language))
425
+ first.set(language, first.size);
426
+ counts.set(language, (counts.get(language) ?? 0) + 1);
427
+ }
428
+ return [...counts.keys()].sort((a, b) => (counts.get(b) - counts.get(a)) || (first.get(a) - first.get(b)));
429
+ }
375
430
  export const NO_KEY_MESSAGE = "No provider key; auto-prove skipped — add OPENAI_API_KEY / ANTHROPIC_API_KEY, or use the OrangePro MCP in your coding agent.";
431
+ /** Only failures proven to apply across a runner project may quarantine its siblings. */
432
+ const PROJECT_WIDE_BLOCKERS = new Set([
433
+ "engine_mismatch",
434
+ "tsconfig_missing",
435
+ "runner_missing",
436
+ "module_root_missing",
437
+ "go_package_build_failure",
438
+ "environment_unavailable",
439
+ "collection_error"
440
+ ]);
376
441
  export function isRoastSurvivor(attempt) {
377
442
  return attempt.classification === "non_killing" && attempt.mutant_status === "associated_survived";
378
443
  }
@@ -434,16 +499,38 @@ function fileReaderFor(root) {
434
499
  function classifyProof(result, ctx) {
435
500
  if ("status" in result && result.status === "unrunnable") {
436
501
  // Setup did not run (env non-event) — nothing was minted, target needs setup.
437
- return { classification: "needs_setup", reason: result.reason };
502
+ const reason = result.reason;
503
+ const category = /runner binary not found|unsupported or unknown test runner/i.test(reason)
504
+ ? "runner_missing"
505
+ : /no go\.mod found|no pom\.xml|no maven or gradle|build\.gradle|no python project root/i.test(reason)
506
+ ? "module_root_missing"
507
+ : undefined;
508
+ return { classification: "needs_setup", reason, category };
438
509
  }
439
510
  const dyn = result;
440
511
  const record = dyn.record;
441
512
  if (record.closed)
442
513
  return { classification: "proven" };
514
+ // Language spikes may refuse before a baseline exists (for example, Python's
515
+ // collection preflight, an explicit project boundary, or a type-unsafe mutation).
516
+ // These are structured runner outcomes, not a raw-stderr heuristic. Never mint a
517
+ // proof and never mislabel a mutation refusal as a surviving test.
518
+ const runnerCategory = dyn.oracle.category;
519
+ if (runnerCategory === "environment_unavailable" || runnerCategory === "collection_error" || runnerCategory === "project_boundary" || runnerCategory === "mutation_unsupported") {
520
+ return { classification: "needs_setup", reason: dyn.oracle.reason, category: runnerCategory };
521
+ }
443
522
  const cert = record.dynamic_proof;
444
523
  if (cert && cert.baseline_green === false) {
524
+ const failureSummary = dyn.oracle.baseline?.failureSummary;
525
+ if (ctx.targetFileRel.endsWith(".go") && /^#\s+\S+/m.test(failureSummary ?? "")) {
526
+ return {
527
+ classification: "needs_setup",
528
+ reason: "The Go package did not compile before the selected test could run.",
529
+ category: "go_package_build_failure"
530
+ };
531
+ }
445
532
  const { category, reason } = classifyBaselineFailure({
446
- failureSummary: dyn.oracle.baseline?.failureSummary,
533
+ failureSummary,
447
534
  enginesNode: readEnginesNode(ctx.sourceRoot, ctx.targetFileRel),
448
535
  runnerNode: ctx.runnerNode ?? process.version
449
536
  });
@@ -457,27 +544,53 @@ function mutantStatusOf(result) {
457
544
  return "unrunnable";
458
545
  return result.record.dynamic_proof?.mutant_status;
459
546
  }
460
- /**
461
- * R-1 sibling-dedup key: a baseline-red import-time failure is a deterministic property of
462
- * loading the TARGET FILE with a given runner, independent of which test runs it — so
463
- * same-file siblings share it. Keyed on the TARGET FILE only (every caller pins runner to
464
- * undefined): the sole deduped cause is engine_mismatch, a package-level fact both lanes
465
- * classify against the SAME process.version, so the runner must not be in the key. The old
466
- * {runner, target file} key split the cache across lanes (lane 1 wrote "auto <file>", the
467
- * generation lane read "<runner> <file>" -> never matched), silently disabling cross-lane
468
- * dedup. NEVER merges across different files or a different failure class.
469
- */
470
- function dedupKey(runner, targetFileRel) {
471
- return `${runner ?? "auto"}\u0000${targetFileRel}`;
547
+ /** A baseline-green target consumes the conservative generated-proof target quota. */
548
+ function baselineGreenOf(result) {
549
+ if ("status" in result && result.status === "unrunnable")
550
+ return false;
551
+ return result.record.dynamic_proof?.baseline_green === true;
472
552
  }
473
- /** A same-file sibling deduped WITHOUT re-running: shares the first attempt's redacted reason. */
474
- function dedupedAttempt(targetSymbol, testPath, targetFileRel, blocked) {
553
+ function proofDiagnosticsOf(result) {
554
+ if ("status" in result && result.status === "unrunnable")
555
+ return {};
556
+ const cert = result.record.dynamic_proof;
557
+ if (!cert?.command || !cert.cwd || cert.duration_ms === undefined)
558
+ return {};
559
+ return {
560
+ command: cert.command,
561
+ cwd: cert.cwd,
562
+ exit_code: cert.exit_code,
563
+ duration_ms: cert.duration_ms,
564
+ failure_class: cert.failure_class,
565
+ stdout_tail: cert.stdout_tail,
566
+ stderr_tail: cert.stderr_tail,
567
+ runner_fallback: cert.runner_fallback
568
+ };
569
+ }
570
+ function projectBlockKey(language, projectRoot) {
571
+ return `${language}\u0000${projectRoot}`;
572
+ }
573
+ function projectBlockKeys(language, projectRoot, targetFileRel) {
574
+ return [
575
+ projectBlockKey(language, projectRoot),
576
+ `${projectBlockKey(language, projectRoot)}\u0000package:${dirname(targetFileRel).split(sep).join("/")}`
577
+ ];
578
+ }
579
+ function classifiedProjectBlockKey(category, language, projectRoot, targetFileRel) {
580
+ return category === "go_package_build_failure"
581
+ ? projectBlockKeys(language, projectRoot, targetFileRel)[1]
582
+ : projectBlockKey(language, projectRoot);
583
+ }
584
+ /** A same-project candidate skipped WITHOUT re-running after a classified project-wide failure. */
585
+ function dedupedAttempt(targetSymbol, testPath, projectRoot, blocked) {
475
586
  return {
476
587
  target_symbol: targetSymbol,
477
588
  test_path: testPath,
478
589
  classification: "needs_setup",
479
- reason: `${blocked.reason} (shared root cause with a sibling in ${targetFileRel}; not re-run).`,
590
+ reason: `${blocked.reason} (shared project/toolchain cause in ${blocked.scopeLabel}; not re-run).`,
480
591
  category: blocked.category,
592
+ project_root: projectRoot,
593
+ blocked_by: blocked.blockedBy,
481
594
  deduped: true
482
595
  };
483
596
  }
@@ -606,17 +719,58 @@ export function existingAssociatedTests(graph, nodeById) {
606
719
  add(symId, testRel, false);
607
720
  }
608
721
  }
722
+ // Python proof must have an exact hard edge carrying a test selector. A weak
723
+ // file relation (or an empty pre-edge set) is not a runnable assertion link.
724
+ for (const [symId, tests] of out) {
725
+ if (!isPythonFile(symId.split("#")[0].slice(4)))
726
+ continue;
727
+ const preEdges = new Set(proofEdgesFor(graph, symId));
728
+ const linked = tests.filter((t) => t.hard && !!t.testName && PYTHON_TEST_NODEID_SUFFIX_RE.test(t.testName) && preEdges.has(`test:${t.test}->${symId}`));
729
+ if (linked.length)
730
+ out.set(symId, linked);
731
+ else
732
+ out.delete(symId);
733
+ }
609
734
  return out;
610
735
  }
736
+ function declaredPytestTestpaths(sourceRoot, projectRoot) {
737
+ const declared = new Set();
738
+ const root = resolve(sourceRoot, projectRoot);
739
+ for (const config of ["pyproject.toml", "pytest.ini", ".pytest.ini", "tox.ini", "setup.cfg"]) {
740
+ let text = "";
741
+ try {
742
+ text = readFileSync(join(root, config), "utf8");
743
+ }
744
+ catch {
745
+ continue;
746
+ }
747
+ const testpaths = /(?:^|\n)\s*testpaths\s*=\s*([^\n]*(?:\n(?:\s+[^\n#;][^\n]*)?)*)/im.exec(text)?.[1];
748
+ if (!testpaths)
749
+ continue;
750
+ for (const path of testpaths.match(/['"][^'"]+['"]|[^\s,\[\]]+/g) ?? []) {
751
+ const normalized = path.trim().replace(/^['"]|['"]$/g, "").replace(/^\.\//, "").replace(/\/+$/, "");
752
+ if (normalized)
753
+ declared.add(normalized);
754
+ }
755
+ }
756
+ return declared;
757
+ }
758
+ function isInDeclaredPytestTestpath(sourceRoot, projectRoot, testRel, testpaths) {
759
+ const file = testRel.split("::", 1)[0] ?? testRel;
760
+ const projectAbs = resolve(sourceRoot, projectRoot);
761
+ const testAbs = resolve(sourceRoot, file);
762
+ if (testAbs !== projectAbs && !testAbs.startsWith(projectAbs + sep))
763
+ return false;
764
+ const projectTestRel = relative(projectAbs, testAbs).split(sep).join("/");
765
+ return [...testpaths].some((path) => projectTestRel === path || projectTestRel.startsWith(`${path}/`));
766
+ }
611
767
  /**
612
- * Fix 3 — deterministic hard-first, weak-capped attempt order. Every hard TESTED_BY/COVERS
613
- * pair precedes ANY weak MAY_* pair (a provable hard-edge symbol is tried before weak fan-out
614
- * burns the shared budget), and each symbol contributes at most `maxWeakPerSymbol` weak pairs
615
- * so one hot MAY_RELATE_TO file (its every eligible symbol × every importing test) can't starve
616
- * the budget. Order within each tier follows the Map's insertion order (graph node/edge order),
617
- * so the result is deterministic.
768
+ * Deterministic hard-first, weak-capped attempt order. Python is the sole exception to the
769
+ * legacy source order: its existing-test retries are ordered by the unpersisted ORS worklist
770
+ * (including Associated symbols), then direct hard-edge tests, declared pytest `testpaths`, and
771
+ * original source order. Non-Python entries retain the v0.2.44 queue exactly.
618
772
  */
619
- export function orderExistingAttempts(testsBySymbol, maxWeakPerSymbol = EXISTING_LANE_MAX_WEAK_PER_SYMBOL) {
773
+ export function orderExistingAttempts(testsBySymbol, maxWeakPerSymbol = EXISTING_LANE_MAX_WEAK_PER_SYMBOL, schedule) {
620
774
  const hard = [];
621
775
  const weak = [];
622
776
  for (const [symId, tests] of testsBySymbol) {
@@ -628,7 +782,67 @@ export function orderExistingAttempts(testsBySymbol, maxWeakPerSymbol = EXISTING
628
782
  weak.push({ symId, testRel: t.test, hard: false });
629
783
  }
630
784
  }
631
- return [...hard, ...weak];
785
+ if (!schedule)
786
+ return [...hard, ...weak];
787
+ const rankedPython = rankRiskGaps(schedule.graph, { repoRoot: schedule.sourceRoot, limit: 500, includeAssociated: true });
788
+ const pythonRank = new Map(rankedPython.map((gap, index) => [gap.id, { score: gap.risk_score, rank: index }]));
789
+ const decorate = (attempt, index) => {
790
+ const node = schedule.nodeById.get(attempt.symId);
791
+ const targetRel = node ? symbolFileOf(node) : attempt.symId.replace(/^sym:/, "").split("#")[0];
792
+ const python = isPythonFile(targetRel);
793
+ const risk = pythonRank.get(attempt.symId);
794
+ const projectRoot = proofRunnerRoot(schedule.sourceRoot, targetRel, attempt.testRel);
795
+ const pytestTestpaths = python ? declaredPytestTestpaths(schedule.sourceRoot, projectRoot) : new Set();
796
+ return {
797
+ ...attempt,
798
+ language: targetLanguage(attempt.symId),
799
+ projectRoot,
800
+ index,
801
+ python,
802
+ score: risk?.score ?? Number.NEGATIVE_INFINITY,
803
+ rank: risk?.rank ?? Number.MAX_SAFE_INTEGER,
804
+ declaredTestpath: python && isInDeclaredPytestTestpath(schedule.sourceRoot, projectRoot, attempt.testRel, pytestTestpaths)
805
+ };
806
+ };
807
+ const pythonOrsOrder = (attempts) => {
808
+ const decorated = attempts.map(decorate);
809
+ return decorated
810
+ .sort((a, b) =>
811
+ // Python ORS is primary. Direct hard and pytest-config testpaths are strictly
812
+ // equal-ORS tie breakers; project grouping deliberately has no precedence.
813
+ (b.score - a.score)
814
+ || Number(b.hard) - Number(a.hard)
815
+ || Number(b.declaredTestpath) - Number(a.declaredTestpath)
816
+ || (a.index - b.index))
817
+ .map(({ index: _index, python: _python, score: _score, rank: _rank, declaredTestpath: _declaredTestpath, ...attempt }) => attempt);
818
+ };
819
+ const queue = [...hard, ...weak];
820
+ const python = pythonOrsOrder(queue.filter((attempt) => {
821
+ const node = schedule.nodeById.get(attempt.symId);
822
+ return isPythonFile(node ? symbolFileOf(node) : attempt.symId.replace(/^sym:/, "").split("#")[0]);
823
+ }));
824
+ if (python.length === 0)
825
+ return queue;
826
+ let nextPython = 0;
827
+ const scheduled = queue.map((attempt) => {
828
+ const node = schedule.nodeById.get(attempt.symId);
829
+ const targetRel = node ? symbolFileOf(node) : attempt.symId.replace(/^sym:/, "").split("#")[0];
830
+ return isPythonFile(targetRel) ? python[nextPython++] : attempt;
831
+ });
832
+ const primary = proofLanguageOrder(schedule.graph)[0];
833
+ return primary ? scheduled.sort((a, b) => Number(targetLanguage(b.symId) === primary) - Number(targetLanguage(a.symId) === primary)) : scheduled;
834
+ }
835
+ /**
836
+ * Stable scheduling view over the existing ORS-ranked candidate list. This changes
837
+ * only which runner receives a scarce proof attempt first; it never changes scores,
838
+ * evidence tiers, or the persisted priority-gap order.
839
+ */
840
+ export function orderRankedProofCandidates(candidates, graph, sourceRoot) {
841
+ // Generation was v0.2.44 source-ranking order. Python retries are scheduled separately
842
+ // in `orderExistingAttempts`; leave this queue untouched for every language.
843
+ void graph;
844
+ void sourceRoot;
845
+ return candidates;
632
846
  }
633
847
  /**
634
848
  * PR 1.5 lane — prove the repo's OWN existing tests, NO provider key. For each eligible
@@ -642,19 +856,26 @@ export function orderExistingAttempts(testsBySymbol, maxWeakPerSymbol = EXISTING
642
856
  * — the generation lane gets whatever this lane leaves unspent, so TOTAL attempts
643
857
  * (existing + generation) never exceed the budget.
644
858
  */
645
- function proveExistingAssociatedTests(root, graph, sourceRoot, nodeById, opts, proveLoop, proveDeps, alreadyProven, importTimeBlocked, budget) {
859
+ function proveExistingAssociatedTests(root, graph, sourceRoot, nodeById, opts, proveLoop, proveDeps, alreadyProven, projectBlocked, pythonAttemptLimit, pythonGreenTarget, nonPythonAttemptLimit) {
646
860
  const attempts = [];
647
861
  const needsSetup = [];
648
862
  const provenSymbols = new Set();
649
863
  let proven = 0;
650
864
  let attempted = 0;
865
+ let pythonCountedAttempts = 0;
866
+ let pythonGreenBaselines = 0;
867
+ let nonPythonCountedAttempts = 0;
868
+ const pythonPreflightGreen = new Set();
869
+ const pythonBlockedCounts = new Map();
651
870
  const changed = opts.changedFiles && opts.changedFiles.length > 0 ? new Set(opts.changedFiles) : null;
652
871
  const reader = fileReaderFor(sourceRoot); // R-2: source scan for the node:sqlite env profile
653
872
  // Fix 3: hard TESTED_BY/COVERS pairs first, weak MAY_* pairs after and capped per symbol.
654
- const queue = orderExistingAttempts(existingAssociatedTests(graph, nodeById));
655
- for (const { symId, testRel, hard, testName } of queue) {
656
- if (attempted >= budget)
657
- break;
873
+ const queue = orderExistingAttempts(existingAssociatedTests(graph, nodeById), EXISTING_LANE_MAX_WEAK_PER_SYMBOL, {
874
+ graph,
875
+ nodeById,
876
+ sourceRoot
877
+ });
878
+ for (const { symId, testRel, hard, testName, language, projectRoot } of queue) {
658
879
  const node = nodeById.get(symId);
659
880
  // Redundant with existingAssociatedTests' own filter, but the eligibility barrier is
660
881
  // the sole guard against handing plumbing to the guard-less prove path — assert it here too.
@@ -671,20 +892,32 @@ function proveExistingAssociatedTests(root, graph, sourceRoot, nodeById, opts, p
671
892
  if (provenSymbols.has(symId))
672
893
  continue;
673
894
  const targetFileRel = symbolFileOf(node);
674
- // R-1 sibling dedup: a prior same-file attempt hit a package-level env root cause
675
- // (engine_mismatch: runner Node outside the declared engines range). Every sibling in this
676
- // file fails baseline identically → mark it
677
- // needs_setup WITHOUT re-running (and WITHOUT consuming the attempt budget).
895
+ const targetLanguageName = language ?? targetLanguage(symId);
896
+ const runnerRoot = projectRoot ?? proofRunnerRoot(sourceRoot, targetFileRel, testRel);
897
+ // A prior candidate in this exact runner project hit a classified project-wide
898
+ // toolchain/environment failure. Preserve the budget and explain the skip.
678
899
  const isPython = isPythonFile(targetFileRel);
900
+ if (isPython ? (pythonCountedAttempts >= pythonAttemptLimit || pythonGreenBaselines >= pythonGreenTarget) : nonPythonCountedAttempts >= nonPythonAttemptLimit)
901
+ continue;
679
902
  const candidateTestRels = isPython
680
- ? pytestNodeidsForTarget(sourceRoot, testRel, testName).filter((candidate) => isRunnableTestForTarget(node, candidate))
903
+ ? (hard && testName && PYTHON_TEST_NODEID_SUFFIX_RE.test(testName) ? [`${testRel}::${testName}`] : []).filter((candidate) => isRunnableTestForTarget(node, candidate))
681
904
  : [testRel];
682
905
  if (candidateTestRels.length === 0)
683
906
  continue;
684
907
  const proofTestRel = candidateTestRels[0];
685
- const blocked = importTimeBlocked.get(dedupKey(undefined, targetFileRel));
908
+ const blockedKey = projectBlockKeys(targetLanguageName, runnerRoot, targetFileRel).find((key) => projectBlocked.has(key));
909
+ const blocked = blockedKey ? projectBlocked.get(blockedKey) : undefined;
686
910
  if (blocked) {
687
- const attempt = dedupedAttempt(symId, proofTestRel, targetFileRel, blocked);
911
+ // Python preflight is once per project root. The first classified failure
912
+ // is already persisted with full diagnostics; siblings are skipped without
913
+ // becoming fake attempts or thousands of duplicate sidecar rows.
914
+ if (isPython) {
915
+ const key = blockedKey;
916
+ const prior = pythonBlockedCounts.get(key);
917
+ pythonBlockedCounts.set(key, { count: (prior?.count ?? 0) + 1, block: blocked, root: runnerRoot });
918
+ continue;
919
+ }
920
+ const attempt = dedupedAttempt(symId, proofTestRel, runnerRoot, blocked);
688
921
  attempts.push(attempt);
689
922
  needsSetup.push(attempt);
690
923
  continue;
@@ -710,6 +943,8 @@ function proveExistingAssociatedTests(root, graph, sourceRoot, nodeById, opts, p
710
943
  continue;
711
944
  }
712
945
  attempted++;
946
+ if (!isPython)
947
+ nonPythonCountedAttempts++;
713
948
  // R-2: inject NODE_OPTIONS=--experimental-sqlite via the existing test_env path when the
714
949
  // target references node:sqlite. Makes the baseline runnable only; never mints Proven.
715
950
  const testEnv = isGo || isJava || isPython ? undefined : experimentalSqliteTestEnv(reader, targetFileRel);
@@ -726,7 +961,7 @@ function proveExistingAssociatedTests(root, graph, sourceRoot, nodeById, opts, p
726
961
  let result;
727
962
  try {
728
963
  result = proveLoop(root, nativeTestRun
729
- ? { target_symbol: symId, source: sourceRoot, test_run: nativeTestRun, ...(goAssertionLine !== undefined ? { go_assertion_line: goAssertionLine } : {}), run_id: `auto-prove-existing-${attempted}` }
964
+ ? { target_symbol: symId, source: sourceRoot, test_run: nativeTestRun, ...(goAssertionLine !== undefined ? { go_assertion_line: goAssertionLine } : {}), run_id: `auto-prove-${attempted}` }
730
965
  // link_node_modules: the isolated proof copy excludes node_modules; without linking,
731
966
  // any target/test importing a repo dependency fails baseline → needs_setup. Linking only
732
967
  // makes real tests runnable — Proven still requires the dynamic oracle's sentinel kill.
@@ -736,29 +971,38 @@ function proveExistingAssociatedTests(root, graph, sourceRoot, nodeById, opts, p
736
971
  test_path: proofTestRel,
737
972
  replacement: replacementForTarget(targetFileRel),
738
973
  link_node_modules: true,
974
+ ...(isPython ? { skip_preflight: pythonPreflightGreen.has(runnerRoot) } : {}),
739
975
  ...(testEnv ? { test_env: testEnv } : {}),
740
- run_id: `auto-prove-existing-${attempted}`
976
+ run_id: `auto-prove-${attempted}`
741
977
  }, proveDeps);
742
978
  }
743
979
  catch (e) {
744
- const attempt = {
745
- target_symbol: symId,
746
- test_path: displayTest,
747
- classification: "needs_setup",
748
- reason: `Proof could not run: ${redactSecrets(errMsg(e))}`
749
- };
750
- attempts.push(attempt);
751
- needsSetup.push(attempt);
980
+ attempted--;
981
+ if (!isPython)
982
+ nonPythonCountedAttempts--;
983
+ reportProgress(`proof skipped: ${symId}: ${redactSecrets(errMsg(e))}`);
752
984
  continue;
753
985
  }
754
986
  const { classification, reason, category } = classifyProof(result, { sourceRoot, targetFileRel });
987
+ const baselineGreen = baselineGreenOf(result);
988
+ if (isPython) {
989
+ const preflightUnavailable = category === "environment_unavailable" || category === "collection_error";
990
+ if (!preflightUnavailable) {
991
+ pythonCountedAttempts++;
992
+ pythonPreflightGreen.add(runnerRoot);
993
+ }
994
+ if (baselineGreen)
995
+ pythonGreenBaselines++;
996
+ }
755
997
  const attempt = {
756
998
  target_symbol: symId,
757
999
  test_path: displayTest,
758
1000
  classification,
759
1001
  reason,
760
1002
  category,
761
- mutant_status: mutantStatusOf(result)
1003
+ project_root: runnerRoot,
1004
+ mutant_status: mutantStatusOf(result),
1005
+ ...proofDiagnosticsOf(result)
762
1006
  };
763
1007
  attempts.push(attempt);
764
1008
  if (classification === "proven") {
@@ -768,14 +1012,31 @@ function proveExistingAssociatedTests(root, graph, sourceRoot, nodeById, opts, p
768
1012
  }
769
1013
  if (classification === "needs_setup") {
770
1014
  needsSetup.push(attempt);
771
- // Cache an import-time root cause so same-file siblings dedup instead of re-running.
772
- if (category && IMPORT_TIME_CATEGORIES.has(category)) {
773
- importTimeBlocked.set(dedupKey(undefined, targetFileRel), { category, reason: reason ?? "" });
1015
+ // Quarantine only a classified project-wide root cause. Target/test-specific
1016
+ // red baselines and surviving mutants never suppress siblings.
1017
+ if (category && PROJECT_WIDE_BLOCKERS.has(category)) {
1018
+ const scopeLabel = category === "go_package_build_failure"
1019
+ ? dirname(targetFileRel).split(sep).join("/") || "."
1020
+ : runnerRoot;
1021
+ projectBlocked.set(classifiedProjectBlockKey(category, targetLanguageName, runnerRoot, targetFileRel), {
1022
+ category,
1023
+ reason: reason ?? "",
1024
+ blockedBy: symId,
1025
+ scopeLabel
1026
+ });
774
1027
  }
775
1028
  }
776
1029
  // non_killing → keep trying this symbol's other associated tests, if any.
777
1030
  }
778
- return { attempts, needsSetup, proven, attempted, provenSymbols };
1031
+ const skipped = [...pythonBlockedCounts.values()].map(({ count, block, root }) => ({
1032
+ title: `Python project ${root}`,
1033
+ reason: `${count} remaining linked target${count === 1 ? " was" : "s were"} in a project root marked ${block.category}; no further runner calls were made in that root.`,
1034
+ category: block.category,
1035
+ language: "python",
1036
+ project_root: root,
1037
+ blocked_by: block.blockedBy
1038
+ }));
1039
+ return { attempts, needsSetup, skipped, proven, attempted, nonPythonAttempted: nonPythonCountedAttempts, provenSymbols };
779
1040
  }
780
1041
  /**
781
1042
  * Drive prove for the top provable TS/JS targets. Two lanes: (1) prove the repo's OWN
@@ -810,16 +1071,18 @@ export async function autoProve(root, opts, deps) {
810
1071
  .filter((r) => r.evidence_tier === "proven")
811
1072
  .map((r) => r.code_symbol)
812
1073
  .filter(Boolean));
813
- // R-1: shared sibling-dedup cache of import-time baseline failures. Spans BOTH lanes so a
814
- // node:sqlite-style root cause found once is never re-run across same-file siblings.
815
- const importTimeBlocked = new Map();
816
- // ONE unified dynamic-proof budget for the whole pass (existing-first → then generation).
817
- // Default 5 ("dynamically prove top 5"); `--auto-limit N` overrides it, clamped to
818
- // MAX_AUTO_LIMIT. The existing lane consumes from this budget and the generation lane gets
819
- // only the remainder, so TOTAL attempts (existing + generation) are ≤ budget.
1074
+ // Shared, run-local quarantine for classified project-wide toolchain failures.
1075
+ // Keyed by language + bounded runner root; never populated by a test-specific red
1076
+ // baseline, an unknown failure, or a surviving mutant.
1077
+ const projectBlocked = new Map();
1078
+ // R7.1 adds a Python-only existing-test retry lane. Generic/non-Python calls retain
1079
+ // the released `autoLimit`; Python can inspect more existing targets to find
1080
+ // the requested number of baseline-green tests without widening any other language.
820
1081
  const autoLimit = Math.max(1, Math.min(MAX_AUTO_LIMIT, Math.floor(opts.autoLimit ?? DEFAULT_AUTO_LIMIT)));
1082
+ const existingLimit = Math.max(1, Math.min(MAX_AUTO_LIMIT, Math.floor(opts.proof_existing_limit ?? DEFAULT_PROOF_EXISTING_LIMIT)));
1083
+ const genLimit = Math.max(1, Math.min(MAX_AUTO_LIMIT, Math.floor(opts.proof_gen_limit ?? DEFAULT_PROOF_GEN_LIMIT)));
821
1084
  // ── Lane 1: existing associated tests — NO key required, runs FIRST (PR 1.5). ──
822
- const ex = proveExistingAssociatedTests(root, graph, sourceRoot, nodeById, opts, proveLoop, proveDeps, alreadyProven, importTimeBlocked, autoLimit);
1085
+ const ex = proveExistingAssociatedTests(root, graph, sourceRoot, nodeById, opts, proveLoop, proveDeps, alreadyProven, projectBlocked, existingLimit, genLimit, autoLimit);
823
1086
  if (opts.existingOnly) {
824
1087
  const status = ex.proven > 0 ? "proven-run" : ex.attempted > 0 ? "ran-no-proof" : "no-targets";
825
1088
  return {
@@ -829,7 +1092,7 @@ export async function autoProve(root, opts, deps) {
829
1092
  attempted: ex.attempted,
830
1093
  proven: ex.proven,
831
1094
  needs_setup: ex.needsSetup,
832
- skipped: [],
1095
+ skipped: ex.skipped,
833
1096
  generated_files: [],
834
1097
  attempts: ex.attempts
835
1098
  };
@@ -848,7 +1111,7 @@ export async function autoProve(root, opts, deps) {
848
1111
  attempted: ex.attempted,
849
1112
  proven: ex.proven,
850
1113
  needs_setup: ex.needsSetup,
851
- skipped: [],
1114
+ skipped: ex.skipped,
852
1115
  generated_files: [],
853
1116
  attempts: ex.attempts
854
1117
  };
@@ -856,9 +1119,9 @@ export async function autoProve(root, opts, deps) {
856
1119
  const provider = deterministic ? new DeterministicProvider() : buildProvider(providerConfig);
857
1120
  const generate = deps.generate ?? generateTests;
858
1121
  const reader = fileReaderFor(sourceRoot);
859
- // Generation gets only the budget the existing-tests lane left unspent, so existing +
860
- // generation attempts total ≤ autoLimit. Exhausted budget ⇒ genBudget 0 ⇒ no provider call.
861
- const genBudget = Math.max(0, autoLimit - ex.attempted);
1122
+ // Generated proofs retain the released generic limit. Python's green-target cap
1123
+ // applies only to existing-test retries above, not to non-Python generation.
1124
+ const generationLimit = Math.max(0, autoLimit - ex.nonPythonAttempted);
862
1125
  // Candidates = ORS-ranked provable CodeSymbols. rankRiskGaps ranks by OrangePro Risk
863
1126
  // Score and excludes hard-confirmed symbols; we ALSO enforce the eligibility barrier
864
1127
  // explicitly (isEligibleProvableTarget) at selection so an excluded infra symbol can
@@ -872,6 +1135,7 @@ export async function autoProve(root, opts, deps) {
872
1135
  const changed = new Set(opts.changedFiles);
873
1136
  candidates = candidates.filter((g) => changed.has(g.file));
874
1137
  }
1138
+ candidates = orderRankedProofCandidates(candidates, graph, sourceRoot);
875
1139
  const attempts = [];
876
1140
  const needsSetup = [];
877
1141
  const skipped = [];
@@ -879,16 +1143,17 @@ export async function autoProve(root, opts, deps) {
879
1143
  const declaredDeps = readDeclaredDeps(sourceRoot);
880
1144
  let proven = 0;
881
1145
  let attempted = 0;
882
- for (let start = 0; start < candidates.length && attempted < genBudget; start += GEN_WINDOW) {
1146
+ let baselineGreenTargets = 0;
1147
+ for (let start = 0; start < candidates.length && attempted < autoLimit && baselineGreenTargets < generationLimit; start += GEN_WINDOW) {
883
1148
  const window = candidates.slice(start, start + GEN_WINDOW);
884
- const need = genBudget - attempted;
1149
+ const need = Math.max(1, Math.min(GEN_WINDOW, generationLimit - baselineGreenTargets));
885
1150
  const windowIds = window.map((g) => g.id);
886
1151
  const gen = await generate(graph, { target_ids: windowIds, limit: Math.min(windowIds.length, need), ...(opts.prompt_version ? { prompt_version: opts.prompt_version } : {}) }, provider, reader, clock);
887
1152
  const tests = gen.generated_tests;
888
1153
  // Global start offset so filenames stay unique across windows — runHintsFor
889
1154
  // otherwise resets its index to 0 per window and same-slug targets collide.
890
1155
  const hints = runHintsFor(tests, sourceRoot, start);
891
- for (let i = 0; i < tests.length && attempted < genBudget; i++) {
1156
+ for (let i = 0; i < tests.length && attempted < autoLimit && baselineGreenTargets < generationLimit; i++) {
892
1157
  const test = tests[i];
893
1158
  const hint = hints[i];
894
1159
  if (!hint.prove_run) {
@@ -917,14 +1182,24 @@ export async function autoProve(root, opts, deps) {
917
1182
  }
918
1183
  const { target_symbol, replacement, runner } = hint.prove_run.args;
919
1184
  const targetFileRel = symbolFile(target_symbol);
920
- // R-1 sibling dedup (cross-lane): a same-file target already hit an import-time env root
921
- // cause → a fresh generated test importing the same module fails identically. Skip it
922
- // WITHOUT generating/writing/running or consuming the attempt budget.
923
- const blocked = importTimeBlocked.get(dedupKey(undefined, targetFileRel));
1185
+ const language = targetLanguage(target_symbol);
1186
+ const projectRoot = proofRunnerRoot(sourceRoot, targetFileRel);
1187
+ // Cross-lane quarantine: a classified project-wide runner failure already
1188
+ // explains why this generated candidate cannot run. Skip without spending budget.
1189
+ const blockedKey = projectBlockKeys(language, projectRoot, targetFileRel).find((key) => projectBlocked.has(key));
1190
+ const blocked = blockedKey ? projectBlocked.get(blockedKey) : undefined;
924
1191
  if (blocked) {
925
- const attempt = dedupedAttempt(target_symbol, "", targetFileRel, blocked);
1192
+ const attempt = dedupedAttempt(target_symbol, "", projectRoot, blocked);
926
1193
  attempts.push(attempt);
927
1194
  needsSetup.push(attempt);
1195
+ skipped.push({
1196
+ target_symbol,
1197
+ title: test.title,
1198
+ reason: attempt.reason ?? "Skipped after a project-wide proof failure.",
1199
+ language,
1200
+ project_root: projectRoot,
1201
+ blocked_by: blocked.blockedBy
1202
+ });
928
1203
  continue;
929
1204
  }
930
1205
  const filename = basename(hint.prove_run.args.test_path);
@@ -976,18 +1251,12 @@ export async function autoProve(root, opts, deps) {
976
1251
  // See lane 1: link node_modules so a generated test importing a repo dep can boot.
977
1252
  link_node_modules: true,
978
1253
  ...(testEnv ? { test_env: testEnv } : {}),
979
- run_id: `auto-prove-${start + i + 1}`
1254
+ run_id: `auto-prove-${ex.attempted + attempted}`
980
1255
  }, proveDeps);
981
1256
  }
982
1257
  catch (e) {
983
- const attempt = {
984
- target_symbol,
985
- test_path: writeRel,
986
- classification: "needs_setup",
987
- reason: `Proof could not run: ${redactSecrets(errMsg(e))}`
988
- };
989
- attempts.push(attempt);
990
- needsSetup.push(attempt);
1258
+ attempted--;
1259
+ skipped.push({ target_symbol, title: test.title, reason: `Proof could not run: ${redactSecrets(errMsg(e))}`, language, project_root: projectRoot });
991
1260
  continue;
992
1261
  }
993
1262
  const { classification, reason, category } = classifyProof(result, { sourceRoot, targetFileRel });
@@ -997,16 +1266,27 @@ export async function autoProve(root, opts, deps) {
997
1266
  classification,
998
1267
  reason,
999
1268
  category,
1000
- mutant_status: mutantStatusOf(result)
1269
+ project_root: projectRoot,
1270
+ mutant_status: mutantStatusOf(result),
1271
+ ...proofDiagnosticsOf(result)
1001
1272
  };
1002
1273
  attempts.push(attempt);
1274
+ if (baselineGreenOf(result))
1275
+ baselineGreenTargets++;
1003
1276
  if (classification === "proven")
1004
1277
  proven++;
1005
1278
  else if (classification === "needs_setup") {
1006
1279
  needsSetup.push(attempt);
1007
- // Cache an import-time root cause so same-file siblings dedup instead of re-running.
1008
- if (category && IMPORT_TIME_CATEGORIES.has(category)) {
1009
- importTimeBlocked.set(dedupKey(undefined, targetFileRel), { category, reason: reason ?? "" });
1280
+ if (category && PROJECT_WIDE_BLOCKERS.has(category)) {
1281
+ const scopeLabel = category === "go_package_build_failure"
1282
+ ? dirname(targetFileRel).split(sep).join("/") || "."
1283
+ : projectRoot;
1284
+ projectBlocked.set(classifiedProjectBlockKey(category, language, projectRoot, targetFileRel), {
1285
+ category,
1286
+ reason: reason ?? "",
1287
+ blockedBy: target_symbol,
1288
+ scopeLabel
1289
+ });
1010
1290
  }
1011
1291
  }
1012
1292
  // non_killing stays in `attempts` only — an honest skip, never Proven.
@@ -1025,7 +1305,7 @@ export async function autoProve(root, opts, deps) {
1025
1305
  attempted: totalAttempted,
1026
1306
  proven: totalProven,
1027
1307
  needs_setup: [...ex.needsSetup, ...needsSetup],
1028
- skipped,
1308
+ skipped: [...ex.skipped, ...skipped],
1029
1309
  generated_files: generatedFiles,
1030
1310
  attempts: [...ex.attempts, ...attempts]
1031
1311
  };