@orangepro/orangepro-mcp 0.2.44 → 0.2.46

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -21,7 +21,7 @@ import { resolveProviderConfig } from "./localConfig.js";
21
21
  import { buildProvider, DeterministicProvider } from "./generate/providers.js";
22
22
  import { resolveContained } from "./reprove/paths.js";
23
23
  import { buildRtm } from "./rtm.js";
24
- import { loadLedger, targetLanguage } from "./ledger.js";
24
+ import { loadLedger, proofEdgesFor, targetLanguage } from "./ledger.js";
25
25
  import { reportProgress } from "./util/progress.js";
26
26
  import { loadGraph, workspacePaths } from "./workspace.js";
27
27
  import { systemClock } from "./util/time.js";
@@ -82,7 +82,7 @@ function pytestNodeidsForFile(sourceRoot, testRel) {
82
82
  }
83
83
  if (currentClass && indent <= classIndent && line.trim() !== "" && !line.startsWith(" "))
84
84
  currentClass = null;
85
- const fnMatch = /^(\s*)def\s+(test_[A-Za-z0-9_]*)\s*\(/.exec(line);
85
+ const fnMatch = /^(\s*)(?:async\s+)?def\s+(test_[A-Za-z0-9_]*)\s*\(/.exec(line);
86
86
  if (!fnMatch)
87
87
  continue;
88
88
  const fnIndent = fnMatch[1].length;
@@ -358,13 +358,12 @@ export function javaTestForTarget(nodeById, graph, symId) {
358
358
  return selectors.size === 1 ? [...selectors][0] : null;
359
359
  }
360
360
  const GENERATED_HEADER = "// Generated by OrangePro — do not edit";
361
- // "Static map first, dynamically prove top 5": ONE unified dynamic-proof budget spans
362
- // BOTH lanes (existing-tests first, then generation for the remaining eligible gaps).
363
- // TOTAL attempts (existing + generation) are capped at this budget; `--auto-limit N`
364
- // overrides it (clamped to MAX_AUTO_LIMIT) for deeper runs (`opro start --auto-limit 25`,
365
- // `opro prove`, `opro prove-loop`). Static breadth (behaviors/flows/associated/risk) is
366
- // never gated on this budget — only the dynamic verification pass is.
361
+ // Released v0.2.44 generic dynamic-proof limit. Python's existing-test lane has its
362
+ // own retry limit below; do not widen this generic default or TS/JS, Go, Java, and
363
+ // generated-test scheduling changes before a caller opts in.
367
364
  const DEFAULT_AUTO_LIMIT = 5;
365
+ const DEFAULT_PROOF_EXISTING_LIMIT = 20;
366
+ const DEFAULT_PROOF_GEN_LIMIT = 5;
368
367
  const MAX_AUTO_LIMIT = 50;
369
368
  /** Generator caps a single call at 5 (MAX_LIMIT); page candidates in windows of that. */
370
369
  const GEN_WINDOW = 5;
@@ -381,10 +380,10 @@ const RUNNER_ROOT_MARKERS = {
381
380
  rs: ["Cargo.toml"]
382
381
  };
383
382
  /**
384
- * The nearest language runner root for a target, bounded by the analyzed source root.
385
- * A nested root is accepted only when the selected existing test is inside it too;
386
- * otherwise proof execution falls back to the analyzed root, matching the oracle's
387
- * containment rule rather than inventing a cross-project plan.
383
+ * The nearest language runner root, bounded by the analyzed source root. Python is
384
+ * test-owned: pytest configuration/environment follows the selected test file even when
385
+ * that test exercises a target elsewhere in the same repository. Other languages keep
386
+ * their target-owned v0.2.44 behavior.
388
387
  */
389
388
  export function proofRunnerRoot(sourceRoot, targetRel, testRel) {
390
389
  const stop = resolve(sourceRoot);
@@ -393,7 +392,11 @@ export function proofRunnerRoot(sourceRoot, targetRel, testRel) {
393
392
  return ".";
394
393
  const language = targetLanguage(`sym:${targetRel}#target`);
395
394
  const markers = RUNNER_ROOT_MARKERS[language] ?? [];
396
- let dir = dirname(targetAbs);
395
+ const testFile = testRel?.split("::", 1)[0] ?? testRel;
396
+ const testAbs = testFile ? resolve(stop, testFile) : undefined;
397
+ if (language === "python" && testAbs && testAbs !== stop && !testAbs.startsWith(stop + sep))
398
+ return ".";
399
+ let dir = dirname(language === "python" && testAbs ? testAbs : targetAbs);
397
400
  let found = stop;
398
401
  for (;;) {
399
402
  if (markers.some((marker) => existsSync(join(dir, marker)))) {
@@ -407,13 +410,6 @@ export function proofRunnerRoot(sourceRoot, targetRel, testRel) {
407
410
  break;
408
411
  dir = parent;
409
412
  }
410
- if (testRel && found !== stop) {
411
- const testFile = testRel.split("::", 1)[0] ?? testRel;
412
- const testAbs = resolve(stop, testFile);
413
- if ((testAbs !== stop && !testAbs.startsWith(stop + sep))
414
- || (testAbs !== found && !testAbs.startsWith(found + sep)))
415
- found = stop;
416
- }
417
413
  const rel = relative(stop, found).split(sep).join("/");
418
414
  return rel || ".";
419
415
  }
@@ -438,7 +434,9 @@ const PROJECT_WIDE_BLOCKERS = new Set([
438
434
  "tsconfig_missing",
439
435
  "runner_missing",
440
436
  "module_root_missing",
441
- "go_package_build_failure"
437
+ "go_package_build_failure",
438
+ "environment_unavailable",
439
+ "collection_error"
442
440
  ]);
443
441
  export function isRoastSurvivor(attempt) {
444
442
  return attempt.classification === "non_killing" && attempt.mutant_status === "associated_survived";
@@ -513,6 +511,14 @@ function classifyProof(result, ctx) {
513
511
  const record = dyn.record;
514
512
  if (record.closed)
515
513
  return { classification: "proven" };
514
+ // Language spikes may refuse before a baseline exists (for example, Python's
515
+ // collection preflight, an explicit project boundary, or a type-unsafe mutation).
516
+ // These are structured runner outcomes, not a raw-stderr heuristic. Never mint a
517
+ // proof and never mislabel a mutation refusal as a surviving test.
518
+ const runnerCategory = dyn.oracle.category;
519
+ if (runnerCategory === "environment_unavailable" || runnerCategory === "collection_error" || runnerCategory === "project_boundary" || runnerCategory === "mutation_unsupported") {
520
+ return { classification: "needs_setup", reason: dyn.oracle.reason, category: runnerCategory };
521
+ }
516
522
  const cert = record.dynamic_proof;
517
523
  if (cert && cert.baseline_green === false) {
518
524
  const failureSummary = dyn.oracle.baseline?.failureSummary;
@@ -538,6 +544,29 @@ function mutantStatusOf(result) {
538
544
  return "unrunnable";
539
545
  return result.record.dynamic_proof?.mutant_status;
540
546
  }
547
+ /** A baseline-green target consumes the conservative generated-proof target quota. */
548
+ function baselineGreenOf(result) {
549
+ if ("status" in result && result.status === "unrunnable")
550
+ return false;
551
+ return result.record.dynamic_proof?.baseline_green === true;
552
+ }
553
+ function proofDiagnosticsOf(result) {
554
+ if ("status" in result && result.status === "unrunnable")
555
+ return {};
556
+ const cert = result.record.dynamic_proof;
557
+ if (!cert?.command || !cert.cwd || cert.duration_ms === undefined)
558
+ return {};
559
+ return {
560
+ command: cert.command,
561
+ cwd: cert.cwd,
562
+ exit_code: cert.exit_code,
563
+ duration_ms: cert.duration_ms,
564
+ failure_class: cert.failure_class,
565
+ stdout_tail: cert.stdout_tail,
566
+ stderr_tail: cert.stderr_tail,
567
+ runner_fallback: cert.runner_fallback
568
+ };
569
+ }
541
570
  function projectBlockKey(language, projectRoot) {
542
571
  return `${language}\u0000${projectRoot}`;
543
572
  }
@@ -690,15 +719,56 @@ export function existingAssociatedTests(graph, nodeById) {
690
719
  add(symId, testRel, false);
691
720
  }
692
721
  }
722
+ // Python proof must have an exact hard edge carrying a test selector. A weak
723
+ // file relation (or an empty pre-edge set) is not a runnable assertion link.
724
+ for (const [symId, tests] of out) {
725
+ if (!isPythonFile(symId.split("#")[0].slice(4)))
726
+ continue;
727
+ const preEdges = new Set(proofEdgesFor(graph, symId));
728
+ const linked = tests.filter((t) => t.hard && !!t.testName && PYTHON_TEST_NODEID_SUFFIX_RE.test(t.testName) && preEdges.has(`test:${t.test}->${symId}`));
729
+ if (linked.length)
730
+ out.set(symId, linked);
731
+ else
732
+ out.delete(symId);
733
+ }
693
734
  return out;
694
735
  }
736
+ function declaredPytestTestpaths(sourceRoot, projectRoot) {
737
+ const declared = new Set();
738
+ const root = resolve(sourceRoot, projectRoot);
739
+ for (const config of ["pyproject.toml", "pytest.ini", ".pytest.ini", "tox.ini", "setup.cfg"]) {
740
+ let text = "";
741
+ try {
742
+ text = readFileSync(join(root, config), "utf8");
743
+ }
744
+ catch {
745
+ continue;
746
+ }
747
+ const testpaths = /(?:^|\n)\s*testpaths\s*=\s*([^\n]*(?:\n(?:\s+[^\n#;][^\n]*)?)*)/im.exec(text)?.[1];
748
+ if (!testpaths)
749
+ continue;
750
+ for (const path of testpaths.match(/['"][^'"]+['"]|[^\s,\[\]]+/g) ?? []) {
751
+ const normalized = path.trim().replace(/^['"]|['"]$/g, "").replace(/^\.\//, "").replace(/\/+$/, "");
752
+ if (normalized)
753
+ declared.add(normalized);
754
+ }
755
+ }
756
+ return declared;
757
+ }
758
+ function isInDeclaredPytestTestpath(sourceRoot, projectRoot, testRel, testpaths) {
759
+ const file = testRel.split("::", 1)[0] ?? testRel;
760
+ const projectAbs = resolve(sourceRoot, projectRoot);
761
+ const testAbs = resolve(sourceRoot, file);
762
+ if (testAbs !== projectAbs && !testAbs.startsWith(projectAbs + sep))
763
+ return false;
764
+ const projectTestRel = relative(projectAbs, testAbs).split(sep).join("/");
765
+ return [...testpaths].some((path) => projectTestRel === path || projectTestRel.startsWith(`${path}/`));
766
+ }
695
767
  /**
696
- * Fix 3 — deterministic hard-first, weak-capped attempt order. Every hard TESTED_BY/COVERS
697
- * pair precedes ANY weak MAY_* pair (a provable hard-edge symbol is tried before weak fan-out
698
- * burns the shared budget), and each symbol contributes at most `maxWeakPerSymbol` weak pairs
699
- * so one hot MAY_RELATE_TO file (its every eligible symbol × every importing test) can't starve
700
- * the budget. Order within each tier follows the Map's insertion order (graph node/edge order),
701
- * so the result is deterministic.
768
+ * Deterministic hard-first, weak-capped attempt order. Python is the sole exception to the
769
+ * legacy source order: its existing-test retries are ordered by the unpersisted ORS worklist
770
+ * (including Associated symbols), then direct hard-edge tests, declared pytest `testpaths`, and
771
+ * original source order. Non-Python entries retain the v0.2.44 queue exactly.
702
772
  */
703
773
  export function orderExistingAttempts(testsBySymbol, maxWeakPerSymbol = EXISTING_LANE_MAX_WEAK_PER_SYMBOL, schedule) {
704
774
  const hard = [];
@@ -714,33 +784,53 @@ export function orderExistingAttempts(testsBySymbol, maxWeakPerSymbol = EXISTING
714
784
  }
715
785
  if (!schedule)
716
786
  return [...hard, ...weak];
717
- const languageOrder = proofLanguageOrder(schedule.graph);
718
- const languageRank = new Map(languageOrder.map((language, index) => [language, index]));
787
+ const rankedPython = rankRiskGaps(schedule.graph, { repoRoot: schedule.sourceRoot, limit: 500, includeAssociated: true });
788
+ const pythonRank = new Map(rankedPython.map((gap, index) => [gap.id, { score: gap.risk_score, rank: index }]));
719
789
  const decorate = (attempt, index) => {
720
790
  const node = schedule.nodeById.get(attempt.symId);
721
791
  const targetRel = node ? symbolFileOf(node) : attempt.symId.replace(/^sym:/, "").split("#")[0];
792
+ const python = isPythonFile(targetRel);
793
+ const risk = pythonRank.get(attempt.symId);
794
+ const projectRoot = proofRunnerRoot(schedule.sourceRoot, targetRel, attempt.testRel);
795
+ const pytestTestpaths = python ? declaredPytestTestpaths(schedule.sourceRoot, projectRoot) : new Set();
722
796
  return {
723
797
  ...attempt,
724
798
  language: targetLanguage(attempt.symId),
725
- projectRoot: proofRunnerRoot(schedule.sourceRoot, targetRel, attempt.testRel),
726
- index
799
+ projectRoot,
800
+ index,
801
+ python,
802
+ score: risk?.score ?? Number.NEGATIVE_INFINITY,
803
+ rank: risk?.rank ?? Number.MAX_SAFE_INTEGER,
804
+ declaredTestpath: python && isInDeclaredPytestTestpath(schedule.sourceRoot, projectRoot, attempt.testRel, pytestTestpaths)
727
805
  };
728
806
  };
729
- const stableProjectOrder = (attempts) => {
807
+ const pythonOrsOrder = (attempts) => {
730
808
  const decorated = attempts.map(decorate);
731
- const firstProject = new Map();
732
- for (const item of decorated) {
733
- const key = `${item.language}\u0000${item.projectRoot}`;
734
- if (!firstProject.has(key))
735
- firstProject.set(key, firstProject.size);
736
- }
737
809
  return decorated
738
- .sort((a, b) => ((languageRank.get(a.language) ?? Number.MAX_SAFE_INTEGER) - (languageRank.get(b.language) ?? Number.MAX_SAFE_INTEGER))
739
- || ((firstProject.get(`${a.language}\u0000${a.projectRoot}`) ?? 0) - (firstProject.get(`${b.language}\u0000${b.projectRoot}`) ?? 0))
810
+ .sort((a, b) =>
811
+ // Python ORS is primary. Direct hard and pytest-config testpaths are strictly
812
+ // equal-ORS tie breakers; project grouping deliberately has no precedence.
813
+ (b.score - a.score)
814
+ || Number(b.hard) - Number(a.hard)
815
+ || Number(b.declaredTestpath) - Number(a.declaredTestpath)
740
816
  || (a.index - b.index))
741
- .map(({ index: _index, ...attempt }) => attempt);
817
+ .map(({ index: _index, python: _python, score: _score, rank: _rank, declaredTestpath: _declaredTestpath, ...attempt }) => attempt);
742
818
  };
743
- return [...stableProjectOrder(hard), ...stableProjectOrder(weak)];
819
+ const queue = [...hard, ...weak];
820
+ const python = pythonOrsOrder(queue.filter((attempt) => {
821
+ const node = schedule.nodeById.get(attempt.symId);
822
+ return isPythonFile(node ? symbolFileOf(node) : attempt.symId.replace(/^sym:/, "").split("#")[0]);
823
+ }));
824
+ if (python.length === 0)
825
+ return queue;
826
+ let nextPython = 0;
827
+ const scheduled = queue.map((attempt) => {
828
+ const node = schedule.nodeById.get(attempt.symId);
829
+ const targetRel = node ? symbolFileOf(node) : attempt.symId.replace(/^sym:/, "").split("#")[0];
830
+ return isPythonFile(targetRel) ? python[nextPython++] : attempt;
831
+ });
832
+ const primary = proofLanguageOrder(schedule.graph)[0];
833
+ return primary ? scheduled.sort((a, b) => Number(targetLanguage(b.symId) === primary) - Number(targetLanguage(a.symId) === primary)) : scheduled;
744
834
  }
745
835
  /**
746
836
  * Stable scheduling view over the existing ORS-ranked candidate list. This changes
@@ -748,24 +838,11 @@ export function orderExistingAttempts(testsBySymbol, maxWeakPerSymbol = EXISTING
748
838
  * evidence tiers, or the persisted priority-gap order.
749
839
  */
750
840
  export function orderRankedProofCandidates(candidates, graph, sourceRoot) {
751
- const languageOrder = proofLanguageOrder(graph);
752
- if (languageOrder.length <= 1)
753
- return candidates;
754
- const languageRank = new Map(languageOrder.map((language, index) => [language, index]));
755
- const firstProject = new Map();
756
- const decorated = candidates.map((candidate, index) => {
757
- const language = targetLanguage(candidate.id);
758
- const projectRoot = proofRunnerRoot(sourceRoot, candidate.file);
759
- const projectKey = `${language}\u0000${projectRoot}`;
760
- if (!firstProject.has(projectKey))
761
- firstProject.set(projectKey, firstProject.size);
762
- return { candidate, index, language, projectKey };
763
- });
764
- return decorated
765
- .sort((a, b) => ((languageRank.get(a.language) ?? Number.MAX_SAFE_INTEGER) - (languageRank.get(b.language) ?? Number.MAX_SAFE_INTEGER))
766
- || ((firstProject.get(a.projectKey) ?? 0) - (firstProject.get(b.projectKey) ?? 0))
767
- || (a.index - b.index))
768
- .map(({ candidate }) => candidate);
841
+ // Generation was v0.2.44 source-ranking order. Python retries are scheduled separately
842
+ // in `orderExistingAttempts`; leave this queue untouched for every language.
843
+ void graph;
844
+ void sourceRoot;
845
+ return candidates;
769
846
  }
770
847
  /**
771
848
  * PR 1.5 lane — prove the repo's OWN existing tests, NO provider key. For each eligible
@@ -779,12 +856,17 @@ export function orderRankedProofCandidates(candidates, graph, sourceRoot) {
779
856
  * — the generation lane gets whatever this lane leaves unspent, so TOTAL attempts
780
857
  * (existing + generation) never exceed the budget.
781
858
  */
782
- function proveExistingAssociatedTests(root, graph, sourceRoot, nodeById, opts, proveLoop, proveDeps, alreadyProven, projectBlocked, budget) {
859
+ function proveExistingAssociatedTests(root, graph, sourceRoot, nodeById, opts, proveLoop, proveDeps, alreadyProven, projectBlocked, pythonAttemptLimit, pythonGreenTarget, nonPythonAttemptLimit) {
783
860
  const attempts = [];
784
861
  const needsSetup = [];
785
862
  const provenSymbols = new Set();
786
863
  let proven = 0;
787
864
  let attempted = 0;
865
+ let pythonCountedAttempts = 0;
866
+ let pythonGreenBaselines = 0;
867
+ let nonPythonCountedAttempts = 0;
868
+ const pythonPreflightGreen = new Set();
869
+ const pythonBlockedCounts = new Map();
788
870
  const changed = opts.changedFiles && opts.changedFiles.length > 0 ? new Set(opts.changedFiles) : null;
789
871
  const reader = fileReaderFor(sourceRoot); // R-2: source scan for the node:sqlite env profile
790
872
  // Fix 3: hard TESTED_BY/COVERS pairs first, weak MAY_* pairs after and capped per symbol.
@@ -794,8 +876,6 @@ function proveExistingAssociatedTests(root, graph, sourceRoot, nodeById, opts, p
794
876
  sourceRoot
795
877
  });
796
878
  for (const { symId, testRel, hard, testName, language, projectRoot } of queue) {
797
- if (attempted >= budget)
798
- break;
799
879
  const node = nodeById.get(symId);
800
880
  // Redundant with existingAssociatedTests' own filter, but the eligibility barrier is
801
881
  // the sole guard against handing plumbing to the guard-less prove path — assert it here too.
@@ -817,8 +897,10 @@ function proveExistingAssociatedTests(root, graph, sourceRoot, nodeById, opts, p
817
897
  // A prior candidate in this exact runner project hit a classified project-wide
818
898
  // toolchain/environment failure. Preserve the budget and explain the skip.
819
899
  const isPython = isPythonFile(targetFileRel);
900
+ if (isPython ? (pythonCountedAttempts >= pythonAttemptLimit || pythonGreenBaselines >= pythonGreenTarget) : nonPythonCountedAttempts >= nonPythonAttemptLimit)
901
+ continue;
820
902
  const candidateTestRels = isPython
821
- ? pytestNodeidsForTarget(sourceRoot, testRel, testName).filter((candidate) => isRunnableTestForTarget(node, candidate))
903
+ ? (hard && testName && PYTHON_TEST_NODEID_SUFFIX_RE.test(testName) ? [`${testRel}::${testName}`] : []).filter((candidate) => isRunnableTestForTarget(node, candidate))
822
904
  : [testRel];
823
905
  if (candidateTestRels.length === 0)
824
906
  continue;
@@ -826,6 +908,15 @@ function proveExistingAssociatedTests(root, graph, sourceRoot, nodeById, opts, p
826
908
  const blockedKey = projectBlockKeys(targetLanguageName, runnerRoot, targetFileRel).find((key) => projectBlocked.has(key));
827
909
  const blocked = blockedKey ? projectBlocked.get(blockedKey) : undefined;
828
910
  if (blocked) {
911
+ // Python preflight is once per project root. The first classified failure
912
+ // is already persisted with full diagnostics; siblings are skipped without
913
+ // becoming fake attempts or thousands of duplicate sidecar rows.
914
+ if (isPython) {
915
+ const key = blockedKey;
916
+ const prior = pythonBlockedCounts.get(key);
917
+ pythonBlockedCounts.set(key, { count: (prior?.count ?? 0) + 1, block: blocked, root: runnerRoot });
918
+ continue;
919
+ }
829
920
  const attempt = dedupedAttempt(symId, proofTestRel, runnerRoot, blocked);
830
921
  attempts.push(attempt);
831
922
  needsSetup.push(attempt);
@@ -852,6 +943,8 @@ function proveExistingAssociatedTests(root, graph, sourceRoot, nodeById, opts, p
852
943
  continue;
853
944
  }
854
945
  attempted++;
946
+ if (!isPython)
947
+ nonPythonCountedAttempts++;
855
948
  // R-2: inject NODE_OPTIONS=--experimental-sqlite via the existing test_env path when the
856
949
  // target references node:sqlite. Makes the baseline runnable only; never mints Proven.
857
950
  const testEnv = isGo || isJava || isPython ? undefined : experimentalSqliteTestEnv(reader, targetFileRel);
@@ -868,7 +961,7 @@ function proveExistingAssociatedTests(root, graph, sourceRoot, nodeById, opts, p
868
961
  let result;
869
962
  try {
870
963
  result = proveLoop(root, nativeTestRun
871
- ? { target_symbol: symId, source: sourceRoot, test_run: nativeTestRun, ...(goAssertionLine !== undefined ? { go_assertion_line: goAssertionLine } : {}), run_id: `auto-prove-existing-${attempted}` }
964
+ ? { target_symbol: symId, source: sourceRoot, test_run: nativeTestRun, ...(goAssertionLine !== undefined ? { go_assertion_line: goAssertionLine } : {}), run_id: `auto-prove-${attempted}` }
872
965
  // link_node_modules: the isolated proof copy excludes node_modules; without linking,
873
966
  // any target/test importing a repo dependency fails baseline → needs_setup. Linking only
874
967
  // makes real tests runnable — Proven still requires the dynamic oracle's sentinel kill.
@@ -878,23 +971,29 @@ function proveExistingAssociatedTests(root, graph, sourceRoot, nodeById, opts, p
878
971
  test_path: proofTestRel,
879
972
  replacement: replacementForTarget(targetFileRel),
880
973
  link_node_modules: true,
974
+ ...(isPython ? { skip_preflight: pythonPreflightGreen.has(runnerRoot) } : {}),
881
975
  ...(testEnv ? { test_env: testEnv } : {}),
882
- run_id: `auto-prove-existing-${attempted}`
976
+ run_id: `auto-prove-${attempted}`
883
977
  }, proveDeps);
884
978
  }
885
979
  catch (e) {
886
- const attempt = {
887
- target_symbol: symId,
888
- test_path: displayTest,
889
- classification: "needs_setup",
890
- reason: `Proof could not run: ${redactSecrets(errMsg(e))}`,
891
- project_root: runnerRoot
892
- };
893
- attempts.push(attempt);
894
- needsSetup.push(attempt);
980
+ attempted--;
981
+ if (!isPython)
982
+ nonPythonCountedAttempts--;
983
+ reportProgress(`proof skipped: ${symId}: ${redactSecrets(errMsg(e))}`);
895
984
  continue;
896
985
  }
897
986
  const { classification, reason, category } = classifyProof(result, { sourceRoot, targetFileRel });
987
+ const baselineGreen = baselineGreenOf(result);
988
+ if (isPython) {
989
+ const preflightUnavailable = category === "environment_unavailable" || category === "collection_error";
990
+ if (!preflightUnavailable) {
991
+ pythonCountedAttempts++;
992
+ pythonPreflightGreen.add(runnerRoot);
993
+ }
994
+ if (baselineGreen)
995
+ pythonGreenBaselines++;
996
+ }
898
997
  const attempt = {
899
998
  target_symbol: symId,
900
999
  test_path: displayTest,
@@ -902,7 +1001,8 @@ function proveExistingAssociatedTests(root, graph, sourceRoot, nodeById, opts, p
902
1001
  reason,
903
1002
  category,
904
1003
  project_root: runnerRoot,
905
- mutant_status: mutantStatusOf(result)
1004
+ mutant_status: mutantStatusOf(result),
1005
+ ...proofDiagnosticsOf(result)
906
1006
  };
907
1007
  attempts.push(attempt);
908
1008
  if (classification === "proven") {
@@ -928,7 +1028,15 @@ function proveExistingAssociatedTests(root, graph, sourceRoot, nodeById, opts, p
928
1028
  }
929
1029
  // non_killing → keep trying this symbol's other associated tests, if any.
930
1030
  }
931
- return { attempts, needsSetup, proven, attempted, provenSymbols };
1031
+ const skipped = [...pythonBlockedCounts.values()].map(({ count, block, root }) => ({
1032
+ title: `Python project ${root}`,
1033
+ reason: `${count} remaining linked target${count === 1 ? " was" : "s were"} in a project root marked ${block.category}; no further runner calls were made in that root.`,
1034
+ category: block.category,
1035
+ language: "python",
1036
+ project_root: root,
1037
+ blocked_by: block.blockedBy
1038
+ }));
1039
+ return { attempts, needsSetup, skipped, proven, attempted, nonPythonAttempted: nonPythonCountedAttempts, provenSymbols };
932
1040
  }
933
1041
  /**
934
1042
  * Drive prove for the top provable TS/JS targets. Two lanes: (1) prove the repo's OWN
@@ -967,13 +1075,14 @@ export async function autoProve(root, opts, deps) {
967
1075
  // Keyed by language + bounded runner root; never populated by a test-specific red
968
1076
  // baseline, an unknown failure, or a surviving mutant.
969
1077
  const projectBlocked = new Map();
970
- // ONE unified dynamic-proof budget for the whole pass (existing-first → then generation).
971
- // Default 5 ("dynamically prove top 5"); `--auto-limit N` overrides it, clamped to
972
- // MAX_AUTO_LIMIT. The existing lane consumes from this budget and the generation lane gets
973
- // only the remainder, so TOTAL attempts (existing + generation) are ≤ budget.
1078
+ // R7.1 adds a Python-only existing-test retry lane. Generic/non-Python calls retain
1079
+ // the released `autoLimit`; Python can inspect more existing targets to find
1080
+ // the requested number of baseline-green tests without widening any other language.
974
1081
  const autoLimit = Math.max(1, Math.min(MAX_AUTO_LIMIT, Math.floor(opts.autoLimit ?? DEFAULT_AUTO_LIMIT)));
1082
+ const existingLimit = Math.max(1, Math.min(MAX_AUTO_LIMIT, Math.floor(opts.proof_existing_limit ?? DEFAULT_PROOF_EXISTING_LIMIT)));
1083
+ const genLimit = Math.max(1, Math.min(MAX_AUTO_LIMIT, Math.floor(opts.proof_gen_limit ?? DEFAULT_PROOF_GEN_LIMIT)));
975
1084
  // ── Lane 1: existing associated tests — NO key required, runs FIRST (PR 1.5). ──
976
- const ex = proveExistingAssociatedTests(root, graph, sourceRoot, nodeById, opts, proveLoop, proveDeps, alreadyProven, projectBlocked, autoLimit);
1085
+ const ex = proveExistingAssociatedTests(root, graph, sourceRoot, nodeById, opts, proveLoop, proveDeps, alreadyProven, projectBlocked, existingLimit, genLimit, autoLimit);
977
1086
  if (opts.existingOnly) {
978
1087
  const status = ex.proven > 0 ? "proven-run" : ex.attempted > 0 ? "ran-no-proof" : "no-targets";
979
1088
  return {
@@ -983,7 +1092,7 @@ export async function autoProve(root, opts, deps) {
983
1092
  attempted: ex.attempted,
984
1093
  proven: ex.proven,
985
1094
  needs_setup: ex.needsSetup,
986
- skipped: [],
1095
+ skipped: ex.skipped,
987
1096
  generated_files: [],
988
1097
  attempts: ex.attempts
989
1098
  };
@@ -1002,7 +1111,7 @@ export async function autoProve(root, opts, deps) {
1002
1111
  attempted: ex.attempted,
1003
1112
  proven: ex.proven,
1004
1113
  needs_setup: ex.needsSetup,
1005
- skipped: [],
1114
+ skipped: ex.skipped,
1006
1115
  generated_files: [],
1007
1116
  attempts: ex.attempts
1008
1117
  };
@@ -1010,9 +1119,9 @@ export async function autoProve(root, opts, deps) {
1010
1119
  const provider = deterministic ? new DeterministicProvider() : buildProvider(providerConfig);
1011
1120
  const generate = deps.generate ?? generateTests;
1012
1121
  const reader = fileReaderFor(sourceRoot);
1013
- // Generation gets only the budget the existing-tests lane left unspent, so existing +
1014
- // generation attempts total ≤ autoLimit. Exhausted budget ⇒ genBudget 0 ⇒ no provider call.
1015
- const genBudget = Math.max(0, autoLimit - ex.attempted);
1122
+ // Generated proofs retain the released generic limit. Python's green-target cap
1123
+ // applies only to existing-test retries above, not to non-Python generation.
1124
+ const generationLimit = Math.max(0, autoLimit - ex.nonPythonAttempted);
1016
1125
  // Candidates = ORS-ranked provable CodeSymbols. rankRiskGaps ranks by OrangePro Risk
1017
1126
  // Score and excludes hard-confirmed symbols; we ALSO enforce the eligibility barrier
1018
1127
  // explicitly (isEligibleProvableTarget) at selection so an excluded infra symbol can
@@ -1034,16 +1143,17 @@ export async function autoProve(root, opts, deps) {
1034
1143
  const declaredDeps = readDeclaredDeps(sourceRoot);
1035
1144
  let proven = 0;
1036
1145
  let attempted = 0;
1037
- for (let start = 0; start < candidates.length && attempted < genBudget; start += GEN_WINDOW) {
1146
+ let baselineGreenTargets = 0;
1147
+ for (let start = 0; start < candidates.length && attempted < autoLimit && baselineGreenTargets < generationLimit; start += GEN_WINDOW) {
1038
1148
  const window = candidates.slice(start, start + GEN_WINDOW);
1039
- const need = genBudget - attempted;
1149
+ const need = Math.max(1, Math.min(GEN_WINDOW, generationLimit - baselineGreenTargets));
1040
1150
  const windowIds = window.map((g) => g.id);
1041
1151
  const gen = await generate(graph, { target_ids: windowIds, limit: Math.min(windowIds.length, need), ...(opts.prompt_version ? { prompt_version: opts.prompt_version } : {}) }, provider, reader, clock);
1042
1152
  const tests = gen.generated_tests;
1043
1153
  // Global start offset so filenames stay unique across windows — runHintsFor
1044
1154
  // otherwise resets its index to 0 per window and same-slug targets collide.
1045
1155
  const hints = runHintsFor(tests, sourceRoot, start);
1046
- for (let i = 0; i < tests.length && attempted < genBudget; i++) {
1156
+ for (let i = 0; i < tests.length && attempted < autoLimit && baselineGreenTargets < generationLimit; i++) {
1047
1157
  const test = tests[i];
1048
1158
  const hint = hints[i];
1049
1159
  if (!hint.prove_run) {
@@ -1141,19 +1251,12 @@ export async function autoProve(root, opts, deps) {
1141
1251
  // See lane 1: link node_modules so a generated test importing a repo dep can boot.
1142
1252
  link_node_modules: true,
1143
1253
  ...(testEnv ? { test_env: testEnv } : {}),
1144
- run_id: `auto-prove-${start + i + 1}`
1254
+ run_id: `auto-prove-${ex.attempted + attempted}`
1145
1255
  }, proveDeps);
1146
1256
  }
1147
1257
  catch (e) {
1148
- const attempt = {
1149
- target_symbol,
1150
- test_path: writeRel,
1151
- classification: "needs_setup",
1152
- reason: `Proof could not run: ${redactSecrets(errMsg(e))}`,
1153
- project_root: projectRoot
1154
- };
1155
- attempts.push(attempt);
1156
- needsSetup.push(attempt);
1258
+ attempted--;
1259
+ skipped.push({ target_symbol, title: test.title, reason: `Proof could not run: ${redactSecrets(errMsg(e))}`, language, project_root: projectRoot });
1157
1260
  continue;
1158
1261
  }
1159
1262
  const { classification, reason, category } = classifyProof(result, { sourceRoot, targetFileRel });
@@ -1164,9 +1267,12 @@ export async function autoProve(root, opts, deps) {
1164
1267
  reason,
1165
1268
  category,
1166
1269
  project_root: projectRoot,
1167
- mutant_status: mutantStatusOf(result)
1270
+ mutant_status: mutantStatusOf(result),
1271
+ ...proofDiagnosticsOf(result)
1168
1272
  };
1169
1273
  attempts.push(attempt);
1274
+ if (baselineGreenOf(result))
1275
+ baselineGreenTargets++;
1170
1276
  if (classification === "proven")
1171
1277
  proven++;
1172
1278
  else if (classification === "needs_setup") {
@@ -1199,7 +1305,7 @@ export async function autoProve(root, opts, deps) {
1199
1305
  attempted: totalAttempted,
1200
1306
  proven: totalProven,
1201
1307
  needs_setup: [...ex.needsSetup, ...needsSetup],
1202
- skipped,
1308
+ skipped: [...ex.skipped, ...skipped],
1203
1309
  generated_files: generatedFiles,
1204
1310
  attempts: [...ex.attempts, ...attempts]
1205
1311
  };