@orangepro/orangepro-mcp 0.2.44 → 0.2.46
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/local/analyze/analyzer.js +322 -30
- package/dist/local/analyze/treeSitter/engine.js +1126 -90
- package/dist/local/autoProve.js +208 -102
- package/dist/local/operations.js +82 -18
- package/dist/local/proofDoctor.js +23 -3
- package/dist/local/provenance.js +2 -2
- package/dist/local/score/risk.js +339 -111
- package/dist/local/score/riskConfig.js +21 -1
- package/dist/local/viz/behaviorReportData.js +83 -20
- package/dist/local/viz/behaviorReportHtml.js +13 -2
- package/package.json +1 -1
- package/scripts/spikes/python-dynamic-proof-spike.mjs +399 -103
- package/scripts/spikes/python-mutate.py +72 -24
package/dist/local/autoProve.js
CHANGED
|
@@ -21,7 +21,7 @@ import { resolveProviderConfig } from "./localConfig.js";
|
|
|
21
21
|
import { buildProvider, DeterministicProvider } from "./generate/providers.js";
|
|
22
22
|
import { resolveContained } from "./reprove/paths.js";
|
|
23
23
|
import { buildRtm } from "./rtm.js";
|
|
24
|
-
import { loadLedger, targetLanguage } from "./ledger.js";
|
|
24
|
+
import { loadLedger, proofEdgesFor, targetLanguage } from "./ledger.js";
|
|
25
25
|
import { reportProgress } from "./util/progress.js";
|
|
26
26
|
import { loadGraph, workspacePaths } from "./workspace.js";
|
|
27
27
|
import { systemClock } from "./util/time.js";
|
|
@@ -82,7 +82,7 @@ function pytestNodeidsForFile(sourceRoot, testRel) {
|
|
|
82
82
|
}
|
|
83
83
|
if (currentClass && indent <= classIndent && line.trim() !== "" && !line.startsWith(" "))
|
|
84
84
|
currentClass = null;
|
|
85
|
-
const fnMatch = /^(\s*)def\s+(test_[A-Za-z0-9_]*)\s*\(/.exec(line);
|
|
85
|
+
const fnMatch = /^(\s*)(?:async\s+)?def\s+(test_[A-Za-z0-9_]*)\s*\(/.exec(line);
|
|
86
86
|
if (!fnMatch)
|
|
87
87
|
continue;
|
|
88
88
|
const fnIndent = fnMatch[1].length;
|
|
@@ -358,13 +358,12 @@ export function javaTestForTarget(nodeById, graph, symId) {
|
|
|
358
358
|
return selectors.size === 1 ? [...selectors][0] : null;
|
|
359
359
|
}
|
|
360
360
|
const GENERATED_HEADER = "// Generated by OrangePro — do not edit";
|
|
361
|
-
//
|
|
362
|
-
//
|
|
363
|
-
//
|
|
364
|
-
// overrides it (clamped to MAX_AUTO_LIMIT) for deeper runs (`opro start --auto-limit 25`,
|
|
365
|
-
// `opro prove`, `opro prove-loop`). Static breadth (behaviors/flows/associated/risk) is
|
|
366
|
-
// never gated on this budget — only the dynamic verification pass is.
|
|
361
|
+
// Released v0.2.44 generic dynamic-proof limit. Python's existing-test lane has its
|
|
362
|
+
// own retry limit below; do not widen this generic default or TS/JS, Go, Java, and
|
|
363
|
+
// generated-test scheduling changes before a caller opts in.
|
|
367
364
|
const DEFAULT_AUTO_LIMIT = 5;
|
|
365
|
+
const DEFAULT_PROOF_EXISTING_LIMIT = 20;
|
|
366
|
+
const DEFAULT_PROOF_GEN_LIMIT = 5;
|
|
368
367
|
const MAX_AUTO_LIMIT = 50;
|
|
369
368
|
/** Generator caps a single call at 5 (MAX_LIMIT); page candidates in windows of that. */
|
|
370
369
|
const GEN_WINDOW = 5;
|
|
@@ -381,10 +380,10 @@ const RUNNER_ROOT_MARKERS = {
|
|
|
381
380
|
rs: ["Cargo.toml"]
|
|
382
381
|
};
|
|
383
382
|
/**
|
|
384
|
-
* The nearest language runner root
|
|
385
|
-
*
|
|
386
|
-
*
|
|
387
|
-
*
|
|
383
|
+
* The nearest language runner root, bounded by the analyzed source root. Python is
|
|
384
|
+
* test-owned: pytest configuration/environment follows the selected test file even when
|
|
385
|
+
* that test exercises a target elsewhere in the same repository. Other languages keep
|
|
386
|
+
* their target-owned v0.2.44 behavior.
|
|
388
387
|
*/
|
|
389
388
|
export function proofRunnerRoot(sourceRoot, targetRel, testRel) {
|
|
390
389
|
const stop = resolve(sourceRoot);
|
|
@@ -393,7 +392,11 @@ export function proofRunnerRoot(sourceRoot, targetRel, testRel) {
|
|
|
393
392
|
return ".";
|
|
394
393
|
const language = targetLanguage(`sym:${targetRel}#target`);
|
|
395
394
|
const markers = RUNNER_ROOT_MARKERS[language] ?? [];
|
|
396
|
-
|
|
395
|
+
const testFile = testRel?.split("::", 1)[0] ?? testRel;
|
|
396
|
+
const testAbs = testFile ? resolve(stop, testFile) : undefined;
|
|
397
|
+
if (language === "python" && testAbs && testAbs !== stop && !testAbs.startsWith(stop + sep))
|
|
398
|
+
return ".";
|
|
399
|
+
let dir = dirname(language === "python" && testAbs ? testAbs : targetAbs);
|
|
397
400
|
let found = stop;
|
|
398
401
|
for (;;) {
|
|
399
402
|
if (markers.some((marker) => existsSync(join(dir, marker)))) {
|
|
@@ -407,13 +410,6 @@ export function proofRunnerRoot(sourceRoot, targetRel, testRel) {
|
|
|
407
410
|
break;
|
|
408
411
|
dir = parent;
|
|
409
412
|
}
|
|
410
|
-
if (testRel && found !== stop) {
|
|
411
|
-
const testFile = testRel.split("::", 1)[0] ?? testRel;
|
|
412
|
-
const testAbs = resolve(stop, testFile);
|
|
413
|
-
if ((testAbs !== stop && !testAbs.startsWith(stop + sep))
|
|
414
|
-
|| (testAbs !== found && !testAbs.startsWith(found + sep)))
|
|
415
|
-
found = stop;
|
|
416
|
-
}
|
|
417
413
|
const rel = relative(stop, found).split(sep).join("/");
|
|
418
414
|
return rel || ".";
|
|
419
415
|
}
|
|
@@ -438,7 +434,9 @@ const PROJECT_WIDE_BLOCKERS = new Set([
|
|
|
438
434
|
"tsconfig_missing",
|
|
439
435
|
"runner_missing",
|
|
440
436
|
"module_root_missing",
|
|
441
|
-
"go_package_build_failure"
|
|
437
|
+
"go_package_build_failure",
|
|
438
|
+
"environment_unavailable",
|
|
439
|
+
"collection_error"
|
|
442
440
|
]);
|
|
443
441
|
export function isRoastSurvivor(attempt) {
|
|
444
442
|
return attempt.classification === "non_killing" && attempt.mutant_status === "associated_survived";
|
|
@@ -513,6 +511,14 @@ function classifyProof(result, ctx) {
|
|
|
513
511
|
const record = dyn.record;
|
|
514
512
|
if (record.closed)
|
|
515
513
|
return { classification: "proven" };
|
|
514
|
+
// Language spikes may refuse before a baseline exists (for example, Python's
|
|
515
|
+
// collection preflight, an explicit project boundary, or a type-unsafe mutation).
|
|
516
|
+
// These are structured runner outcomes, not a raw-stderr heuristic. Never mint a
|
|
517
|
+
// proof and never mislabel a mutation refusal as a surviving test.
|
|
518
|
+
const runnerCategory = dyn.oracle.category;
|
|
519
|
+
if (runnerCategory === "environment_unavailable" || runnerCategory === "collection_error" || runnerCategory === "project_boundary" || runnerCategory === "mutation_unsupported") {
|
|
520
|
+
return { classification: "needs_setup", reason: dyn.oracle.reason, category: runnerCategory };
|
|
521
|
+
}
|
|
516
522
|
const cert = record.dynamic_proof;
|
|
517
523
|
if (cert && cert.baseline_green === false) {
|
|
518
524
|
const failureSummary = dyn.oracle.baseline?.failureSummary;
|
|
@@ -538,6 +544,29 @@ function mutantStatusOf(result) {
|
|
|
538
544
|
return "unrunnable";
|
|
539
545
|
return result.record.dynamic_proof?.mutant_status;
|
|
540
546
|
}
|
|
547
|
+
/** A baseline-green target consumes the conservative generated-proof target quota. */
|
|
548
|
+
function baselineGreenOf(result) {
|
|
549
|
+
if ("status" in result && result.status === "unrunnable")
|
|
550
|
+
return false;
|
|
551
|
+
return result.record.dynamic_proof?.baseline_green === true;
|
|
552
|
+
}
|
|
553
|
+
function proofDiagnosticsOf(result) {
|
|
554
|
+
if ("status" in result && result.status === "unrunnable")
|
|
555
|
+
return {};
|
|
556
|
+
const cert = result.record.dynamic_proof;
|
|
557
|
+
if (!cert?.command || !cert.cwd || cert.duration_ms === undefined)
|
|
558
|
+
return {};
|
|
559
|
+
return {
|
|
560
|
+
command: cert.command,
|
|
561
|
+
cwd: cert.cwd,
|
|
562
|
+
exit_code: cert.exit_code,
|
|
563
|
+
duration_ms: cert.duration_ms,
|
|
564
|
+
failure_class: cert.failure_class,
|
|
565
|
+
stdout_tail: cert.stdout_tail,
|
|
566
|
+
stderr_tail: cert.stderr_tail,
|
|
567
|
+
runner_fallback: cert.runner_fallback
|
|
568
|
+
};
|
|
569
|
+
}
|
|
541
570
|
function projectBlockKey(language, projectRoot) {
|
|
542
571
|
return `${language}\u0000${projectRoot}`;
|
|
543
572
|
}
|
|
@@ -690,15 +719,56 @@ export function existingAssociatedTests(graph, nodeById) {
|
|
|
690
719
|
add(symId, testRel, false);
|
|
691
720
|
}
|
|
692
721
|
}
|
|
722
|
+
// Python proof must have an exact hard edge carrying a test selector. A weak
|
|
723
|
+
// file relation (or an empty pre-edge set) is not a runnable assertion link.
|
|
724
|
+
for (const [symId, tests] of out) {
|
|
725
|
+
if (!isPythonFile(symId.split("#")[0].slice(4)))
|
|
726
|
+
continue;
|
|
727
|
+
const preEdges = new Set(proofEdgesFor(graph, symId));
|
|
728
|
+
const linked = tests.filter((t) => t.hard && !!t.testName && PYTHON_TEST_NODEID_SUFFIX_RE.test(t.testName) && preEdges.has(`test:${t.test}->${symId}`));
|
|
729
|
+
if (linked.length)
|
|
730
|
+
out.set(symId, linked);
|
|
731
|
+
else
|
|
732
|
+
out.delete(symId);
|
|
733
|
+
}
|
|
693
734
|
return out;
|
|
694
735
|
}
|
|
736
|
+
function declaredPytestTestpaths(sourceRoot, projectRoot) {
|
|
737
|
+
const declared = new Set();
|
|
738
|
+
const root = resolve(sourceRoot, projectRoot);
|
|
739
|
+
for (const config of ["pyproject.toml", "pytest.ini", ".pytest.ini", "tox.ini", "setup.cfg"]) {
|
|
740
|
+
let text = "";
|
|
741
|
+
try {
|
|
742
|
+
text = readFileSync(join(root, config), "utf8");
|
|
743
|
+
}
|
|
744
|
+
catch {
|
|
745
|
+
continue;
|
|
746
|
+
}
|
|
747
|
+
const testpaths = /(?:^|\n)\s*testpaths\s*=\s*([^\n]*(?:\n(?:\s+[^\n#;][^\n]*)?)*)/im.exec(text)?.[1];
|
|
748
|
+
if (!testpaths)
|
|
749
|
+
continue;
|
|
750
|
+
for (const path of testpaths.match(/['"][^'"]+['"]|[^\s,\[\]]+/g) ?? []) {
|
|
751
|
+
const normalized = path.trim().replace(/^['"]|['"]$/g, "").replace(/^\.\//, "").replace(/\/+$/, "");
|
|
752
|
+
if (normalized)
|
|
753
|
+
declared.add(normalized);
|
|
754
|
+
}
|
|
755
|
+
}
|
|
756
|
+
return declared;
|
|
757
|
+
}
|
|
758
|
+
function isInDeclaredPytestTestpath(sourceRoot, projectRoot, testRel, testpaths) {
|
|
759
|
+
const file = testRel.split("::", 1)[0] ?? testRel;
|
|
760
|
+
const projectAbs = resolve(sourceRoot, projectRoot);
|
|
761
|
+
const testAbs = resolve(sourceRoot, file);
|
|
762
|
+
if (testAbs !== projectAbs && !testAbs.startsWith(projectAbs + sep))
|
|
763
|
+
return false;
|
|
764
|
+
const projectTestRel = relative(projectAbs, testAbs).split(sep).join("/");
|
|
765
|
+
return [...testpaths].some((path) => projectTestRel === path || projectTestRel.startsWith(`${path}/`));
|
|
766
|
+
}
|
|
695
767
|
/**
|
|
696
|
-
*
|
|
697
|
-
*
|
|
698
|
-
*
|
|
699
|
-
*
|
|
700
|
-
* the budget. Order within each tier follows the Map's insertion order (graph node/edge order),
|
|
701
|
-
* so the result is deterministic.
|
|
768
|
+
* Deterministic hard-first, weak-capped attempt order. Python is the sole exception to the
|
|
769
|
+
* legacy source order: its existing-test retries are ordered by the unpersisted ORS worklist
|
|
770
|
+
* (including Associated symbols), then direct hard-edge tests, declared pytest `testpaths`, and
|
|
771
|
+
* original source order. Non-Python entries retain the v0.2.44 queue exactly.
|
|
702
772
|
*/
|
|
703
773
|
export function orderExistingAttempts(testsBySymbol, maxWeakPerSymbol = EXISTING_LANE_MAX_WEAK_PER_SYMBOL, schedule) {
|
|
704
774
|
const hard = [];
|
|
@@ -714,33 +784,53 @@ export function orderExistingAttempts(testsBySymbol, maxWeakPerSymbol = EXISTING
|
|
|
714
784
|
}
|
|
715
785
|
if (!schedule)
|
|
716
786
|
return [...hard, ...weak];
|
|
717
|
-
const
|
|
718
|
-
const
|
|
787
|
+
const rankedPython = rankRiskGaps(schedule.graph, { repoRoot: schedule.sourceRoot, limit: 500, includeAssociated: true });
|
|
788
|
+
const pythonRank = new Map(rankedPython.map((gap, index) => [gap.id, { score: gap.risk_score, rank: index }]));
|
|
719
789
|
const decorate = (attempt, index) => {
|
|
720
790
|
const node = schedule.nodeById.get(attempt.symId);
|
|
721
791
|
const targetRel = node ? symbolFileOf(node) : attempt.symId.replace(/^sym:/, "").split("#")[0];
|
|
792
|
+
const python = isPythonFile(targetRel);
|
|
793
|
+
const risk = pythonRank.get(attempt.symId);
|
|
794
|
+
const projectRoot = proofRunnerRoot(schedule.sourceRoot, targetRel, attempt.testRel);
|
|
795
|
+
const pytestTestpaths = python ? declaredPytestTestpaths(schedule.sourceRoot, projectRoot) : new Set();
|
|
722
796
|
return {
|
|
723
797
|
...attempt,
|
|
724
798
|
language: targetLanguage(attempt.symId),
|
|
725
|
-
projectRoot
|
|
726
|
-
index
|
|
799
|
+
projectRoot,
|
|
800
|
+
index,
|
|
801
|
+
python,
|
|
802
|
+
score: risk?.score ?? Number.NEGATIVE_INFINITY,
|
|
803
|
+
rank: risk?.rank ?? Number.MAX_SAFE_INTEGER,
|
|
804
|
+
declaredTestpath: python && isInDeclaredPytestTestpath(schedule.sourceRoot, projectRoot, attempt.testRel, pytestTestpaths)
|
|
727
805
|
};
|
|
728
806
|
};
|
|
729
|
-
const
|
|
807
|
+
const pythonOrsOrder = (attempts) => {
|
|
730
808
|
const decorated = attempts.map(decorate);
|
|
731
|
-
const firstProject = new Map();
|
|
732
|
-
for (const item of decorated) {
|
|
733
|
-
const key = `${item.language}\u0000${item.projectRoot}`;
|
|
734
|
-
if (!firstProject.has(key))
|
|
735
|
-
firstProject.set(key, firstProject.size);
|
|
736
|
-
}
|
|
737
809
|
return decorated
|
|
738
|
-
.sort((a, b) =>
|
|
739
|
-
|
|
810
|
+
.sort((a, b) =>
|
|
811
|
+
// Python ORS is primary. Direct hard and pytest-config testpaths are strictly
|
|
812
|
+
// equal-ORS tie breakers; project grouping deliberately has no precedence.
|
|
813
|
+
(b.score - a.score)
|
|
814
|
+
|| Number(b.hard) - Number(a.hard)
|
|
815
|
+
|| Number(b.declaredTestpath) - Number(a.declaredTestpath)
|
|
740
816
|
|| (a.index - b.index))
|
|
741
|
-
.map(({ index: _index, ...attempt }) => attempt);
|
|
817
|
+
.map(({ index: _index, python: _python, score: _score, rank: _rank, declaredTestpath: _declaredTestpath, ...attempt }) => attempt);
|
|
742
818
|
};
|
|
743
|
-
|
|
819
|
+
const queue = [...hard, ...weak];
|
|
820
|
+
const python = pythonOrsOrder(queue.filter((attempt) => {
|
|
821
|
+
const node = schedule.nodeById.get(attempt.symId);
|
|
822
|
+
return isPythonFile(node ? symbolFileOf(node) : attempt.symId.replace(/^sym:/, "").split("#")[0]);
|
|
823
|
+
}));
|
|
824
|
+
if (python.length === 0)
|
|
825
|
+
return queue;
|
|
826
|
+
let nextPython = 0;
|
|
827
|
+
const scheduled = queue.map((attempt) => {
|
|
828
|
+
const node = schedule.nodeById.get(attempt.symId);
|
|
829
|
+
const targetRel = node ? symbolFileOf(node) : attempt.symId.replace(/^sym:/, "").split("#")[0];
|
|
830
|
+
return isPythonFile(targetRel) ? python[nextPython++] : attempt;
|
|
831
|
+
});
|
|
832
|
+
const primary = proofLanguageOrder(schedule.graph)[0];
|
|
833
|
+
return primary ? scheduled.sort((a, b) => Number(targetLanguage(b.symId) === primary) - Number(targetLanguage(a.symId) === primary)) : scheduled;
|
|
744
834
|
}
|
|
745
835
|
/**
|
|
746
836
|
* Stable scheduling view over the existing ORS-ranked candidate list. This changes
|
|
@@ -748,24 +838,11 @@ export function orderExistingAttempts(testsBySymbol, maxWeakPerSymbol = EXISTING
|
|
|
748
838
|
* evidence tiers, or the persisted priority-gap order.
|
|
749
839
|
*/
|
|
750
840
|
export function orderRankedProofCandidates(candidates, graph, sourceRoot) {
|
|
751
|
-
|
|
752
|
-
|
|
753
|
-
|
|
754
|
-
|
|
755
|
-
|
|
756
|
-
const decorated = candidates.map((candidate, index) => {
|
|
757
|
-
const language = targetLanguage(candidate.id);
|
|
758
|
-
const projectRoot = proofRunnerRoot(sourceRoot, candidate.file);
|
|
759
|
-
const projectKey = `${language}\u0000${projectRoot}`;
|
|
760
|
-
if (!firstProject.has(projectKey))
|
|
761
|
-
firstProject.set(projectKey, firstProject.size);
|
|
762
|
-
return { candidate, index, language, projectKey };
|
|
763
|
-
});
|
|
764
|
-
return decorated
|
|
765
|
-
.sort((a, b) => ((languageRank.get(a.language) ?? Number.MAX_SAFE_INTEGER) - (languageRank.get(b.language) ?? Number.MAX_SAFE_INTEGER))
|
|
766
|
-
|| ((firstProject.get(a.projectKey) ?? 0) - (firstProject.get(b.projectKey) ?? 0))
|
|
767
|
-
|| (a.index - b.index))
|
|
768
|
-
.map(({ candidate }) => candidate);
|
|
841
|
+
// Generation was v0.2.44 source-ranking order. Python retries are scheduled separately
|
|
842
|
+
// in `orderExistingAttempts`; leave this queue untouched for every language.
|
|
843
|
+
void graph;
|
|
844
|
+
void sourceRoot;
|
|
845
|
+
return candidates;
|
|
769
846
|
}
|
|
770
847
|
/**
|
|
771
848
|
* PR 1.5 lane — prove the repo's OWN existing tests, NO provider key. For each eligible
|
|
@@ -779,12 +856,17 @@ export function orderRankedProofCandidates(candidates, graph, sourceRoot) {
|
|
|
779
856
|
* — the generation lane gets whatever this lane leaves unspent, so TOTAL attempts
|
|
780
857
|
* (existing + generation) never exceed the budget.
|
|
781
858
|
*/
|
|
782
|
-
function proveExistingAssociatedTests(root, graph, sourceRoot, nodeById, opts, proveLoop, proveDeps, alreadyProven, projectBlocked,
|
|
859
|
+
function proveExistingAssociatedTests(root, graph, sourceRoot, nodeById, opts, proveLoop, proveDeps, alreadyProven, projectBlocked, pythonAttemptLimit, pythonGreenTarget, nonPythonAttemptLimit) {
|
|
783
860
|
const attempts = [];
|
|
784
861
|
const needsSetup = [];
|
|
785
862
|
const provenSymbols = new Set();
|
|
786
863
|
let proven = 0;
|
|
787
864
|
let attempted = 0;
|
|
865
|
+
let pythonCountedAttempts = 0;
|
|
866
|
+
let pythonGreenBaselines = 0;
|
|
867
|
+
let nonPythonCountedAttempts = 0;
|
|
868
|
+
const pythonPreflightGreen = new Set();
|
|
869
|
+
const pythonBlockedCounts = new Map();
|
|
788
870
|
const changed = opts.changedFiles && opts.changedFiles.length > 0 ? new Set(opts.changedFiles) : null;
|
|
789
871
|
const reader = fileReaderFor(sourceRoot); // R-2: source scan for the node:sqlite env profile
|
|
790
872
|
// Fix 3: hard TESTED_BY/COVERS pairs first, weak MAY_* pairs after and capped per symbol.
|
|
@@ -794,8 +876,6 @@ function proveExistingAssociatedTests(root, graph, sourceRoot, nodeById, opts, p
|
|
|
794
876
|
sourceRoot
|
|
795
877
|
});
|
|
796
878
|
for (const { symId, testRel, hard, testName, language, projectRoot } of queue) {
|
|
797
|
-
if (attempted >= budget)
|
|
798
|
-
break;
|
|
799
879
|
const node = nodeById.get(symId);
|
|
800
880
|
// Redundant with existingAssociatedTests' own filter, but the eligibility barrier is
|
|
801
881
|
// the sole guard against handing plumbing to the guard-less prove path — assert it here too.
|
|
@@ -817,8 +897,10 @@ function proveExistingAssociatedTests(root, graph, sourceRoot, nodeById, opts, p
|
|
|
817
897
|
// A prior candidate in this exact runner project hit a classified project-wide
|
|
818
898
|
// toolchain/environment failure. Preserve the budget and explain the skip.
|
|
819
899
|
const isPython = isPythonFile(targetFileRel);
|
|
900
|
+
if (isPython ? (pythonCountedAttempts >= pythonAttemptLimit || pythonGreenBaselines >= pythonGreenTarget) : nonPythonCountedAttempts >= nonPythonAttemptLimit)
|
|
901
|
+
continue;
|
|
820
902
|
const candidateTestRels = isPython
|
|
821
|
-
?
|
|
903
|
+
? (hard && testName && PYTHON_TEST_NODEID_SUFFIX_RE.test(testName) ? [`${testRel}::${testName}`] : []).filter((candidate) => isRunnableTestForTarget(node, candidate))
|
|
822
904
|
: [testRel];
|
|
823
905
|
if (candidateTestRels.length === 0)
|
|
824
906
|
continue;
|
|
@@ -826,6 +908,15 @@ function proveExistingAssociatedTests(root, graph, sourceRoot, nodeById, opts, p
|
|
|
826
908
|
const blockedKey = projectBlockKeys(targetLanguageName, runnerRoot, targetFileRel).find((key) => projectBlocked.has(key));
|
|
827
909
|
const blocked = blockedKey ? projectBlocked.get(blockedKey) : undefined;
|
|
828
910
|
if (blocked) {
|
|
911
|
+
// Python preflight is once per project root. The first classified failure
|
|
912
|
+
// is already persisted with full diagnostics; siblings are skipped without
|
|
913
|
+
// becoming fake attempts or thousands of duplicate sidecar rows.
|
|
914
|
+
if (isPython) {
|
|
915
|
+
const key = blockedKey;
|
|
916
|
+
const prior = pythonBlockedCounts.get(key);
|
|
917
|
+
pythonBlockedCounts.set(key, { count: (prior?.count ?? 0) + 1, block: blocked, root: runnerRoot });
|
|
918
|
+
continue;
|
|
919
|
+
}
|
|
829
920
|
const attempt = dedupedAttempt(symId, proofTestRel, runnerRoot, blocked);
|
|
830
921
|
attempts.push(attempt);
|
|
831
922
|
needsSetup.push(attempt);
|
|
@@ -852,6 +943,8 @@ function proveExistingAssociatedTests(root, graph, sourceRoot, nodeById, opts, p
|
|
|
852
943
|
continue;
|
|
853
944
|
}
|
|
854
945
|
attempted++;
|
|
946
|
+
if (!isPython)
|
|
947
|
+
nonPythonCountedAttempts++;
|
|
855
948
|
// R-2: inject NODE_OPTIONS=--experimental-sqlite via the existing test_env path when the
|
|
856
949
|
// target references node:sqlite. Makes the baseline runnable only; never mints Proven.
|
|
857
950
|
const testEnv = isGo || isJava || isPython ? undefined : experimentalSqliteTestEnv(reader, targetFileRel);
|
|
@@ -868,7 +961,7 @@ function proveExistingAssociatedTests(root, graph, sourceRoot, nodeById, opts, p
|
|
|
868
961
|
let result;
|
|
869
962
|
try {
|
|
870
963
|
result = proveLoop(root, nativeTestRun
|
|
871
|
-
? { target_symbol: symId, source: sourceRoot, test_run: nativeTestRun, ...(goAssertionLine !== undefined ? { go_assertion_line: goAssertionLine } : {}), run_id: `auto-prove
|
|
964
|
+
? { target_symbol: symId, source: sourceRoot, test_run: nativeTestRun, ...(goAssertionLine !== undefined ? { go_assertion_line: goAssertionLine } : {}), run_id: `auto-prove-${attempted}` }
|
|
872
965
|
// link_node_modules: the isolated proof copy excludes node_modules; without linking,
|
|
873
966
|
// any target/test importing a repo dependency fails baseline → needs_setup. Linking only
|
|
874
967
|
// makes real tests runnable — Proven still requires the dynamic oracle's sentinel kill.
|
|
@@ -878,23 +971,29 @@ function proveExistingAssociatedTests(root, graph, sourceRoot, nodeById, opts, p
|
|
|
878
971
|
test_path: proofTestRel,
|
|
879
972
|
replacement: replacementForTarget(targetFileRel),
|
|
880
973
|
link_node_modules: true,
|
|
974
|
+
...(isPython ? { skip_preflight: pythonPreflightGreen.has(runnerRoot) } : {}),
|
|
881
975
|
...(testEnv ? { test_env: testEnv } : {}),
|
|
882
|
-
run_id: `auto-prove
|
|
976
|
+
run_id: `auto-prove-${attempted}`
|
|
883
977
|
}, proveDeps);
|
|
884
978
|
}
|
|
885
979
|
catch (e) {
|
|
886
|
-
|
|
887
|
-
|
|
888
|
-
|
|
889
|
-
|
|
890
|
-
reason: `Proof could not run: ${redactSecrets(errMsg(e))}`,
|
|
891
|
-
project_root: runnerRoot
|
|
892
|
-
};
|
|
893
|
-
attempts.push(attempt);
|
|
894
|
-
needsSetup.push(attempt);
|
|
980
|
+
attempted--;
|
|
981
|
+
if (!isPython)
|
|
982
|
+
nonPythonCountedAttempts--;
|
|
983
|
+
reportProgress(`proof skipped: ${symId}: ${redactSecrets(errMsg(e))}`);
|
|
895
984
|
continue;
|
|
896
985
|
}
|
|
897
986
|
const { classification, reason, category } = classifyProof(result, { sourceRoot, targetFileRel });
|
|
987
|
+
const baselineGreen = baselineGreenOf(result);
|
|
988
|
+
if (isPython) {
|
|
989
|
+
const preflightUnavailable = category === "environment_unavailable" || category === "collection_error";
|
|
990
|
+
if (!preflightUnavailable) {
|
|
991
|
+
pythonCountedAttempts++;
|
|
992
|
+
pythonPreflightGreen.add(runnerRoot);
|
|
993
|
+
}
|
|
994
|
+
if (baselineGreen)
|
|
995
|
+
pythonGreenBaselines++;
|
|
996
|
+
}
|
|
898
997
|
const attempt = {
|
|
899
998
|
target_symbol: symId,
|
|
900
999
|
test_path: displayTest,
|
|
@@ -902,7 +1001,8 @@ function proveExistingAssociatedTests(root, graph, sourceRoot, nodeById, opts, p
|
|
|
902
1001
|
reason,
|
|
903
1002
|
category,
|
|
904
1003
|
project_root: runnerRoot,
|
|
905
|
-
mutant_status: mutantStatusOf(result)
|
|
1004
|
+
mutant_status: mutantStatusOf(result),
|
|
1005
|
+
...proofDiagnosticsOf(result)
|
|
906
1006
|
};
|
|
907
1007
|
attempts.push(attempt);
|
|
908
1008
|
if (classification === "proven") {
|
|
@@ -928,7 +1028,15 @@ function proveExistingAssociatedTests(root, graph, sourceRoot, nodeById, opts, p
|
|
|
928
1028
|
}
|
|
929
1029
|
// non_killing → keep trying this symbol's other associated tests, if any.
|
|
930
1030
|
}
|
|
931
|
-
|
|
1031
|
+
const skipped = [...pythonBlockedCounts.values()].map(({ count, block, root }) => ({
|
|
1032
|
+
title: `Python project ${root}`,
|
|
1033
|
+
reason: `${count} remaining linked target${count === 1 ? " was" : "s were"} in a project root marked ${block.category}; no further runner calls were made in that root.`,
|
|
1034
|
+
category: block.category,
|
|
1035
|
+
language: "python",
|
|
1036
|
+
project_root: root,
|
|
1037
|
+
blocked_by: block.blockedBy
|
|
1038
|
+
}));
|
|
1039
|
+
return { attempts, needsSetup, skipped, proven, attempted, nonPythonAttempted: nonPythonCountedAttempts, provenSymbols };
|
|
932
1040
|
}
|
|
933
1041
|
/**
|
|
934
1042
|
* Drive prove for the top provable TS/JS targets. Two lanes: (1) prove the repo's OWN
|
|
@@ -967,13 +1075,14 @@ export async function autoProve(root, opts, deps) {
|
|
|
967
1075
|
// Keyed by language + bounded runner root; never populated by a test-specific red
|
|
968
1076
|
// baseline, an unknown failure, or a surviving mutant.
|
|
969
1077
|
const projectBlocked = new Map();
|
|
970
|
-
//
|
|
971
|
-
//
|
|
972
|
-
//
|
|
973
|
-
// only the remainder, so TOTAL attempts (existing + generation) are ≤ budget.
|
|
1078
|
+
// R7.1 adds a Python-only existing-test retry lane. Generic/non-Python calls retain
|
|
1079
|
+
// the released `autoLimit`; Python can inspect more existing targets to find
|
|
1080
|
+
// the requested number of baseline-green tests without widening any other language.
|
|
974
1081
|
const autoLimit = Math.max(1, Math.min(MAX_AUTO_LIMIT, Math.floor(opts.autoLimit ?? DEFAULT_AUTO_LIMIT)));
|
|
1082
|
+
const existingLimit = Math.max(1, Math.min(MAX_AUTO_LIMIT, Math.floor(opts.proof_existing_limit ?? DEFAULT_PROOF_EXISTING_LIMIT)));
|
|
1083
|
+
const genLimit = Math.max(1, Math.min(MAX_AUTO_LIMIT, Math.floor(opts.proof_gen_limit ?? DEFAULT_PROOF_GEN_LIMIT)));
|
|
975
1084
|
// ── Lane 1: existing associated tests — NO key required, runs FIRST (PR 1.5). ──
|
|
976
|
-
const ex = proveExistingAssociatedTests(root, graph, sourceRoot, nodeById, opts, proveLoop, proveDeps, alreadyProven, projectBlocked, autoLimit);
|
|
1085
|
+
const ex = proveExistingAssociatedTests(root, graph, sourceRoot, nodeById, opts, proveLoop, proveDeps, alreadyProven, projectBlocked, existingLimit, genLimit, autoLimit);
|
|
977
1086
|
if (opts.existingOnly) {
|
|
978
1087
|
const status = ex.proven > 0 ? "proven-run" : ex.attempted > 0 ? "ran-no-proof" : "no-targets";
|
|
979
1088
|
return {
|
|
@@ -983,7 +1092,7 @@ export async function autoProve(root, opts, deps) {
|
|
|
983
1092
|
attempted: ex.attempted,
|
|
984
1093
|
proven: ex.proven,
|
|
985
1094
|
needs_setup: ex.needsSetup,
|
|
986
|
-
skipped:
|
|
1095
|
+
skipped: ex.skipped,
|
|
987
1096
|
generated_files: [],
|
|
988
1097
|
attempts: ex.attempts
|
|
989
1098
|
};
|
|
@@ -1002,7 +1111,7 @@ export async function autoProve(root, opts, deps) {
|
|
|
1002
1111
|
attempted: ex.attempted,
|
|
1003
1112
|
proven: ex.proven,
|
|
1004
1113
|
needs_setup: ex.needsSetup,
|
|
1005
|
-
skipped:
|
|
1114
|
+
skipped: ex.skipped,
|
|
1006
1115
|
generated_files: [],
|
|
1007
1116
|
attempts: ex.attempts
|
|
1008
1117
|
};
|
|
@@ -1010,9 +1119,9 @@ export async function autoProve(root, opts, deps) {
|
|
|
1010
1119
|
const provider = deterministic ? new DeterministicProvider() : buildProvider(providerConfig);
|
|
1011
1120
|
const generate = deps.generate ?? generateTests;
|
|
1012
1121
|
const reader = fileReaderFor(sourceRoot);
|
|
1013
|
-
//
|
|
1014
|
-
//
|
|
1015
|
-
const
|
|
1122
|
+
// Generated proofs retain the released generic limit. Python's green-target cap
|
|
1123
|
+
// applies only to existing-test retries above, not to non-Python generation.
|
|
1124
|
+
const generationLimit = Math.max(0, autoLimit - ex.nonPythonAttempted);
|
|
1016
1125
|
// Candidates = ORS-ranked provable CodeSymbols. rankRiskGaps ranks by OrangePro Risk
|
|
1017
1126
|
// Score and excludes hard-confirmed symbols; we ALSO enforce the eligibility barrier
|
|
1018
1127
|
// explicitly (isEligibleProvableTarget) at selection so an excluded infra symbol can
|
|
@@ -1034,16 +1143,17 @@ export async function autoProve(root, opts, deps) {
|
|
|
1034
1143
|
const declaredDeps = readDeclaredDeps(sourceRoot);
|
|
1035
1144
|
let proven = 0;
|
|
1036
1145
|
let attempted = 0;
|
|
1037
|
-
|
|
1146
|
+
let baselineGreenTargets = 0;
|
|
1147
|
+
for (let start = 0; start < candidates.length && attempted < autoLimit && baselineGreenTargets < generationLimit; start += GEN_WINDOW) {
|
|
1038
1148
|
const window = candidates.slice(start, start + GEN_WINDOW);
|
|
1039
|
-
const need =
|
|
1149
|
+
const need = Math.max(1, Math.min(GEN_WINDOW, generationLimit - baselineGreenTargets));
|
|
1040
1150
|
const windowIds = window.map((g) => g.id);
|
|
1041
1151
|
const gen = await generate(graph, { target_ids: windowIds, limit: Math.min(windowIds.length, need), ...(opts.prompt_version ? { prompt_version: opts.prompt_version } : {}) }, provider, reader, clock);
|
|
1042
1152
|
const tests = gen.generated_tests;
|
|
1043
1153
|
// Global start offset so filenames stay unique across windows — runHintsFor
|
|
1044
1154
|
// otherwise resets its index to 0 per window and same-slug targets collide.
|
|
1045
1155
|
const hints = runHintsFor(tests, sourceRoot, start);
|
|
1046
|
-
for (let i = 0; i < tests.length && attempted <
|
|
1156
|
+
for (let i = 0; i < tests.length && attempted < autoLimit && baselineGreenTargets < generationLimit; i++) {
|
|
1047
1157
|
const test = tests[i];
|
|
1048
1158
|
const hint = hints[i];
|
|
1049
1159
|
if (!hint.prove_run) {
|
|
@@ -1141,19 +1251,12 @@ export async function autoProve(root, opts, deps) {
|
|
|
1141
1251
|
// See lane 1: link node_modules so a generated test importing a repo dep can boot.
|
|
1142
1252
|
link_node_modules: true,
|
|
1143
1253
|
...(testEnv ? { test_env: testEnv } : {}),
|
|
1144
|
-
run_id: `auto-prove-${
|
|
1254
|
+
run_id: `auto-prove-${ex.attempted + attempted}`
|
|
1145
1255
|
}, proveDeps);
|
|
1146
1256
|
}
|
|
1147
1257
|
catch (e) {
|
|
1148
|
-
|
|
1149
|
-
|
|
1150
|
-
test_path: writeRel,
|
|
1151
|
-
classification: "needs_setup",
|
|
1152
|
-
reason: `Proof could not run: ${redactSecrets(errMsg(e))}`,
|
|
1153
|
-
project_root: projectRoot
|
|
1154
|
-
};
|
|
1155
|
-
attempts.push(attempt);
|
|
1156
|
-
needsSetup.push(attempt);
|
|
1258
|
+
attempted--;
|
|
1259
|
+
skipped.push({ target_symbol, title: test.title, reason: `Proof could not run: ${redactSecrets(errMsg(e))}`, language, project_root: projectRoot });
|
|
1157
1260
|
continue;
|
|
1158
1261
|
}
|
|
1159
1262
|
const { classification, reason, category } = classifyProof(result, { sourceRoot, targetFileRel });
|
|
@@ -1164,9 +1267,12 @@ export async function autoProve(root, opts, deps) {
|
|
|
1164
1267
|
reason,
|
|
1165
1268
|
category,
|
|
1166
1269
|
project_root: projectRoot,
|
|
1167
|
-
mutant_status: mutantStatusOf(result)
|
|
1270
|
+
mutant_status: mutantStatusOf(result),
|
|
1271
|
+
...proofDiagnosticsOf(result)
|
|
1168
1272
|
};
|
|
1169
1273
|
attempts.push(attempt);
|
|
1274
|
+
if (baselineGreenOf(result))
|
|
1275
|
+
baselineGreenTargets++;
|
|
1170
1276
|
if (classification === "proven")
|
|
1171
1277
|
proven++;
|
|
1172
1278
|
else if (classification === "needs_setup") {
|
|
@@ -1199,7 +1305,7 @@ export async function autoProve(root, opts, deps) {
|
|
|
1199
1305
|
attempted: totalAttempted,
|
|
1200
1306
|
proven: totalProven,
|
|
1201
1307
|
needs_setup: [...ex.needsSetup, ...needsSetup],
|
|
1202
|
-
skipped,
|
|
1308
|
+
skipped: [...ex.skipped, ...skipped],
|
|
1203
1309
|
generated_files: generatedFiles,
|
|
1204
1310
|
attempts: [...ex.attempts, ...attempts]
|
|
1205
1311
|
};
|