progmune-runtime 2.1.6 → 3.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +108 -468
- package/dist/ablation-study.js +144 -0
- package/dist/ablation-study.test.js +18 -0
- package/dist/action-runtime.js +3 -1
- package/dist/active-learning.js +211 -0
- package/dist/analytics.js +139 -0
- package/dist/asset-factory.js +309 -0
- package/dist/asset-growth.js +244 -0
- package/dist/asset-promotion.js +382 -0
- package/dist/asset-quality.js +550 -0
- package/dist/audit/business-translator.js +285 -0
- package/dist/audit/cli.js +66 -0
- package/dist/audit/formatters/html.js +379 -0
- package/dist/audit/formatters/json.js +11 -0
- package/dist/audit/formatters/markdown.js +192 -0
- package/dist/audit/formatters/terminal.js +189 -0
- package/dist/audit/index.js +25 -0
- package/dist/audit/report-builder.js +318 -0
- package/dist/audit/types.js +8 -0
- package/dist/audit.js +3 -3
- package/dist/auto-benchmark-generator.js +137 -0
- package/dist/auto-benchmark-generator.test.js +45 -0
- package/dist/auto-protocol-synthesizer.js +362 -0
- package/dist/auto-protocol-synthesizer.test.js +82 -0
- package/dist/autonomous-patch.js +175 -0
- package/dist/autonomous-patch.test.js +128 -0
- package/dist/badge/badge-server.js +98 -0
- package/dist/behavior-miner.js +442 -0
- package/dist/belief-layer.js +475 -0
- package/dist/benchmark-count.js +5 -0
- package/dist/benchmark-generator.js +211 -0
- package/dist/benchmark-harness.js +201 -0
- package/dist/benchmark-pass-rate.js +7 -0
- package/dist/benchmark-report.js +8 -3
- package/dist/benchmark-save.js +14 -1
- package/dist/bootstrap-validation.js +197 -0
- package/dist/bootstrap-validation.test.js +51 -0
- package/dist/branch-ledger.js +1 -1
- package/dist/capability-gap.js +130 -0
- package/dist/certify-html.js +351 -0
- package/dist/certify.js +326 -0
- package/dist/check.js +4 -4
- package/dist/compliance-miner.js +447 -0
- package/dist/continuous-benchmark.js +194 -0
- package/dist/continuous-benchmark.test.js +116 -0
- package/dist/corpus-stats.js +173 -0
- package/dist/counterfactual-engine.js +288 -0
- package/dist/coverage-dashboard.js +109 -0
- package/dist/coverage-system.test.js +205 -0
- package/dist/cross-repo-precision.js +352 -0
- package/dist/cve-benchmark.js +180 -0
- package/dist/cve-benchmark.test.js +28 -0
- package/dist/cve-collector.js +73 -0
- package/dist/data-quality.js +141 -0
- package/dist/decision-engine.js +388 -0
- package/dist/derive-metadata.js +250 -0
- package/dist/difficulty-active.test.js +198 -0
- package/dist/difficulty-map.js +244 -0
- package/dist/discovery-analytics.js +125 -0
- package/dist/discovery-model.js +149 -0
- package/dist/discovery-optimize.test.js +199 -0
- package/dist/discovery-trace.js +276 -0
- package/dist/discovery-trace.test.js +97 -0
- package/dist/emitter.js +83 -1
- package/dist/enterprise-dashboard.js +405 -0
- package/dist/eval-hardening.js +297 -0
- package/dist/eval-hardening.test.js +85 -0
- package/dist/evaluation-campaign.js +359 -0
- package/dist/evaluation-campaign.test.js +181 -0
- package/dist/evidence-growth.js +143 -0
- package/dist/evidence-repository.js +209 -0
- package/dist/evidence-system.js +441 -0
- package/dist/execute.js +15 -7
- package/dist/experimental/software-physics.js +291 -0
- package/dist/experimental/state-inference.js +516 -0
- package/dist/experimental/unsupervised-physics.js +230 -0
- package/dist/extract-ir-python.js +54 -7
- package/dist/extract-ir.js +376 -12
- package/dist/failure-collector.js +2 -2
- package/dist/failure-corpus.js +322 -9
- package/dist/feedback.js +16 -5
- package/dist/feedback.test.js +49 -0
- package/dist/file-lock.js +1 -1
- package/dist/flywheel-batch.js +292 -0
- package/dist/frameworks/express-cli.js +237 -0
- package/dist/frameworks/express-detector.js +445 -0
- package/dist/frameworks/express-detector.test.js +206 -0
- package/dist/frameworks/index.js +30 -0
- package/dist/frameworks/nestjs-detector.js +302 -0
- package/dist/frameworks/trpc-detector.js +161 -0
- package/dist/frameworks/version-awareness.js +179 -0
- package/dist/function-synonyms.js +164 -0
- package/dist/function-synonyms.test.js +68 -0
- package/dist/generalization.test.js +352 -0
- package/dist/goal-annotator.js +113 -0
- package/dist/goal-planner.js +563 -0
- package/dist/gold-cve.js +164 -0
- package/dist/gold-cve.test.js +104 -0
- package/dist/gold-quality.js +206 -0
- package/dist/gold-tiers.js +241 -0
- package/dist/governance-dashboard.js +327 -0
- package/dist/graph-viz.js +240 -0
- package/dist/guided-frontier.js +195 -0
- package/dist/hierarchical-planner.js +148 -0
- package/dist/identifier-parser.js +260 -0
- package/dist/immune-metrics.js +93 -0
- package/dist/immune-receiver.js +158 -0
- package/dist/immune-reporter.js +1 -1
- package/dist/improvement-orchestrator.js +206 -0
- package/dist/inject-p0-vocabulary.js +300 -0
- package/dist/intent-parser.js +218 -0
- package/dist/invariant-algebra.js +476 -0
- package/dist/invariant-calculus.js +533 -0
- package/dist/ir-utils.js +70 -0
- package/dist/ir-utils.test.js +50 -0
- package/dist/knowledge-api.js +312 -0
- package/dist/knowledge-evolution.js +452 -0
- package/dist/knowledge-explorer.js +506 -0
- package/dist/knowledge-flywheel.js +274 -0
- package/dist/knowledge-governance.js +338 -0
- package/dist/knowledge-governance.test.js +150 -0
- package/dist/knowledge-graph.js +181 -0
- package/dist/knowledge-guided-synth.js +246 -0
- package/dist/knowledge-loop.test.js +77 -0
- package/dist/knowledge-object.js +316 -0
- package/dist/knowledge-package.js +98 -0
- package/dist/kpi-dashboard.js +561 -0
- package/dist/l3-cross-function.js +280 -0
- package/dist/learning-ranker.js +148 -0
- package/dist/learning-ranker.test.js +291 -0
- package/dist/ledger/accountability.js +322 -0
- package/dist/ledger/chain-builder.js +185 -0
- package/dist/ledger/cli.js +222 -0
- package/dist/ledger/index.js +13 -0
- package/dist/ledger/signatures.js +193 -0
- package/dist/ledger/types.js +9 -0
- package/dist/llm.js +74 -3
- package/dist/load-benchmarks.js +8 -3
- package/dist/logger.js +66 -0
- package/dist/logger.test.js +37 -0
- package/dist/logistic-reward.js +339 -0
- package/dist/logistic-reward.test.js +180 -0
- package/dist/macro-graph.js +193 -0
- package/dist/macro-repair.js +183 -0
- package/dist/mcp-server.mjs +1202 -483
- package/dist/memory-layer.js +42 -5
- package/dist/multi-repo-precision.js +422 -0
- package/dist/name-free-protocol.js +425 -0
- package/dist/name-free-protocol.test.js +170 -0
- package/dist/name-scrambling.js +138 -0
- package/dist/name-scrambling.test.js +16 -0
- package/dist/p3-observability.test.js +281 -0
- package/dist/p5-orchestrator.test.js +225 -0
- package/dist/pairwise-preference.js +294 -0
- package/dist/pairwise-preference.test.js +140 -0
- package/dist/planner-constraints.js +104 -0
- package/dist/planner-prompts.js +155 -0
- package/dist/planner-telemetry.js +415 -0
- package/dist/planner-trace.js +214 -0
- package/dist/planner.js +162 -167
- package/dist/plsb/artifact.js +116 -0
- package/dist/plsb/cli.js +71 -0
- package/dist/plsb/index.js +19 -0
- package/dist/plsb/leaderboard.js +249 -0
- package/dist/plsb/report-md.js +156 -0
- package/dist/plsb/schema.js +179 -0
- package/dist/plsb-benchmark.js +284 -0
- package/dist/plsb-benchmark.test.js +119 -0
- package/dist/policy/cli.js +134 -0
- package/dist/policy/engine.js +333 -0
- package/dist/policy/index.js +12 -0
- package/dist/policy/types.js +59 -0
- package/dist/policy-miner.js +505 -0
- package/dist/precision-analyze.js +229 -0
- package/dist/precision-benchmark.js +147 -0
- package/dist/precision-label-c.js +134 -0
- package/dist/precision-label.js +193 -0
- package/dist/precision-report-c.js +149 -0
- package/dist/precision-report.js +246 -0
- package/dist/progmune-status.js +108 -0
- package/dist/proof-engine.js +479 -0
- package/dist/proof-provenance.js +315 -0
- package/dist/protocol-coverage.js +294 -0
- package/dist/protocol-detector.js +1189 -0
- package/dist/protocol-embedding-expanded.js +297 -0
- package/dist/protocol-embedding-expanded.test.js +97 -0
- package/dist/protocol-embedding.js +195 -0
- package/dist/protocol-embedding.test.js +82 -0
- package/dist/protocol-extractor-v2.js +354 -0
- package/dist/protocol-extractor-v2.test.js +140 -0
- package/dist/protocol-extractor.js +310 -0
- package/dist/protocol-extractor.test.js +113 -0
- package/dist/protocol-foundation.js +322 -0
- package/dist/protocol-foundation.test.js +163 -0
- package/dist/protocol-frontier.js +243 -0
- package/dist/protocol-frontier.test.js +92 -0
- package/dist/protocol-gap-analyzer.js +228 -0
- package/dist/protocol-gap-analyzer.test.js +49 -0
- package/dist/protocol-invariants.js +276 -0
- package/dist/protocol-invariants.test.js +111 -0
- package/dist/protocol-knowledge.js +464 -0
- package/dist/protocol-miner.js +343 -0
- package/dist/protocol-mining.js +207 -0
- package/dist/protocol-mining.test.js +37 -0
- package/dist/protocol-registry.js +1 -1
- package/dist/protocol-security-benchmark.js +222 -0
- package/dist/protocol-vulnerability.js +257 -0
- package/dist/protocol-vulnerability.test.js +60 -0
- package/dist/python-benchmark.js +120 -0
- package/dist/python-emitter.js +163 -45
- package/dist/python-protocol-extractor.js +187 -0
- package/dist/python-protocol-extractor.test.js +116 -0
- package/dist/realworld-benchmark.js +646 -0
- package/dist/realworld-benchmark.test.js +36 -0
- package/dist/repair-arch.test.js +411 -0
- package/dist/repair-evolution.test.js +454 -0
- package/dist/repair-executor.js +719 -0
- package/dist/repair-proposal.js +4 -4
- package/dist/repair-ranker.js +141 -0
- package/dist/repair-strategies.js +419 -0
- package/dist/repair-taxonomy.js +234 -0
- package/dist/repair-types.js +12 -0
- package/dist/repo-evaluator.js +250 -0
- package/dist/repo-evaluator.test.js +128 -0
- package/dist/resource-abstraction.js +242 -0
- package/dist/resource-detector.js +211 -0
- package/dist/result.test.js +43 -0
- package/dist/reward-system.js +411 -0
- package/dist/reward-system.test.js +175 -0
- package/dist/risk-model.js +215 -0
- package/dist/rule-miner.js +234 -7
- package/dist/rule-specificity.js +254 -0
- package/dist/runtime-types.js +27 -0
- package/dist/scaffold.js +208 -0
- package/dist/scale-collector.test.js +101 -0
- package/dist/scale-trajectory-collector.js +128 -0
- package/dist/sdk.js +250 -0
- package/dist/search-planner.js +4 -41
- package/dist/semantic-snapshot.js +1 -1
- package/dist/semantic-topology.js +121 -0
- package/dist/semantic-trace.js +310 -317
- package/dist/sequence-extractor.js +343 -0
- package/dist/skill-library.js +245 -0
- package/dist/skill-planner.test.js +189 -0
- package/dist/software-physics.js +291 -0
- package/dist/software-physics.test.js +81 -0
- package/dist/ssg-precision.js +478 -0
- package/dist/ssg-validator.js +71 -21
- package/dist/state-inference-doubleblind.test.js +160 -0
- package/dist/state-inference.js +516 -0
- package/dist/state-inference.test.js +115 -0
- package/dist/state-machine-fingerprint.js +345 -0
- package/dist/state-machine-fingerprint.test.js +120 -0
- package/dist/state-miner.js +386 -0
- package/dist/state-name-inference.js +213 -0
- package/dist/state-name-inference.test.js +69 -0
- package/dist/strategy-planner.js +262 -96
- package/dist/strategy-planner.test.js +135 -0
- package/dist/telemetry-analytics.test.js +402 -0
- package/dist/terminal-format.js +68 -0
- package/dist/terminal-format.test.js +83 -0
- package/dist/topology-factory.js +196 -0
- package/dist/topology-representation.js +242 -0
- package/dist/topology-representation.test.js +27 -0
- package/dist/trajectory-augmentation.js +254 -0
- package/dist/trajectory-augmentation.test.js +63 -0
- package/dist/trajectory-corpus.js +440 -0
- package/dist/trajectory-corpus.test.js +32 -0
- package/dist/trajectory-feedback.test.js +116 -0
- package/dist/transition-synthesizer.js +286 -0
- package/dist/transition-synthesizer.test.js +123 -0
- package/dist/trust/api-semantic-mapper.js +809 -0
- package/dist/trust/call-graph-propagator.js +225 -0
- package/dist/trust/cli.js +122 -0
- package/dist/trust/compliance-scorer.js +283 -0
- package/dist/trust/confidence-calculator.js +261 -0
- package/dist/trust/engine.js +1145 -0
- package/dist/trust/explainability.js +85 -0
- package/dist/trust/formatters/ci.js +42 -0
- package/dist/trust/formatters/json.js +11 -0
- package/dist/trust/formatters/terminal.js +152 -0
- package/dist/trust/index.js +39 -0
- package/dist/trust/phase1-verify.js +171 -0
- package/dist/trust/protocol-domain-validator.js +697 -0
- package/dist/trust/score-calculator.js +282 -0
- package/dist/trust/ssg-bridge.js +641 -0
- package/dist/trust/ssg-bridge.test.js +269 -0
- package/dist/trust/types.js +67 -0
- package/dist/trust/violation-trace.js +335 -0
- package/dist/trust-api.js +179 -0
- package/dist/trust-calibration.js +279 -0
- package/dist/unknown-protocol-discovery.js +339 -0
- package/dist/unknown-protocol-discovery.test.js +102 -0
- package/dist/unsupervised-physics.js +230 -0
- package/dist/unsupervised-physics.test.js +95 -0
- package/dist/utils.test.js +37 -0
- package/dist/validator.js +187 -10
- package/dist/verification-intelligence.js +475 -0
- package/dist/verify-api.js +432 -0
- package/dist/vi-impact-report.js +293 -0
- package/dist/wl-fingerprint.js +162 -0
- package/dist/wl-fingerprint.test.js +130 -0
- package/dist/zeroshot-strategy.js +139 -0
- package/docs/Progmune_/346/212/225/350/265/204/344/272/272/347/231/275/347/232/256/344/271/246_v2.0.html +576 -0
- package/docs/Progmune_/351/241/271/347/233/256/345/205/250/350/247/243.html +710 -0
- package/package.json +74 -7
- package/protocols.json +1956 -50
- package/.dockerignore +0 -14
- package/.mcp.json +0 -11
- package/.progmune_allowlist +0 -50
- package/.test_report/test_report.md +0 -87
- package/Dockerfile +0 -9
- package/FAQ.md +0 -167
- package/WHITEPAPER.md +0 -540
- package/demo-project/auth.ts +0 -55
- package/demo-project/tsconfig.json +0 -8
- package/dist/acl-breakdown.js +0 -13
- package/dist/all-sessions.js +0 -11
- package/dist/antibody-stats.js +0 -11
- package/dist/branch-tree-count.js +0 -14
- package/dist/common-fixpath.js +0 -12
- package/dist/constraint-types.js +0 -12
- package/dist/exec-metrics.js +0 -11
- package/dist/failure-report.js +0 -11
- package/dist/fast-path-hits.js +0 -13
- package/dist/fingerprint-list.js +0 -15
- package/dist/gen-history-log.js +0 -13
- package/dist/heatmap-data.js +0 -11
- package/dist/recent-session.js +0 -12
- package/dist/svl-distribution.js +0 -11
- package/dist/terminal-status.js +0 -11
- package/dist/token-savings.js +0 -11
- package/dist/total-repairs.js +0 -12
- package/dist/unresolved-count.js +0 -12
- package/dist/valid-fingerprints.js +0 -13
- package/dist/verify-ledgers.js +0 -11
- package/docs/whitepaper-style.css +0 -77
- package/docs/whitepaper-v2.1.md +0 -609
- package/docs/whitepaper-v2.2.md +0 -1064
- package/docs/whitepaper-v2.2.pdf +0 -0
- package/fly.toml +0 -31
- package/public/dashboard.html +0 -119
- package/server/hub.js +0 -116
- package/test/replay-golden/sess_1780063202050_mgeld.json +0 -9
- package/test/replay-golden/sess_1780064032560_gocld.json +0 -354
- package/test/replay-golden/sess_1780064413331_s2709.json +0 -606
- package/test/replay-golden/sess_1780064792710_y3avo.json +0 -614
- package/test/replay-golden.ts +0 -84
- package/test_benchmark.js +0 -165
- package/test_comprehensive.mjs +0 -638
- package/test_concurrency.js +0 -129
- package/test_ir_robustness.js +0 -85
- package/test_semantic_contracts.js +0 -269
- package/test_ssg_stress.js +0 -156
- package/test_svl3.js +0 -58
- package/tsconfig.json +0 -17
|
@@ -0,0 +1,359 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* P3.9: Evaluation Campaign
|
|
4
|
+
*
|
|
5
|
+
* Shifts from "building modules" to "validating hypotheses".
|
|
6
|
+
*
|
|
7
|
+
* Three tools:
|
|
8
|
+
* 1. Failure Attribution — classify WHY each benchmark case fails
|
|
9
|
+
* 2. Error Budget Dashboard — aggregate failure reasons
|
|
10
|
+
* 3. Offline Replay Engine — replay history with new rankers, compute accuracy
|
|
11
|
+
*
|
|
12
|
+
* Key metric: Replay Accuracy — how often would the ranker have matched
|
|
13
|
+
* what the user actually chose? If LearningRanker > LinearRanker by 10%+,
|
|
14
|
+
* the Telemetry→Feedback→Learning loop is proven effective.
|
|
15
|
+
*/
|
|
16
|
+
var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
|
|
17
|
+
if (k2 === undefined) k2 = k;
|
|
18
|
+
var desc = Object.getOwnPropertyDescriptor(m, k);
|
|
19
|
+
if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
|
|
20
|
+
desc = { enumerable: true, get: function() { return m[k]; } };
|
|
21
|
+
}
|
|
22
|
+
Object.defineProperty(o, k2, desc);
|
|
23
|
+
}) : (function(o, m, k, k2) {
|
|
24
|
+
if (k2 === undefined) k2 = k;
|
|
25
|
+
o[k2] = m[k];
|
|
26
|
+
}));
|
|
27
|
+
var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
|
|
28
|
+
Object.defineProperty(o, "default", { enumerable: true, value: v });
|
|
29
|
+
}) : function(o, v) {
|
|
30
|
+
o["default"] = v;
|
|
31
|
+
});
|
|
32
|
+
var __importStar = (this && this.__importStar) || (function () {
|
|
33
|
+
var ownKeys = function(o) {
|
|
34
|
+
ownKeys = Object.getOwnPropertyNames || function (o) {
|
|
35
|
+
var ar = [];
|
|
36
|
+
for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
|
|
37
|
+
return ar;
|
|
38
|
+
};
|
|
39
|
+
return ownKeys(o);
|
|
40
|
+
};
|
|
41
|
+
return function (mod) {
|
|
42
|
+
if (mod && mod.__esModule) return mod;
|
|
43
|
+
var result = {};
|
|
44
|
+
if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
|
|
45
|
+
__setModuleDefault(result, mod);
|
|
46
|
+
return result;
|
|
47
|
+
};
|
|
48
|
+
})();
|
|
49
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
50
|
+
exports.runFailureAttribution = runFailureAttribution;
|
|
51
|
+
exports.computeErrorBudget = computeErrorBudget;
|
|
52
|
+
exports.printErrorBudget = printErrorBudget;
|
|
53
|
+
exports.replayDecisions = replayDecisions;
|
|
54
|
+
exports.compareRankers = compareRankers;
|
|
55
|
+
exports.printReplayReport = printReplayReport;
|
|
56
|
+
exports.printRankerComparison = printRankerComparison;
|
|
57
|
+
const fs = __importStar(require("fs"));
|
|
58
|
+
const path = __importStar(require("path"));
|
|
59
|
+
const counterfactual_engine_1 = require("./counterfactual-engine");
|
|
60
|
+
const ssg_validator_1 = require("./ssg-validator");
|
|
61
|
+
function expectedSignature(expected) {
|
|
62
|
+
return [...expected].sort().join("→");
|
|
63
|
+
}
|
|
64
|
+
function resultSignature(fixPath) {
|
|
65
|
+
return [...fixPath].sort().join("→");
|
|
66
|
+
}
|
|
67
|
+
async function classifyFailure(tc, result, alts) {
|
|
68
|
+
if (result.top1Hit)
|
|
69
|
+
return "success";
|
|
70
|
+
const expSig = expectedSignature(tc.expected);
|
|
71
|
+
// No candidates at all → protocol model can't find the path
|
|
72
|
+
if (alts.length === 0) {
|
|
73
|
+
// Check if it's a resource leak (ProtocolStrategy handles these)
|
|
74
|
+
if (tc.violationType === "resource_leak")
|
|
75
|
+
return "bad_protocol_model";
|
|
76
|
+
return "bad_protocol_model";
|
|
77
|
+
}
|
|
78
|
+
// Check if ANY candidate has the expected repair (even if ranked wrong)
|
|
79
|
+
const anyMatch = alts.some(a => resultSignature(a.fixPath) === expSig);
|
|
80
|
+
if (anyMatch)
|
|
81
|
+
return "bad_ranking";
|
|
82
|
+
// Check if it's a corpus-dependent scenario
|
|
83
|
+
if (tc.violationType === "missing_prerequisite" && alts.length <= 2) {
|
|
84
|
+
return "insufficient_history";
|
|
85
|
+
}
|
|
86
|
+
// If expected includes functions not in protocol rules, goal mismatch
|
|
87
|
+
const protoDef = JSON.parse(fs.readFileSync(path.resolve(__dirname, "..", "protocols.json"), "utf-8"));
|
|
88
|
+
const allRules = new Set(Object.keys(protoDef.rules));
|
|
89
|
+
const unknownFns = tc.expected.filter(fn => !allRules.has(fn));
|
|
90
|
+
if (unknownFns.length > 0)
|
|
91
|
+
return "goal_mismatch";
|
|
92
|
+
// Otherwise: candidate simply wasn't found
|
|
93
|
+
return "missing_candidate";
|
|
94
|
+
}
|
|
95
|
+
async function runFailureAttribution(suitePath) {
|
|
96
|
+
const benchmarksDir = suitePath || path.resolve(__dirname, "..", "benchmarks");
|
|
97
|
+
const files = fs.readdirSync(benchmarksDir).filter(f => f.endsWith(".json") && !f.includes("generated") && !f.includes("priority"));
|
|
98
|
+
const protoDef = JSON.parse(fs.readFileSync(path.resolve(__dirname, "..", "protocols.json"), "utf-8"));
|
|
99
|
+
const protocols = (0, ssg_validator_1.parseProtocolsFromJSON)(protoDef);
|
|
100
|
+
const rules = new Map();
|
|
101
|
+
for (const p of protocols)
|
|
102
|
+
rules.set(p.function, p.protocol);
|
|
103
|
+
const attributed = [];
|
|
104
|
+
for (const file of files) {
|
|
105
|
+
const cases = JSON.parse(fs.readFileSync(path.join(benchmarksDir, file), "utf-8"));
|
|
106
|
+
for (const tc of cases) {
|
|
107
|
+
const currentStates = new Set();
|
|
108
|
+
for (const fn of tc.broken) {
|
|
109
|
+
const rule = rules.get(fn);
|
|
110
|
+
if (rule) {
|
|
111
|
+
for (const post of rule.post_states)
|
|
112
|
+
currentStates.add(post);
|
|
113
|
+
if (rule.invalidate)
|
|
114
|
+
rule.invalidate.forEach(s => currentStates.delete(s));
|
|
115
|
+
}
|
|
116
|
+
}
|
|
117
|
+
const alts = await (0, counterfactual_engine_1.suggestAlternatives)({
|
|
118
|
+
violation: {
|
|
119
|
+
svl: 4,
|
|
120
|
+
violatedConstraint: tc.violationType,
|
|
121
|
+
actionIndex: tc.broken.length,
|
|
122
|
+
currentStates: [...currentStates],
|
|
123
|
+
requiredStates: [],
|
|
124
|
+
description: `Benchmark: ${tc.goal}`,
|
|
125
|
+
},
|
|
126
|
+
protocol: tc.protocol,
|
|
127
|
+
currentState: [...currentStates],
|
|
128
|
+
targetState: [],
|
|
129
|
+
constraints: [],
|
|
130
|
+
rules,
|
|
131
|
+
goal: tc.goal,
|
|
132
|
+
});
|
|
133
|
+
const expSig = expectedSignature(tc.expected);
|
|
134
|
+
let top1Hit = false;
|
|
135
|
+
let rank = null;
|
|
136
|
+
for (let i = 0; i < alts.length; i++) {
|
|
137
|
+
const fullSig = expectedSignature([...tc.broken, ...alts[i].fixPath]);
|
|
138
|
+
if (fullSig === expSig) {
|
|
139
|
+
if (rank === null)
|
|
140
|
+
rank = i + 1;
|
|
141
|
+
if (i === 0)
|
|
142
|
+
top1Hit = true;
|
|
143
|
+
}
|
|
144
|
+
}
|
|
145
|
+
const caseResult = {
|
|
146
|
+
goal: tc.goal, top1Hit, top3Hit: rank !== null && rank <= 3,
|
|
147
|
+
rank, latencyMs: 0, candidatesReturned: alts.length,
|
|
148
|
+
};
|
|
149
|
+
const reason = await classifyFailure(tc, caseResult, alts);
|
|
150
|
+
attributed.push({
|
|
151
|
+
caseId: `${file}:${tc.goal}`,
|
|
152
|
+
goal: tc.goal,
|
|
153
|
+
protocol: tc.protocol,
|
|
154
|
+
violationType: tc.violationType,
|
|
155
|
+
expectedRepair: tc.expected,
|
|
156
|
+
plannerTop1: alts.length > 0 ? alts[0].fixPath : undefined,
|
|
157
|
+
plannerTop3: alts.slice(0, 3).map(a => a.fixPath),
|
|
158
|
+
candidatesReturned: alts.length,
|
|
159
|
+
rank,
|
|
160
|
+
failureReason: reason,
|
|
161
|
+
});
|
|
162
|
+
}
|
|
163
|
+
}
|
|
164
|
+
return attributed;
|
|
165
|
+
}
|
|
166
|
+
function computeErrorBudget(attributed) {
|
|
167
|
+
const total = attributed.length;
|
|
168
|
+
const successes = attributed.filter(a => a.failureReason === "success").length;
|
|
169
|
+
const breakdown = {};
|
|
170
|
+
for (const a of attributed) {
|
|
171
|
+
breakdown[a.failureReason] = (breakdown[a.failureReason] || 0) + 1;
|
|
172
|
+
}
|
|
173
|
+
const percentages = {};
|
|
174
|
+
for (const [k, v] of Object.entries(breakdown)) {
|
|
175
|
+
percentages[k] = total > 0 ? v / total : 0;
|
|
176
|
+
}
|
|
177
|
+
// Generate recommendation
|
|
178
|
+
const missingPct = percentages["missing_candidate"] || 0;
|
|
179
|
+
const rankingPct = percentages["bad_ranking"] || 0;
|
|
180
|
+
const protocolPct = percentages["bad_protocol_model"] || 0;
|
|
181
|
+
let recommendation;
|
|
182
|
+
if (missingPct > 0.3) {
|
|
183
|
+
recommendation = "P0: Fix candidate discovery (ProtocolStrategy BFS). Don't touch Reward Model until candidates are found.";
|
|
184
|
+
}
|
|
185
|
+
else if (rankingPct > 0.3) {
|
|
186
|
+
recommendation = "P0: Improve ranking (LearningRanker, more feedback data). Ranking is the bottleneck.";
|
|
187
|
+
}
|
|
188
|
+
else if (protocolPct > 0.3) {
|
|
189
|
+
recommendation = "P0: Expand protocol rules. Protocol model doesn't cover enough transitions.";
|
|
190
|
+
}
|
|
191
|
+
else {
|
|
192
|
+
recommendation = "Balanced error profile. Proceed with small improvements across all dimensions.";
|
|
193
|
+
}
|
|
194
|
+
return {
|
|
195
|
+
totalCases: total,
|
|
196
|
+
successes,
|
|
197
|
+
successRate: total > 0 ? successes / total : 0,
|
|
198
|
+
breakdown: breakdown,
|
|
199
|
+
percentages: percentages,
|
|
200
|
+
recommendation,
|
|
201
|
+
};
|
|
202
|
+
}
|
|
203
|
+
function printErrorBudget(budget) {
|
|
204
|
+
console.log("\n╔════════════════════════════════════════════════════╗");
|
|
205
|
+
console.log("║ Error Budget Dashboard ║");
|
|
206
|
+
console.log("╚════════════════════════════════════════════════════╝\n");
|
|
207
|
+
console.log(`Total Cases: ${budget.totalCases}`);
|
|
208
|
+
console.log(`Successes: ${budget.successes} (${(budget.successRate * 100).toFixed(0)}%)`);
|
|
209
|
+
console.log();
|
|
210
|
+
console.log("─── Failure Breakdown ───");
|
|
211
|
+
console.log("Reason Count Pct Bar");
|
|
212
|
+
console.log("──────────────────────────────────────────────");
|
|
213
|
+
const order = ["missing_candidate", "bad_ranking", "bad_protocol_model", "goal_mismatch", "insufficient_history", "success"];
|
|
214
|
+
for (const reason of order) {
|
|
215
|
+
const count = budget.breakdown[reason] || 0;
|
|
216
|
+
const pct = budget.percentages[reason] || 0;
|
|
217
|
+
const bar = "█".repeat(Math.round(pct * 40));
|
|
218
|
+
const label = reason.padEnd(22);
|
|
219
|
+
const pctStr = (pct * 100).toFixed(0).padStart(3) + "%";
|
|
220
|
+
console.log(` ${label} ${String(count).padStart(4)} ${pctStr} ${bar}`);
|
|
221
|
+
}
|
|
222
|
+
console.log();
|
|
223
|
+
console.log(`─── Recommendation ───`);
|
|
224
|
+
console.log(` ${budget.recommendation}`);
|
|
225
|
+
console.log();
|
|
226
|
+
}
|
|
227
|
+
/**
|
|
228
|
+
* Replay historical decisions with a new candidate ranking.
|
|
229
|
+
*
|
|
230
|
+
* Given PlannerTrace data (what was shown to the user and what they chose)
|
|
231
|
+
* and a LearningRanker (which re-scores candidates using feedback data),
|
|
232
|
+
* compute how often the new ranker's top-1 matches the user's choice.
|
|
233
|
+
*/
|
|
234
|
+
function replayDecisions(traceStore, telemetry) {
|
|
235
|
+
const traces = traceStore.all();
|
|
236
|
+
const withChoice = traces.filter((t) => t.selectedFingerprint && t.candidates.length > 0);
|
|
237
|
+
const results = [];
|
|
238
|
+
for (const trace of withChoice) {
|
|
239
|
+
const userChose = trace.selectedFingerprint;
|
|
240
|
+
// Build RepairCandidate-like objects from snapshots
|
|
241
|
+
const candidates = trace.candidates.map((c) => ({
|
|
242
|
+
fingerprint: c.fingerprint,
|
|
243
|
+
oldRank: c.rank,
|
|
244
|
+
oldScore: c.score,
|
|
245
|
+
source: c.source,
|
|
246
|
+
actions: c.actions,
|
|
247
|
+
evidenceSources: c.evidenceSources,
|
|
248
|
+
}));
|
|
249
|
+
// Re-rank using telemetry acceptance data
|
|
250
|
+
// Higher acceptance = better rank
|
|
251
|
+
const reranked = candidates.map((c) => ({
|
|
252
|
+
...c,
|
|
253
|
+
acceptance: telemetry.getCandidateAcceptance(c.fingerprint, 1),
|
|
254
|
+
})).sort((a, b) => {
|
|
255
|
+
// Sort by acceptance descending, then by old score
|
|
256
|
+
if (a.acceptance !== b.acceptance)
|
|
257
|
+
return b.acceptance - a.acceptance;
|
|
258
|
+
return b.oldScore - a.oldScore;
|
|
259
|
+
});
|
|
260
|
+
const newRankerChose = reranked[0]?.fingerprint ?? null;
|
|
261
|
+
const matched = newRankerChose === userChose;
|
|
262
|
+
// Find user's choice in new ranking
|
|
263
|
+
const userIdx = reranked.findIndex((c) => c.fingerprint === userChose);
|
|
264
|
+
const userChoiceRank = userIdx >= 0 ? userIdx + 1 : null;
|
|
265
|
+
results.push({
|
|
266
|
+
traceId: trace.traceId,
|
|
267
|
+
goal: trace.goal,
|
|
268
|
+
protocol: trace.protocol,
|
|
269
|
+
userChose,
|
|
270
|
+
userChoseRank: userChoiceRank,
|
|
271
|
+
newRankerChose,
|
|
272
|
+
matched,
|
|
273
|
+
candidates: reranked.map((c, i) => ({
|
|
274
|
+
fingerprint: c.fingerprint,
|
|
275
|
+
oldRank: c.oldRank,
|
|
276
|
+
newRank: i + 1,
|
|
277
|
+
score: c.acceptance,
|
|
278
|
+
})),
|
|
279
|
+
});
|
|
280
|
+
}
|
|
281
|
+
const matches = results.filter(r => r.matched).length;
|
|
282
|
+
const avgRank = results.reduce((s, r) => s + (r.userChoseRank ?? results.length), 0) / Math.max(1, results.length);
|
|
283
|
+
return {
|
|
284
|
+
ranker: "LearningRanker (acceptance-based)",
|
|
285
|
+
totalDecisions: results.length,
|
|
286
|
+
matches,
|
|
287
|
+
matchRate: results.length > 0 ? matches / results.length : 0,
|
|
288
|
+
avgUserChoiceRank: avgRank,
|
|
289
|
+
results,
|
|
290
|
+
};
|
|
291
|
+
}
|
|
292
|
+
/**
|
|
293
|
+
* Compare two ranking strategies by replay accuracy.
|
|
294
|
+
*/
|
|
295
|
+
function compareRankers(traceStore, telemetry) {
|
|
296
|
+
const traces = traceStore.all();
|
|
297
|
+
const withChoice = traces.filter((t) => t.selectedFingerprint && t.candidates.length > 0);
|
|
298
|
+
if (withChoice.length === 0) {
|
|
299
|
+
const empty = { ranker: "", totalDecisions: 0, matches: 0, matchRate: 0, avgUserChoiceRank: 0, results: [] };
|
|
300
|
+
return { baseline: empty, learning: empty, delta: 0 };
|
|
301
|
+
}
|
|
302
|
+
// Baseline: original ranker (rank-1 = what planner showed first)
|
|
303
|
+
let baselineMatches = 0;
|
|
304
|
+
for (const trace of withChoice) {
|
|
305
|
+
const top1 = trace.candidates[0]?.fingerprint;
|
|
306
|
+
if (top1 === trace.selectedFingerprint)
|
|
307
|
+
baselineMatches++;
|
|
308
|
+
}
|
|
309
|
+
const baseline = {
|
|
310
|
+
ranker: "LinearRanker (original)",
|
|
311
|
+
totalDecisions: withChoice.length,
|
|
312
|
+
matches: baselineMatches,
|
|
313
|
+
matchRate: baselineMatches / withChoice.length,
|
|
314
|
+
avgUserChoiceRank: 0, // N/A for baseline
|
|
315
|
+
results: [],
|
|
316
|
+
};
|
|
317
|
+
// Learning: acceptance-based reranking
|
|
318
|
+
const learning = replayDecisions(traceStore, telemetry);
|
|
319
|
+
return {
|
|
320
|
+
baseline,
|
|
321
|
+
learning,
|
|
322
|
+
delta: learning.matchRate - baseline.matchRate,
|
|
323
|
+
};
|
|
324
|
+
}
|
|
325
|
+
function printReplayReport(report) {
|
|
326
|
+
console.log("\n╔════════════════════════════════════════════════════╗");
|
|
327
|
+
console.log("║ Offline Replay Report ║");
|
|
328
|
+
console.log("╚════════════════════════════════════════════════════╝\n");
|
|
329
|
+
console.log(`Ranker: ${report.ranker}`);
|
|
330
|
+
console.log(`Decisions Replayed: ${report.totalDecisions}`);
|
|
331
|
+
console.log(`User Choice Matched: ${report.matches}/${report.totalDecisions}`);
|
|
332
|
+
console.log(`Replay Accuracy: ${(report.matchRate * 100).toFixed(1)}%`);
|
|
333
|
+
console.log(`Avg User Choice Rank: ${report.avgUserChoiceRank.toFixed(1)}`);
|
|
334
|
+
console.log();
|
|
335
|
+
}
|
|
336
|
+
function printRankerComparison(baseline, learning, delta) {
|
|
337
|
+
console.log("\n╔════════════════════════════════════════════════════╗");
|
|
338
|
+
console.log("║ Ranker A/B Comparison ║");
|
|
339
|
+
console.log("╚════════════════════════════════════════════════════╝\n");
|
|
340
|
+
const basePct = (baseline.matchRate * 100).toFixed(1);
|
|
341
|
+
const learnPct = (learning.matchRate * 100).toFixed(1);
|
|
342
|
+
const deltaPct = (delta * 100).toFixed(1);
|
|
343
|
+
const sign = delta > 0 ? "+" : "";
|
|
344
|
+
console.log(` Baseline (LinearRanker): ${basePct}% (${baseline.matches}/${baseline.totalDecisions})`);
|
|
345
|
+
console.log(` Learning (Acceptance): ${learnPct}% (${learning.matches}/${learning.totalDecisions})`);
|
|
346
|
+
console.log(` Δ: ${sign}${deltaPct}%`);
|
|
347
|
+
console.log();
|
|
348
|
+
if (delta > 0.05) {
|
|
349
|
+
console.log(` ✅ LearningRanker outperforms baseline by ${sign}${deltaPct}%`);
|
|
350
|
+
console.log(" The Telemetry→Feedback→Learning loop is effective.");
|
|
351
|
+
}
|
|
352
|
+
else if (delta > 0) {
|
|
353
|
+
console.log(" ⚠️ Marginal improvement. More feedback data needed.");
|
|
354
|
+
}
|
|
355
|
+
else {
|
|
356
|
+
console.log(" ❌ No improvement. Check data quality or increase sample size.");
|
|
357
|
+
}
|
|
358
|
+
console.log();
|
|
359
|
+
}
|
|
@@ -0,0 +1,181 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* P3.9: Evaluation Campaign Tests
|
|
4
|
+
*
|
|
5
|
+
* Verifying:
|
|
6
|
+
* 1. Failure attribution classifies benchmark misses correctly
|
|
7
|
+
* 2. Error budget dashboard produces actionable breakdown
|
|
8
|
+
* 3. Offline replay computes match rate against user choices
|
|
9
|
+
* 4. Ranker A/B comparison produces delta
|
|
10
|
+
*/
|
|
11
|
+
var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
|
|
12
|
+
if (k2 === undefined) k2 = k;
|
|
13
|
+
var desc = Object.getOwnPropertyDescriptor(m, k);
|
|
14
|
+
if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
|
|
15
|
+
desc = { enumerable: true, get: function() { return m[k]; } };
|
|
16
|
+
}
|
|
17
|
+
Object.defineProperty(o, k2, desc);
|
|
18
|
+
}) : (function(o, m, k, k2) {
|
|
19
|
+
if (k2 === undefined) k2 = k;
|
|
20
|
+
o[k2] = m[k];
|
|
21
|
+
}));
|
|
22
|
+
var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
|
|
23
|
+
Object.defineProperty(o, "default", { enumerable: true, value: v });
|
|
24
|
+
}) : function(o, v) {
|
|
25
|
+
o["default"] = v;
|
|
26
|
+
});
|
|
27
|
+
var __importStar = (this && this.__importStar) || (function () {
|
|
28
|
+
var ownKeys = function(o) {
|
|
29
|
+
ownKeys = Object.getOwnPropertyNames || function (o) {
|
|
30
|
+
var ar = [];
|
|
31
|
+
for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
|
|
32
|
+
return ar;
|
|
33
|
+
};
|
|
34
|
+
return ownKeys(o);
|
|
35
|
+
};
|
|
36
|
+
return function (mod) {
|
|
37
|
+
if (mod && mod.__esModule) return mod;
|
|
38
|
+
var result = {};
|
|
39
|
+
if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
|
|
40
|
+
__setModuleDefault(result, mod);
|
|
41
|
+
return result;
|
|
42
|
+
};
|
|
43
|
+
})();
|
|
44
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
45
|
+
const vitest_1 = require("vitest");
|
|
46
|
+
const fs = __importStar(require("fs"));
|
|
47
|
+
const path = __importStar(require("path"));
|
|
48
|
+
const evaluation_campaign_1 = require("./evaluation-campaign");
|
|
49
|
+
const planner_telemetry_1 = require("./planner-telemetry");
|
|
50
|
+
const planner_trace_1 = require("./planner-trace");
|
|
51
|
+
// ═══════════════════════════════════════════════════════════════
|
|
52
|
+
// Failure Attribution
|
|
53
|
+
// ═══════════════════════════════════════════════════════════════
|
|
54
|
+
(0, vitest_1.describe)("Failure Attribution", () => {
|
|
55
|
+
(0, vitest_1.it)("classifies all 49 benchmark cases", async () => {
|
|
56
|
+
const attributed = await (0, evaluation_campaign_1.runFailureAttribution)();
|
|
57
|
+
(0, vitest_1.expect)(attributed.length).toBeGreaterThanOrEqual(49);
|
|
58
|
+
// Count by failure reason
|
|
59
|
+
const counts = {};
|
|
60
|
+
for (const a of attributed) {
|
|
61
|
+
counts[a.failureReason] = (counts[a.failureReason] || 0) + 1;
|
|
62
|
+
}
|
|
63
|
+
// Should have at least some successes and some failures
|
|
64
|
+
(0, vitest_1.expect)(counts["success"]).toBeGreaterThanOrEqual(1);
|
|
65
|
+
(0, vitest_1.expect)(Object.keys(counts).length).toBeGreaterThanOrEqual(2);
|
|
66
|
+
// Every attributed case has required fields
|
|
67
|
+
for (const a of attributed) {
|
|
68
|
+
(0, vitest_1.expect)(a.failureReason).toBeDefined();
|
|
69
|
+
(0, vitest_1.expect)(a.expectedRepair.length).toBeGreaterThan(0);
|
|
70
|
+
(0, vitest_1.expect)(["success", "missing_candidate", "bad_ranking", "bad_protocol_model", "goal_mismatch", "insufficient_history"]).toContain(a.failureReason);
|
|
71
|
+
}
|
|
72
|
+
}, 60000);
|
|
73
|
+
(0, vitest_1.it)("produces actionable error budget", async () => {
|
|
74
|
+
const attributed = await (0, evaluation_campaign_1.runFailureAttribution)();
|
|
75
|
+
const budget = (0, evaluation_campaign_1.computeErrorBudget)(attributed);
|
|
76
|
+
(0, vitest_1.expect)(budget.totalCases).toBeGreaterThanOrEqual(49);
|
|
77
|
+
(0, vitest_1.expect)(budget.successRate).toBeGreaterThanOrEqual(0);
|
|
78
|
+
(0, vitest_1.expect)(budget.successRate).toBeLessThanOrEqual(1);
|
|
79
|
+
(0, vitest_1.expect)(budget.recommendation.length).toBeGreaterThan(0);
|
|
80
|
+
// All failure reasons should sum to total
|
|
81
|
+
const sum = Object.values(budget.breakdown).reduce((s, v) => s + v, 0);
|
|
82
|
+
(0, vitest_1.expect)(sum).toBe(budget.totalCases);
|
|
83
|
+
(0, evaluation_campaign_1.printErrorBudget)(budget);
|
|
84
|
+
}, 60000);
|
|
85
|
+
});
|
|
86
|
+
// ═══════════════════════════════════════════════════════════════
|
|
87
|
+
// Offline Replay
|
|
88
|
+
// ═══════════════════════════════════════════════════════════════
|
|
89
|
+
const REPLAY_DIR = path.resolve(__dirname, "..", "test-evaluation-replay");
|
|
90
|
+
process.env.PROGMUNE_PROJECT_DIR = REPLAY_DIR;
|
|
91
|
+
fs.mkdirSync(REPLAY_DIR, { recursive: true });
|
|
92
|
+
fs.mkdirSync(path.join(REPLAY_DIR, ".progmune_corpus", "telemetry"), { recursive: true });
|
|
93
|
+
fs.mkdirSync(path.join(REPLAY_DIR, ".progmune_corpus", "traces"), { recursive: true });
|
|
94
|
+
function seedReplayData() {
|
|
95
|
+
const telemetry = new planner_telemetry_1.PlannerTelemetry(path.join(REPLAY_DIR, ".progmune_corpus", "telemetry", `replay-${Date.now()}.jsonl`));
|
|
96
|
+
const traceStore = new planner_trace_1.PlannerTraceStore(path.join(REPLAY_DIR, ".progmune_corpus", "traces", `replay-${Date.now()}.jsonl`));
|
|
97
|
+
// Seed: candidate A is safe (high acceptance), candidate B is fast (low acceptance)
|
|
98
|
+
const fpA = (0, planner_telemetry_1.candidateFingerprint)("FileProtocol", ["open_file", "write_file", "close_file"], "resource_leak");
|
|
99
|
+
const fpB = (0, planner_telemetry_1.candidateFingerprint)("FileProtocol", ["atomic_write"], "resource_leak");
|
|
100
|
+
// A accepted 80 times, B rejected 50 times
|
|
101
|
+
for (let i = 0; i < 80; i++) {
|
|
102
|
+
const id = telemetry.recordDecision({
|
|
103
|
+
goal: "safely write file",
|
|
104
|
+
protocol: "FileProtocol",
|
|
105
|
+
violationType: "resource_leak",
|
|
106
|
+
candidates: [
|
|
107
|
+
{ candidateId: fpA, source: "protocol", evidenceSources: ["protocol"], actions: ["open_file", "write_file", "close_file"], explanation: "safe" },
|
|
108
|
+
{ candidateId: fpB, source: "corpus", evidenceSources: ["corpus"], actions: ["atomic_write"], explanation: "fast" },
|
|
109
|
+
],
|
|
110
|
+
selectedCandidateId: fpA,
|
|
111
|
+
});
|
|
112
|
+
telemetry.recordFeedback(id, { decision: "accepted", executionResult: { success: true, violations: [] }, timestamp: Date.now() });
|
|
113
|
+
}
|
|
114
|
+
for (let i = 0; i < 50; i++) {
|
|
115
|
+
const id = telemetry.recordDecision({
|
|
116
|
+
goal: "quick write",
|
|
117
|
+
protocol: "FileProtocol",
|
|
118
|
+
violationType: "resource_leak",
|
|
119
|
+
candidates: [
|
|
120
|
+
{ candidateId: fpA, source: "protocol", evidenceSources: ["protocol"], actions: ["open_file", "write_file", "close_file"], explanation: "safe" },
|
|
121
|
+
{ candidateId: fpB, source: "corpus", evidenceSources: ["corpus"], actions: ["atomic_write"], explanation: "fast" },
|
|
122
|
+
],
|
|
123
|
+
selectedCandidateId: fpB,
|
|
124
|
+
});
|
|
125
|
+
telemetry.recordFeedback(id, { decision: "rejected", timestamp: Date.now() });
|
|
126
|
+
}
|
|
127
|
+
// Create traces where user chose A over B (original ranker put B first, user chose A)
|
|
128
|
+
for (let i = 0; i < 20; i++) {
|
|
129
|
+
traceStore.recordTrace({
|
|
130
|
+
decisionId: `pd-replay-${i}`,
|
|
131
|
+
goal: "safely write config file",
|
|
132
|
+
protocol: "FileProtocol",
|
|
133
|
+
violationType: "resource_leak",
|
|
134
|
+
candidates: [
|
|
135
|
+
{ fingerprint: fpB, source: "corpus", evidenceSources: ["corpus"], actions: ["atomic_write"], score: 0.73, rank: 1 },
|
|
136
|
+
{ fingerprint: fpA, source: "protocol", evidenceSources: ["protocol"], actions: ["open_file", "write_file", "close_file"], score: 0.68, rank: 2 },
|
|
137
|
+
],
|
|
138
|
+
selectedFingerprint: fpA, // user chose A even though B was rank-1
|
|
139
|
+
accepted: true,
|
|
140
|
+
});
|
|
141
|
+
}
|
|
142
|
+
// Traces where user chose rank-1
|
|
143
|
+
for (let i = 20; i < 30; i++) {
|
|
144
|
+
traceStore.recordTrace({
|
|
145
|
+
decisionId: `pd-replay-${i}`,
|
|
146
|
+
goal: "safely write config file",
|
|
147
|
+
protocol: "FileProtocol",
|
|
148
|
+
violationType: "resource_leak",
|
|
149
|
+
candidates: [
|
|
150
|
+
{ fingerprint: fpA, source: "protocol", evidenceSources: ["protocol"], actions: ["open_file", "write_file", "close_file"], score: 0.81, rank: 1 },
|
|
151
|
+
{ fingerprint: fpB, source: "corpus", evidenceSources: ["corpus"], actions: ["atomic_write"], score: 0.73, rank: 2 },
|
|
152
|
+
],
|
|
153
|
+
selectedFingerprint: fpA,
|
|
154
|
+
accepted: true,
|
|
155
|
+
});
|
|
156
|
+
}
|
|
157
|
+
return { telemetry, traceStore };
|
|
158
|
+
}
|
|
159
|
+
(0, vitest_1.describe)("Offline Replay", () => {
|
|
160
|
+
(0, vitest_1.it)("replays decisions and computes match rate", () => {
|
|
161
|
+
const { telemetry, traceStore } = seedReplayData();
|
|
162
|
+
const report = (0, evaluation_campaign_1.replayDecisions)(traceStore, telemetry);
|
|
163
|
+
(0, vitest_1.expect)(report.totalDecisions).toBeGreaterThanOrEqual(30);
|
|
164
|
+
(0, vitest_1.expect)(report.matchRate).toBeGreaterThanOrEqual(0);
|
|
165
|
+
(0, vitest_1.expect)(report.matchRate).toBeLessThanOrEqual(1);
|
|
166
|
+
// LearningRanker should match > 80% (A has 80 accepts vs B has 50 rejects)
|
|
167
|
+
(0, vitest_1.expect)(report.matchRate).toBeGreaterThan(0.8);
|
|
168
|
+
(0, evaluation_campaign_1.printReplayReport)(report);
|
|
169
|
+
});
|
|
170
|
+
(0, vitest_1.it)("compares rankers and shows delta", () => {
|
|
171
|
+
const { telemetry, traceStore } = seedReplayData();
|
|
172
|
+
const { baseline, learning, delta } = (0, evaluation_campaign_1.compareRankers)(traceStore, telemetry);
|
|
173
|
+
(0, vitest_1.expect)(baseline.totalDecisions).toBeGreaterThanOrEqual(30);
|
|
174
|
+
(0, vitest_1.expect)(learning.totalDecisions).toBeGreaterThanOrEqual(30);
|
|
175
|
+
(0, vitest_1.expect)(delta).toBeGreaterThan(0); // LearningRanker outperforms baseline
|
|
176
|
+
// Baseline (rank-1 = what planner showed first): B was rank-1 in 20/30 traces
|
|
177
|
+
// but user chose A. So baseline matches only when A was rank-1 (10/30 ≈ 33%)
|
|
178
|
+
(0, vitest_1.expect)(baseline.matchRate).toBeLessThan(learning.matchRate);
|
|
179
|
+
(0, evaluation_campaign_1.printRankerComparison)(baseline, learning, delta);
|
|
180
|
+
});
|
|
181
|
+
});
|