progmune-runtime 2.1.6 ā 3.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +108 -468
- package/dist/ablation-study.js +144 -0
- package/dist/ablation-study.test.js +18 -0
- package/dist/action-runtime.js +3 -1
- package/dist/active-learning.js +211 -0
- package/dist/analytics.js +139 -0
- package/dist/asset-factory.js +309 -0
- package/dist/asset-growth.js +244 -0
- package/dist/asset-promotion.js +382 -0
- package/dist/asset-quality.js +550 -0
- package/dist/audit/business-translator.js +285 -0
- package/dist/audit/cli.js +66 -0
- package/dist/audit/formatters/html.js +379 -0
- package/dist/audit/formatters/json.js +11 -0
- package/dist/audit/formatters/markdown.js +192 -0
- package/dist/audit/formatters/terminal.js +189 -0
- package/dist/audit/index.js +25 -0
- package/dist/audit/report-builder.js +318 -0
- package/dist/audit/types.js +8 -0
- package/dist/audit.js +3 -3
- package/dist/auto-benchmark-generator.js +137 -0
- package/dist/auto-benchmark-generator.test.js +45 -0
- package/dist/auto-protocol-synthesizer.js +362 -0
- package/dist/auto-protocol-synthesizer.test.js +82 -0
- package/dist/autonomous-patch.js +175 -0
- package/dist/autonomous-patch.test.js +128 -0
- package/dist/badge/badge-server.js +98 -0
- package/dist/behavior-miner.js +442 -0
- package/dist/belief-layer.js +475 -0
- package/dist/benchmark-count.js +5 -0
- package/dist/benchmark-generator.js +211 -0
- package/dist/benchmark-harness.js +201 -0
- package/dist/benchmark-pass-rate.js +7 -0
- package/dist/benchmark-report.js +8 -3
- package/dist/benchmark-save.js +14 -1
- package/dist/bootstrap-validation.js +197 -0
- package/dist/bootstrap-validation.test.js +51 -0
- package/dist/branch-ledger.js +1 -1
- package/dist/capability-gap.js +130 -0
- package/dist/certify-html.js +351 -0
- package/dist/certify.js +326 -0
- package/dist/check.js +4 -4
- package/dist/compliance-miner.js +447 -0
- package/dist/continuous-benchmark.js +194 -0
- package/dist/continuous-benchmark.test.js +116 -0
- package/dist/corpus-stats.js +173 -0
- package/dist/counterfactual-engine.js +288 -0
- package/dist/coverage-dashboard.js +109 -0
- package/dist/coverage-system.test.js +205 -0
- package/dist/cross-repo-precision.js +352 -0
- package/dist/cve-benchmark.js +180 -0
- package/dist/cve-benchmark.test.js +28 -0
- package/dist/cve-collector.js +73 -0
- package/dist/data-quality.js +141 -0
- package/dist/decision-engine.js +388 -0
- package/dist/derive-metadata.js +250 -0
- package/dist/difficulty-active.test.js +198 -0
- package/dist/difficulty-map.js +244 -0
- package/dist/discovery-analytics.js +125 -0
- package/dist/discovery-model.js +149 -0
- package/dist/discovery-optimize.test.js +199 -0
- package/dist/discovery-trace.js +276 -0
- package/dist/discovery-trace.test.js +97 -0
- package/dist/emitter.js +83 -1
- package/dist/enterprise-dashboard.js +405 -0
- package/dist/eval-hardening.js +297 -0
- package/dist/eval-hardening.test.js +85 -0
- package/dist/evaluation-campaign.js +359 -0
- package/dist/evaluation-campaign.test.js +181 -0
- package/dist/evidence-growth.js +143 -0
- package/dist/evidence-repository.js +209 -0
- package/dist/evidence-system.js +441 -0
- package/dist/execute.js +15 -7
- package/dist/experimental/software-physics.js +291 -0
- package/dist/experimental/state-inference.js +516 -0
- package/dist/experimental/unsupervised-physics.js +230 -0
- package/dist/extract-ir-python.js +54 -7
- package/dist/extract-ir.js +376 -12
- package/dist/failure-collector.js +2 -2
- package/dist/failure-corpus.js +322 -9
- package/dist/feedback.js +16 -5
- package/dist/feedback.test.js +49 -0
- package/dist/file-lock.js +1 -1
- package/dist/flywheel-batch.js +292 -0
- package/dist/frameworks/express-cli.js +237 -0
- package/dist/frameworks/express-detector.js +445 -0
- package/dist/frameworks/express-detector.test.js +206 -0
- package/dist/frameworks/index.js +30 -0
- package/dist/frameworks/nestjs-detector.js +302 -0
- package/dist/frameworks/trpc-detector.js +161 -0
- package/dist/frameworks/version-awareness.js +179 -0
- package/dist/function-synonyms.js +164 -0
- package/dist/function-synonyms.test.js +68 -0
- package/dist/generalization.test.js +352 -0
- package/dist/goal-annotator.js +113 -0
- package/dist/goal-planner.js +563 -0
- package/dist/gold-cve.js +164 -0
- package/dist/gold-cve.test.js +104 -0
- package/dist/gold-quality.js +206 -0
- package/dist/gold-tiers.js +241 -0
- package/dist/governance-dashboard.js +327 -0
- package/dist/graph-viz.js +240 -0
- package/dist/guided-frontier.js +195 -0
- package/dist/hierarchical-planner.js +148 -0
- package/dist/identifier-parser.js +260 -0
- package/dist/immune-metrics.js +93 -0
- package/dist/immune-receiver.js +158 -0
- package/dist/immune-reporter.js +1 -1
- package/dist/improvement-orchestrator.js +206 -0
- package/dist/inject-p0-vocabulary.js +300 -0
- package/dist/intent-parser.js +218 -0
- package/dist/invariant-algebra.js +476 -0
- package/dist/invariant-calculus.js +533 -0
- package/dist/ir-utils.js +70 -0
- package/dist/ir-utils.test.js +50 -0
- package/dist/knowledge-api.js +312 -0
- package/dist/knowledge-evolution.js +452 -0
- package/dist/knowledge-explorer.js +506 -0
- package/dist/knowledge-flywheel.js +274 -0
- package/dist/knowledge-governance.js +338 -0
- package/dist/knowledge-governance.test.js +150 -0
- package/dist/knowledge-graph.js +181 -0
- package/dist/knowledge-guided-synth.js +246 -0
- package/dist/knowledge-loop.test.js +77 -0
- package/dist/knowledge-object.js +316 -0
- package/dist/knowledge-package.js +98 -0
- package/dist/kpi-dashboard.js +561 -0
- package/dist/l3-cross-function.js +280 -0
- package/dist/learning-ranker.js +148 -0
- package/dist/learning-ranker.test.js +291 -0
- package/dist/ledger/accountability.js +322 -0
- package/dist/ledger/chain-builder.js +185 -0
- package/dist/ledger/cli.js +222 -0
- package/dist/ledger/index.js +13 -0
- package/dist/ledger/signatures.js +193 -0
- package/dist/ledger/types.js +9 -0
- package/dist/llm.js +74 -3
- package/dist/load-benchmarks.js +8 -3
- package/dist/logger.js +66 -0
- package/dist/logger.test.js +37 -0
- package/dist/logistic-reward.js +339 -0
- package/dist/logistic-reward.test.js +180 -0
- package/dist/macro-graph.js +193 -0
- package/dist/macro-repair.js +183 -0
- package/dist/mcp-server.mjs +1202 -483
- package/dist/memory-layer.js +42 -5
- package/dist/multi-repo-precision.js +422 -0
- package/dist/name-free-protocol.js +425 -0
- package/dist/name-free-protocol.test.js +170 -0
- package/dist/name-scrambling.js +138 -0
- package/dist/name-scrambling.test.js +16 -0
- package/dist/p3-observability.test.js +281 -0
- package/dist/p5-orchestrator.test.js +225 -0
- package/dist/pairwise-preference.js +294 -0
- package/dist/pairwise-preference.test.js +140 -0
- package/dist/planner-constraints.js +104 -0
- package/dist/planner-prompts.js +155 -0
- package/dist/planner-telemetry.js +415 -0
- package/dist/planner-trace.js +214 -0
- package/dist/planner.js +162 -167
- package/dist/plsb/artifact.js +116 -0
- package/dist/plsb/cli.js +71 -0
- package/dist/plsb/index.js +19 -0
- package/dist/plsb/leaderboard.js +249 -0
- package/dist/plsb/report-md.js +156 -0
- package/dist/plsb/schema.js +179 -0
- package/dist/plsb-benchmark.js +284 -0
- package/dist/plsb-benchmark.test.js +119 -0
- package/dist/policy/cli.js +134 -0
- package/dist/policy/engine.js +333 -0
- package/dist/policy/index.js +12 -0
- package/dist/policy/types.js +59 -0
- package/dist/policy-miner.js +505 -0
- package/dist/precision-analyze.js +229 -0
- package/dist/precision-benchmark.js +147 -0
- package/dist/precision-label-c.js +134 -0
- package/dist/precision-label.js +193 -0
- package/dist/precision-report-c.js +149 -0
- package/dist/precision-report.js +246 -0
- package/dist/progmune-status.js +108 -0
- package/dist/proof-engine.js +479 -0
- package/dist/proof-provenance.js +315 -0
- package/dist/protocol-coverage.js +294 -0
- package/dist/protocol-detector.js +1189 -0
- package/dist/protocol-embedding-expanded.js +297 -0
- package/dist/protocol-embedding-expanded.test.js +97 -0
- package/dist/protocol-embedding.js +195 -0
- package/dist/protocol-embedding.test.js +82 -0
- package/dist/protocol-extractor-v2.js +354 -0
- package/dist/protocol-extractor-v2.test.js +140 -0
- package/dist/protocol-extractor.js +310 -0
- package/dist/protocol-extractor.test.js +113 -0
- package/dist/protocol-foundation.js +322 -0
- package/dist/protocol-foundation.test.js +163 -0
- package/dist/protocol-frontier.js +243 -0
- package/dist/protocol-frontier.test.js +92 -0
- package/dist/protocol-gap-analyzer.js +228 -0
- package/dist/protocol-gap-analyzer.test.js +49 -0
- package/dist/protocol-invariants.js +276 -0
- package/dist/protocol-invariants.test.js +111 -0
- package/dist/protocol-knowledge.js +464 -0
- package/dist/protocol-miner.js +343 -0
- package/dist/protocol-mining.js +207 -0
- package/dist/protocol-mining.test.js +37 -0
- package/dist/protocol-registry.js +1 -1
- package/dist/protocol-security-benchmark.js +222 -0
- package/dist/protocol-vulnerability.js +257 -0
- package/dist/protocol-vulnerability.test.js +60 -0
- package/dist/python-benchmark.js +120 -0
- package/dist/python-emitter.js +163 -45
- package/dist/python-protocol-extractor.js +187 -0
- package/dist/python-protocol-extractor.test.js +116 -0
- package/dist/realworld-benchmark.js +646 -0
- package/dist/realworld-benchmark.test.js +36 -0
- package/dist/repair-arch.test.js +411 -0
- package/dist/repair-evolution.test.js +454 -0
- package/dist/repair-executor.js +719 -0
- package/dist/repair-proposal.js +4 -4
- package/dist/repair-ranker.js +141 -0
- package/dist/repair-strategies.js +419 -0
- package/dist/repair-taxonomy.js +234 -0
- package/dist/repair-types.js +12 -0
- package/dist/repo-evaluator.js +250 -0
- package/dist/repo-evaluator.test.js +128 -0
- package/dist/resource-abstraction.js +242 -0
- package/dist/resource-detector.js +211 -0
- package/dist/result.test.js +43 -0
- package/dist/reward-system.js +411 -0
- package/dist/reward-system.test.js +175 -0
- package/dist/risk-model.js +215 -0
- package/dist/rule-miner.js +234 -7
- package/dist/rule-specificity.js +254 -0
- package/dist/runtime-types.js +27 -0
- package/dist/scaffold.js +208 -0
- package/dist/scale-collector.test.js +101 -0
- package/dist/scale-trajectory-collector.js +128 -0
- package/dist/sdk.js +250 -0
- package/dist/search-planner.js +4 -41
- package/dist/semantic-snapshot.js +1 -1
- package/dist/semantic-topology.js +121 -0
- package/dist/semantic-trace.js +310 -317
- package/dist/sequence-extractor.js +343 -0
- package/dist/skill-library.js +245 -0
- package/dist/skill-planner.test.js +189 -0
- package/dist/software-physics.js +291 -0
- package/dist/software-physics.test.js +81 -0
- package/dist/ssg-precision.js +478 -0
- package/dist/ssg-validator.js +71 -21
- package/dist/state-inference-doubleblind.test.js +160 -0
- package/dist/state-inference.js +516 -0
- package/dist/state-inference.test.js +115 -0
- package/dist/state-machine-fingerprint.js +345 -0
- package/dist/state-machine-fingerprint.test.js +120 -0
- package/dist/state-miner.js +386 -0
- package/dist/state-name-inference.js +213 -0
- package/dist/state-name-inference.test.js +69 -0
- package/dist/strategy-planner.js +262 -96
- package/dist/strategy-planner.test.js +135 -0
- package/dist/telemetry-analytics.test.js +402 -0
- package/dist/terminal-format.js +68 -0
- package/dist/terminal-format.test.js +83 -0
- package/dist/topology-factory.js +196 -0
- package/dist/topology-representation.js +242 -0
- package/dist/topology-representation.test.js +27 -0
- package/dist/trajectory-augmentation.js +254 -0
- package/dist/trajectory-augmentation.test.js +63 -0
- package/dist/trajectory-corpus.js +440 -0
- package/dist/trajectory-corpus.test.js +32 -0
- package/dist/trajectory-feedback.test.js +116 -0
- package/dist/transition-synthesizer.js +286 -0
- package/dist/transition-synthesizer.test.js +123 -0
- package/dist/trust/api-semantic-mapper.js +809 -0
- package/dist/trust/call-graph-propagator.js +225 -0
- package/dist/trust/cli.js +122 -0
- package/dist/trust/compliance-scorer.js +283 -0
- package/dist/trust/confidence-calculator.js +261 -0
- package/dist/trust/engine.js +1145 -0
- package/dist/trust/explainability.js +85 -0
- package/dist/trust/formatters/ci.js +42 -0
- package/dist/trust/formatters/json.js +11 -0
- package/dist/trust/formatters/terminal.js +152 -0
- package/dist/trust/index.js +39 -0
- package/dist/trust/phase1-verify.js +171 -0
- package/dist/trust/protocol-domain-validator.js +697 -0
- package/dist/trust/score-calculator.js +282 -0
- package/dist/trust/ssg-bridge.js +641 -0
- package/dist/trust/ssg-bridge.test.js +269 -0
- package/dist/trust/types.js +67 -0
- package/dist/trust/violation-trace.js +335 -0
- package/dist/trust-api.js +179 -0
- package/dist/trust-calibration.js +279 -0
- package/dist/unknown-protocol-discovery.js +339 -0
- package/dist/unknown-protocol-discovery.test.js +102 -0
- package/dist/unsupervised-physics.js +230 -0
- package/dist/unsupervised-physics.test.js +95 -0
- package/dist/utils.test.js +37 -0
- package/dist/validator.js +187 -10
- package/dist/verification-intelligence.js +475 -0
- package/dist/verify-api.js +432 -0
- package/dist/vi-impact-report.js +293 -0
- package/dist/wl-fingerprint.js +162 -0
- package/dist/wl-fingerprint.test.js +130 -0
- package/dist/zeroshot-strategy.js +139 -0
- package/docs/Progmune_/346/212/225/350/265/204/344/272/272/347/231/275/347/232/256/344/271/246_v2.0.html +576 -0
- package/docs/Progmune_/351/241/271/347/233/256/345/205/250/350/247/243.html +710 -0
- package/package.json +74 -7
- package/protocols.json +1956 -50
- package/.dockerignore +0 -14
- package/.mcp.json +0 -11
- package/.progmune_allowlist +0 -50
- package/.test_report/test_report.md +0 -87
- package/Dockerfile +0 -9
- package/FAQ.md +0 -167
- package/WHITEPAPER.md +0 -540
- package/demo-project/auth.ts +0 -55
- package/demo-project/tsconfig.json +0 -8
- package/dist/acl-breakdown.js +0 -13
- package/dist/all-sessions.js +0 -11
- package/dist/antibody-stats.js +0 -11
- package/dist/branch-tree-count.js +0 -14
- package/dist/common-fixpath.js +0 -12
- package/dist/constraint-types.js +0 -12
- package/dist/exec-metrics.js +0 -11
- package/dist/failure-report.js +0 -11
- package/dist/fast-path-hits.js +0 -13
- package/dist/fingerprint-list.js +0 -15
- package/dist/gen-history-log.js +0 -13
- package/dist/heatmap-data.js +0 -11
- package/dist/recent-session.js +0 -12
- package/dist/svl-distribution.js +0 -11
- package/dist/terminal-status.js +0 -11
- package/dist/token-savings.js +0 -11
- package/dist/total-repairs.js +0 -12
- package/dist/unresolved-count.js +0 -12
- package/dist/valid-fingerprints.js +0 -13
- package/dist/verify-ledgers.js +0 -11
- package/docs/whitepaper-style.css +0 -77
- package/docs/whitepaper-v2.1.md +0 -609
- package/docs/whitepaper-v2.2.md +0 -1064
- package/docs/whitepaper-v2.2.pdf +0 -0
- package/fly.toml +0 -31
- package/public/dashboard.html +0 -119
- package/server/hub.js +0 -116
- package/test/replay-golden/sess_1780063202050_mgeld.json +0 -9
- package/test/replay-golden/sess_1780064032560_gocld.json +0 -354
- package/test/replay-golden/sess_1780064413331_s2709.json +0 -606
- package/test/replay-golden/sess_1780064792710_y3avo.json +0 -614
- package/test/replay-golden.ts +0 -84
- package/test_benchmark.js +0 -165
- package/test_comprehensive.mjs +0 -638
- package/test_concurrency.js +0 -129
- package/test_ir_robustness.js +0 -85
- package/test_semantic_contracts.js +0 -269
- package/test_ssg_stress.js +0 -156
- package/test_svl3.js +0 -58
- package/tsconfig.json +0 -17
|
@@ -0,0 +1,198 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* P3.7 + P3.8: Difficulty Map & Active Learning Tests
|
|
4
|
+
*
|
|
5
|
+
* Verifying:
|
|
6
|
+
* 1. TransitionStats computation from trajectory + telemetry data
|
|
7
|
+
* 2. Protocol difficulty ranking (critical/high/medium/low)
|
|
8
|
+
* 3. Active Learning importance scoring
|
|
9
|
+
* 4. Prioritized benchmark generation
|
|
10
|
+
* 5. End-to-end: difficulty ā importance ā prioritized cases
|
|
11
|
+
*/
|
|
12
|
+
var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
|
|
13
|
+
if (k2 === undefined) k2 = k;
|
|
14
|
+
var desc = Object.getOwnPropertyDescriptor(m, k);
|
|
15
|
+
if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
|
|
16
|
+
desc = { enumerable: true, get: function() { return m[k]; } };
|
|
17
|
+
}
|
|
18
|
+
Object.defineProperty(o, k2, desc);
|
|
19
|
+
}) : (function(o, m, k, k2) {
|
|
20
|
+
if (k2 === undefined) k2 = k;
|
|
21
|
+
o[k2] = m[k];
|
|
22
|
+
}));
|
|
23
|
+
var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
|
|
24
|
+
Object.defineProperty(o, "default", { enumerable: true, value: v });
|
|
25
|
+
}) : function(o, v) {
|
|
26
|
+
o["default"] = v;
|
|
27
|
+
});
|
|
28
|
+
var __importStar = (this && this.__importStar) || (function () {
|
|
29
|
+
var ownKeys = function(o) {
|
|
30
|
+
ownKeys = Object.getOwnPropertyNames || function (o) {
|
|
31
|
+
var ar = [];
|
|
32
|
+
for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
|
|
33
|
+
return ar;
|
|
34
|
+
};
|
|
35
|
+
return ownKeys(o);
|
|
36
|
+
};
|
|
37
|
+
return function (mod) {
|
|
38
|
+
if (mod && mod.__esModule) return mod;
|
|
39
|
+
var result = {};
|
|
40
|
+
if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
|
|
41
|
+
__setModuleDefault(result, mod);
|
|
42
|
+
return result;
|
|
43
|
+
};
|
|
44
|
+
})();
|
|
45
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
46
|
+
const vitest_1 = require("vitest");
|
|
47
|
+
const fs = __importStar(require("fs"));
|
|
48
|
+
const path = __importStar(require("path"));
|
|
49
|
+
const difficulty_map_1 = require("./difficulty-map");
|
|
50
|
+
const active_learning_1 = require("./active-learning");
|
|
51
|
+
// āāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāā
|
|
52
|
+
// Difficulty Map
|
|
53
|
+
// āāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāā
|
|
54
|
+
function makeTrajectory(overrides = {}) {
|
|
55
|
+
return {
|
|
56
|
+
id: `t-${Date.now()}-${Math.random().toString(36).slice(2, 6)}`,
|
|
57
|
+
timestamp: new Date().toISOString(),
|
|
58
|
+
protocol: "_global",
|
|
59
|
+
initialState: [],
|
|
60
|
+
finalState: [],
|
|
61
|
+
trajectory: ["open_file", "write_file", "close_file"],
|
|
62
|
+
result: "success",
|
|
63
|
+
context: { nestingDepth: 0, exceptionHandled: false, insideLoop: false, branchCount: 0, asyncContext: false },
|
|
64
|
+
successRate: 1.0,
|
|
65
|
+
metadata: { source: "human" },
|
|
66
|
+
...overrides,
|
|
67
|
+
};
|
|
68
|
+
}
|
|
69
|
+
(0, vitest_1.describe)("Difficulty Map", () => {
|
|
70
|
+
(0, vitest_1.it)("computes transition stats from trajectories", () => {
|
|
71
|
+
const trajectories = [
|
|
72
|
+
makeTrajectory({ result: "success", trajectory: ["open_file", "write_file", "close_file"] }),
|
|
73
|
+
makeTrajectory({ result: "success", trajectory: ["open_file", "write_file", "close_file"] }),
|
|
74
|
+
makeTrajectory({ result: "violation", trajectory: ["open_file", "write_file"] }), // missing close
|
|
75
|
+
makeTrajectory({
|
|
76
|
+
result: "repair", trajectory: ["open_file", "write_file", "close_file"],
|
|
77
|
+
successRate: 1.0,
|
|
78
|
+
violation: { type: "resource_leak", failingStepIndex: 2, expectedStates: [], actualStates: ["FILE_OPEN"], fixPath: ["close_file"], description: "fix" },
|
|
79
|
+
}),
|
|
80
|
+
];
|
|
81
|
+
const statsMap = (0, difficulty_map_1.buildDifficultyMap)(trajectories);
|
|
82
|
+
// FileProtocol should have stats
|
|
83
|
+
const fileKeys = [...statsMap.keys()].filter(k => k.startsWith("FileProtocol:"));
|
|
84
|
+
(0, vitest_1.expect)(fileKeys.length).toBeGreaterThan(0);
|
|
85
|
+
// open_file transition should have attempts
|
|
86
|
+
const openKey = "FileProtocol:INITāFILE_OPEN";
|
|
87
|
+
const openStats = statsMap.get(openKey);
|
|
88
|
+
(0, vitest_1.expect)(openStats).toBeDefined();
|
|
89
|
+
(0, vitest_1.expect)(openStats.attempts).toBeGreaterThanOrEqual(3);
|
|
90
|
+
(0, vitest_1.expect)(openStats.failures).toBeGreaterThanOrEqual(1); // the violation
|
|
91
|
+
// close_file invalidation should be tracked
|
|
92
|
+
const closeInvKey = "FileProtocol:FILE_OPENāā
";
|
|
93
|
+
const closeStats = statsMap.get(closeInvKey);
|
|
94
|
+
(0, vitest_1.expect)(closeStats).toBeDefined();
|
|
95
|
+
});
|
|
96
|
+
(0, vitest_1.it)("difficulty > 0 for transitions with failures", () => {
|
|
97
|
+
const trajectories = [
|
|
98
|
+
makeTrajectory({ result: "success", trajectory: ["verify_password", "generate_jwt", "create_session"] }),
|
|
99
|
+
makeTrajectory({ result: "violation", trajectory: ["verify_password"] }), // missing jwt
|
|
100
|
+
makeTrajectory({ result: "violation", trajectory: ["verify_password"] }),
|
|
101
|
+
makeTrajectory({ result: "violation", trajectory: ["verify_password"] }),
|
|
102
|
+
];
|
|
103
|
+
const statsMap = (0, difficulty_map_1.buildDifficultyMap)(trajectories);
|
|
104
|
+
// The failures are on verify_password transitions (missing the rest)
|
|
105
|
+
// UNAUTHENTICATEDāPASSWORD_VERIFIED should have failures from the violations
|
|
106
|
+
const vpKey = "AuthProtocol:UNAUTHENTICATEDāPASSWORD_VERIFIED";
|
|
107
|
+
const vpStats = statsMap.get(vpKey);
|
|
108
|
+
(0, vitest_1.expect)(vpStats).toBeDefined();
|
|
109
|
+
(0, vitest_1.expect)(vpStats.attempts).toBeGreaterThanOrEqual(4); // 1 success + 3 violations
|
|
110
|
+
(0, vitest_1.expect)(vpStats.failures).toBeGreaterThanOrEqual(3);
|
|
111
|
+
(0, vitest_1.expect)(vpStats.difficulty).toBeGreaterThan(0);
|
|
112
|
+
});
|
|
113
|
+
(0, vitest_1.it)("ranks protocols by difficulty", () => {
|
|
114
|
+
const trajectories = [
|
|
115
|
+
// FileProtocol: mostly successes
|
|
116
|
+
...Array.from({ length: 10 }, () => makeTrajectory({ result: "success", trajectory: ["open_file", "write_file", "close_file"] })),
|
|
117
|
+
// AuthProtocol: many failures
|
|
118
|
+
makeTrajectory({ result: "violation", trajectory: ["verify_password"] }),
|
|
119
|
+
makeTrajectory({ result: "violation", trajectory: ["verify_password"] }),
|
|
120
|
+
makeTrajectory({ result: "violation", trajectory: ["generate_jwt"] }),
|
|
121
|
+
];
|
|
122
|
+
const statsMap = (0, difficulty_map_1.buildDifficultyMap)(trajectories);
|
|
123
|
+
const ranking = (0, difficulty_map_1.rankProtocolsByDifficulty)(statsMap);
|
|
124
|
+
(0, vitest_1.expect)(ranking.length).toBeGreaterThanOrEqual(4); // P7.3: 9 protocol groups
|
|
125
|
+
// AuthProtocol should be highest difficulty
|
|
126
|
+
const auth = ranking.find(r => r.protocol === "AuthProtocol");
|
|
127
|
+
const file = ranking.find(r => r.protocol === "FileProtocol");
|
|
128
|
+
(0, vitest_1.expect)(auth.maxDifficulty).toBeGreaterThan(file.maxDifficulty);
|
|
129
|
+
(0, difficulty_map_1.printDifficultyDashboard)(statsMap, ranking);
|
|
130
|
+
});
|
|
131
|
+
(0, vitest_1.it)("empty data = all zeros", () => {
|
|
132
|
+
const statsMap = (0, difficulty_map_1.buildDifficultyMap)([]);
|
|
133
|
+
const ranking = (0, difficulty_map_1.rankProtocolsByDifficulty)(statsMap);
|
|
134
|
+
for (const r of ranking) {
|
|
135
|
+
(0, vitest_1.expect)(r.avgDifficulty).toBe(0);
|
|
136
|
+
(0, vitest_1.expect)(r.maxDifficulty).toBe(0);
|
|
137
|
+
(0, vitest_1.expect)(r.risk).toBe("low");
|
|
138
|
+
}
|
|
139
|
+
});
|
|
140
|
+
});
|
|
141
|
+
// āāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāā
|
|
142
|
+
// Active Learning
|
|
143
|
+
// āāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāā
|
|
144
|
+
(0, vitest_1.describe)("Active Learning", () => {
|
|
145
|
+
(0, vitest_1.it)("prioritizes gaps by importance (difficulty Ć usage Ć failure)", () => {
|
|
146
|
+
// Seed with skewed data: FileProtocol is easy, AuthProtocol is hard
|
|
147
|
+
const trajectories = [
|
|
148
|
+
// File: many successes, few failures ā low difficulty
|
|
149
|
+
...Array.from({ length: 20 }, () => makeTrajectory({ result: "success", trajectory: ["open_file", "write_file", "close_file"] })),
|
|
150
|
+
// Auth: many failures ā high difficulty
|
|
151
|
+
...Array.from({ length: 5 }, () => makeTrajectory({ result: "violation", trajectory: ["verify_password"] })),
|
|
152
|
+
];
|
|
153
|
+
const report = (0, active_learning_1.generatePrioritizedBenchmarks)(trajectories);
|
|
154
|
+
(0, vitest_1.expect)(report.totalGaps).toBeGreaterThan(0);
|
|
155
|
+
(0, vitest_1.expect)(report.prioritized.length).toBeGreaterThan(0);
|
|
156
|
+
// Auth transitions should have higher importance than File transitions
|
|
157
|
+
// (auth has violations ā higher difficulty)
|
|
158
|
+
const authCases = report.prioritized.filter(c => c.targetsTransition.rule.includes("password") || c.targetsTransition.rule.includes("jwt") || c.targetsTransition.rule.includes("session"));
|
|
159
|
+
const fileCases = report.prioritized.filter(c => c.targetsTransition.rule.includes("file") || c.targetsTransition.rule.includes("open") || c.targetsTransition.rule.includes("write") || c.targetsTransition.rule.includes("close"));
|
|
160
|
+
if (authCases.length > 0 && fileCases.length > 0) {
|
|
161
|
+
// Auth should have non-zero difficulty (has violations)
|
|
162
|
+
const authHasDifficulty = authCases.some(c => c.difficulty > 0);
|
|
163
|
+
(0, vitest_1.expect)(authHasDifficulty).toBe(true);
|
|
164
|
+
}
|
|
165
|
+
(0, active_learning_1.printActiveLearningReport)(report);
|
|
166
|
+
});
|
|
167
|
+
(0, vitest_1.it)("writes top-K priority benchmarks", () => {
|
|
168
|
+
const trajectories = [
|
|
169
|
+
...Array.from({ length: 30 }, () => makeTrajectory({ result: "success", trajectory: ["open_file", "write_file", "close_file"] })),
|
|
170
|
+
];
|
|
171
|
+
const report = (0, active_learning_1.generatePrioritizedBenchmarks)(trajectories);
|
|
172
|
+
const outDir = path.resolve(__dirname, "..", "test-active-learning");
|
|
173
|
+
const written = (0, active_learning_1.writeTopPriorityBenchmarks)(report, 8, outDir);
|
|
174
|
+
(0, vitest_1.expect)(written.length).toBeGreaterThanOrEqual(1);
|
|
175
|
+
// Verify files are valid JSON with importance scores
|
|
176
|
+
for (const fp of written) {
|
|
177
|
+
(0, vitest_1.expect)(fs.existsSync(fp)).toBe(true);
|
|
178
|
+
const content = JSON.parse(fs.readFileSync(fp, "utf-8"));
|
|
179
|
+
(0, vitest_1.expect)(content.source).toBe("active-learning");
|
|
180
|
+
(0, vitest_1.expect)(content.cases.length).toBeGreaterThan(0);
|
|
181
|
+
(0, vitest_1.expect)(content.cases.length).toBeLessThanOrEqual(8);
|
|
182
|
+
for (const c of content.cases) {
|
|
183
|
+
(0, vitest_1.expect)(c.importance).toBeGreaterThanOrEqual(0);
|
|
184
|
+
(0, vitest_1.expect)(c.difficulty).toBeGreaterThanOrEqual(0);
|
|
185
|
+
}
|
|
186
|
+
}
|
|
187
|
+
});
|
|
188
|
+
(0, vitest_1.it)("importance = 0 when no data (all gaps equal priority)", () => {
|
|
189
|
+
const report = (0, active_learning_1.generatePrioritizedBenchmarks)([]);
|
|
190
|
+
// With no data, all gaps have equal importance
|
|
191
|
+
// They're still generated, just not differentiated
|
|
192
|
+
(0, vitest_1.expect)(report.totalGaps).toBeGreaterThan(0);
|
|
193
|
+
for (const c of report.prioritized) {
|
|
194
|
+
(0, vitest_1.expect)(c.importance).toBeGreaterThanOrEqual(0);
|
|
195
|
+
(0, vitest_1.expect)(c.importance).toBeLessThanOrEqual(1);
|
|
196
|
+
}
|
|
197
|
+
});
|
|
198
|
+
});
|
|
@@ -0,0 +1,244 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* P3.7: Protocol Difficulty Map
|
|
4
|
+
*
|
|
5
|
+
* Answers: which transitions are hardest to learn?
|
|
6
|
+
*
|
|
7
|
+
* Uses Telemetry + Trajectory Corpus to compute per-transition statistics:
|
|
8
|
+
* - How often does this transition appear in trajectories?
|
|
9
|
+
* - When it fails, how often is it successfully repaired?
|
|
10
|
+
* - What's the acceptance rate for repairs involving this transition?
|
|
11
|
+
*
|
|
12
|
+
* The difficulty score feeds into:
|
|
13
|
+
* 1. Active Learning (prioritize hard transitions for benchmark generation)
|
|
14
|
+
* 2. Reward Model (weight training samples by difficulty)
|
|
15
|
+
* 3. Coverage System (focus data acquisition on high-difficulty gaps)
|
|
16
|
+
*/
|
|
17
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
18
|
+
exports.buildDifficultyMap = buildDifficultyMap;
|
|
19
|
+
exports.rankProtocolsByDifficulty = rankProtocolsByDifficulty;
|
|
20
|
+
exports.printDifficultyDashboard = printDifficultyDashboard;
|
|
21
|
+
const protocol_coverage_1 = require("./protocol-coverage");
|
|
22
|
+
// āāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāā
|
|
23
|
+
// Difficulty Computation
|
|
24
|
+
// āāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāā
|
|
25
|
+
/**
|
|
26
|
+
* Compute difficulty for a single transition.
|
|
27
|
+
*
|
|
28
|
+
* difficulty = (failureRate Ć 0.4) + (repairFailureRate Ć 0.4) + (rejectionRate Ć 0.2)
|
|
29
|
+
*
|
|
30
|
+
* Where:
|
|
31
|
+
* - failureRate = failures / attempts (how often does it go wrong?)
|
|
32
|
+
* - repairFailureRate = 1 - (repairSuccesses / repairs) (when fixed, how often does the fix fail?)
|
|
33
|
+
* - rejectionRate = 1 - acceptanceRate (how often do users reject the repair?)
|
|
34
|
+
*
|
|
35
|
+
* A transition that always fails AND repairs never work AND users always reject = difficulty 1.0.
|
|
36
|
+
* A transition that never fails = difficulty 0.0.
|
|
37
|
+
*/
|
|
38
|
+
function computeDifficulty(stats) {
|
|
39
|
+
if (stats.attempts === 0)
|
|
40
|
+
return 0;
|
|
41
|
+
const failureRate = stats.failures / stats.attempts;
|
|
42
|
+
const repairFailureRate = stats.repairs > 0
|
|
43
|
+
? 1 - (stats.repairSuccesses / stats.repairs)
|
|
44
|
+
: 0;
|
|
45
|
+
const rejectionRate = stats.repairs > 0
|
|
46
|
+
? 1 - stats.acceptanceRate
|
|
47
|
+
: 0;
|
|
48
|
+
return failureRate * 0.4 + repairFailureRate * 0.4 + rejectionRate * 0.2;
|
|
49
|
+
}
|
|
50
|
+
function emptyStats(protocol, t) {
|
|
51
|
+
return {
|
|
52
|
+
protocol,
|
|
53
|
+
transition: `${t.from}ā${t.to}`,
|
|
54
|
+
rule: t.rule,
|
|
55
|
+
type: t.type,
|
|
56
|
+
attempts: 0, successes: 0, failures: 0,
|
|
57
|
+
repairs: 0, repairSuccesses: 0,
|
|
58
|
+
acceptanceRate: 0, repairRate: 0, avgRepairSteps: 0, avgLatency: 0,
|
|
59
|
+
difficulty: 0,
|
|
60
|
+
};
|
|
61
|
+
}
|
|
62
|
+
/** Extract transition keys that appear in a trajectory. */
|
|
63
|
+
function transitionsInTrajectory(traj, proto) {
|
|
64
|
+
const keys = [];
|
|
65
|
+
let current = new Set();
|
|
66
|
+
if (proto.initialState)
|
|
67
|
+
current.add(proto.initialState);
|
|
68
|
+
for (const fn of traj) {
|
|
69
|
+
const rule = proto.rules.get(fn);
|
|
70
|
+
if (!rule)
|
|
71
|
+
continue;
|
|
72
|
+
for (const pre of (rule.pre_states.length > 0 ? rule.pre_states : ["INIT"])) {
|
|
73
|
+
for (const post of rule.post_states) {
|
|
74
|
+
keys.push(`${pre}ā${post}`);
|
|
75
|
+
}
|
|
76
|
+
}
|
|
77
|
+
if (rule.invalidate) {
|
|
78
|
+
for (const inv of rule.invalidate) {
|
|
79
|
+
keys.push(`${inv}āā
`);
|
|
80
|
+
}
|
|
81
|
+
}
|
|
82
|
+
if (rule.invalidate)
|
|
83
|
+
rule.invalidate.forEach(s => current.delete(s));
|
|
84
|
+
for (const post of rule.post_states)
|
|
85
|
+
current.add(post);
|
|
86
|
+
}
|
|
87
|
+
return keys;
|
|
88
|
+
}
|
|
89
|
+
/**
|
|
90
|
+
* Build a difficulty map from trajectory data and telemetry decisions.
|
|
91
|
+
*/
|
|
92
|
+
function buildDifficultyMap(trajectories, decisions) {
|
|
93
|
+
const protocols = (0, protocol_coverage_1.loadDefaultProtocolDefinitions)();
|
|
94
|
+
const statsMap = new Map();
|
|
95
|
+
// Initialize all transitions with empty stats
|
|
96
|
+
for (const proto of protocols) {
|
|
97
|
+
for (const t of proto.transitions) {
|
|
98
|
+
const key = `${proto.name}:${t.from}ā${t.to}`;
|
|
99
|
+
statsMap.set(key, emptyStats(proto.name, t));
|
|
100
|
+
}
|
|
101
|
+
}
|
|
102
|
+
// Count from trajectories
|
|
103
|
+
for (const traj of trajectories) {
|
|
104
|
+
// Find matching protocol
|
|
105
|
+
for (const proto of protocols) {
|
|
106
|
+
const tKeys = transitionsInTrajectory(traj.trajectory, proto);
|
|
107
|
+
if (tKeys.length === 0)
|
|
108
|
+
continue;
|
|
109
|
+
const isViolation = traj.result === "violation";
|
|
110
|
+
const isRepair = traj.result === "repair";
|
|
111
|
+
const repairSuccess = isRepair && traj.successRate >= 0.5;
|
|
112
|
+
for (const tKey of tKeys) {
|
|
113
|
+
const key = `${proto.name}:${tKey}`;
|
|
114
|
+
const stats = statsMap.get(key);
|
|
115
|
+
if (!stats)
|
|
116
|
+
continue;
|
|
117
|
+
stats.attempts++;
|
|
118
|
+
if (isViolation || (isRepair && !repairSuccess))
|
|
119
|
+
stats.failures++;
|
|
120
|
+
else
|
|
121
|
+
stats.successes++;
|
|
122
|
+
if (isRepair) {
|
|
123
|
+
stats.repairs++;
|
|
124
|
+
if (repairSuccess)
|
|
125
|
+
stats.repairSuccesses++;
|
|
126
|
+
if (traj.violation?.fixPath) {
|
|
127
|
+
const totalSteps = (stats.avgRepairSteps * (stats.repairs - 1) + traj.violation.fixPath.length) / stats.repairs;
|
|
128
|
+
stats.avgRepairSteps = totalSteps;
|
|
129
|
+
}
|
|
130
|
+
}
|
|
131
|
+
if (traj.cost?.latency) {
|
|
132
|
+
const totalLat = (stats.avgLatency * (stats.attempts - 1) + traj.cost.latency) / stats.attempts;
|
|
133
|
+
stats.avgLatency = totalLat;
|
|
134
|
+
}
|
|
135
|
+
}
|
|
136
|
+
}
|
|
137
|
+
}
|
|
138
|
+
// Incorporate telemetry acceptance data if available
|
|
139
|
+
if (decisions) {
|
|
140
|
+
for (const d of decisions) {
|
|
141
|
+
if (!d.feedback || !d.selectedCandidateId)
|
|
142
|
+
continue;
|
|
143
|
+
const sel = d.candidates.find(c => c.candidateId === d.selectedCandidateId);
|
|
144
|
+
if (!sel)
|
|
145
|
+
continue;
|
|
146
|
+
// Map candidate actions to transitions
|
|
147
|
+
for (const proto of protocols) {
|
|
148
|
+
const tKeys = transitionsInTrajectory(sel.actions, proto);
|
|
149
|
+
for (const tKey of tKeys) {
|
|
150
|
+
const key = `${proto.name}:${tKey}`;
|
|
151
|
+
const stats = statsMap.get(key);
|
|
152
|
+
if (!stats)
|
|
153
|
+
continue;
|
|
154
|
+
if (stats.repairs > 0) {
|
|
155
|
+
const accepted = d.feedback.decision === "accepted" ? 1 : 0;
|
|
156
|
+
stats.acceptanceRate = (stats.acceptanceRate * (stats.repairs - 1) + accepted) / stats.repairs;
|
|
157
|
+
}
|
|
158
|
+
}
|
|
159
|
+
}
|
|
160
|
+
}
|
|
161
|
+
}
|
|
162
|
+
// Compute difficulty for each
|
|
163
|
+
for (const [key, stats] of statsMap) {
|
|
164
|
+
stats.difficulty = computeDifficulty(stats);
|
|
165
|
+
if (stats.failures > 0 && stats.repairs > 0) {
|
|
166
|
+
stats.repairRate = stats.repairSuccesses / stats.repairs;
|
|
167
|
+
}
|
|
168
|
+
}
|
|
169
|
+
return statsMap;
|
|
170
|
+
}
|
|
171
|
+
// āāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāā
|
|
172
|
+
// Protocol Aggregation
|
|
173
|
+
// āāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāā
|
|
174
|
+
function rankProtocolsByDifficulty(statsMap) {
|
|
175
|
+
const protocols = (0, protocol_coverage_1.loadDefaultProtocolDefinitions)();
|
|
176
|
+
const result = [];
|
|
177
|
+
for (const proto of protocols) {
|
|
178
|
+
const entries = [];
|
|
179
|
+
for (const [key, stats] of statsMap) {
|
|
180
|
+
if (stats.protocol === proto.name)
|
|
181
|
+
entries.push(stats);
|
|
182
|
+
}
|
|
183
|
+
if (entries.length === 0) {
|
|
184
|
+
result.push({
|
|
185
|
+
protocol: proto.name, transitionCount: 0, avgDifficulty: 0,
|
|
186
|
+
maxDifficulty: 0, hardestTransition: "N/A", risk: "low",
|
|
187
|
+
});
|
|
188
|
+
continue;
|
|
189
|
+
}
|
|
190
|
+
const difficulties = entries.map(e => e.difficulty);
|
|
191
|
+
const avg = difficulties.reduce((s, d) => s + d, 0) / difficulties.length;
|
|
192
|
+
const max = Math.max(...difficulties);
|
|
193
|
+
const hardest = entries.find(e => e.difficulty === max);
|
|
194
|
+
let risk;
|
|
195
|
+
if (max > 0.5)
|
|
196
|
+
risk = "critical";
|
|
197
|
+
else if (max > 0.3)
|
|
198
|
+
risk = "high";
|
|
199
|
+
else if (max > 0.1)
|
|
200
|
+
risk = "medium";
|
|
201
|
+
else
|
|
202
|
+
risk = "low";
|
|
203
|
+
result.push({
|
|
204
|
+
protocol: proto.name,
|
|
205
|
+
transitionCount: entries.length,
|
|
206
|
+
avgDifficulty: avg,
|
|
207
|
+
maxDifficulty: max,
|
|
208
|
+
hardestTransition: hardest.transition,
|
|
209
|
+
risk,
|
|
210
|
+
});
|
|
211
|
+
}
|
|
212
|
+
return result.sort((a, b) => b.maxDifficulty - a.maxDifficulty);
|
|
213
|
+
}
|
|
214
|
+
// āāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāā
|
|
215
|
+
// Dashboard
|
|
216
|
+
// āāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāā
|
|
217
|
+
function printDifficultyDashboard(statsMap, ranking) {
|
|
218
|
+
console.log("\nāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāā");
|
|
219
|
+
console.log("ā Protocol Difficulty Map ā");
|
|
220
|
+
console.log("āāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāā\n");
|
|
221
|
+
console.log("āāā Protocol Difficulty Ranking āāā");
|
|
222
|
+
console.log("Protocol AvgDiff MaxDiff HardestTransition");
|
|
223
|
+
console.log("āāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāā");
|
|
224
|
+
for (const r of ranking) {
|
|
225
|
+
const icon = r.risk === "critical" ? "š“" : r.risk === "high" ? "š " : r.risk === "medium" ? "š”" : "š¢";
|
|
226
|
+
const avg = (r.avgDifficulty * 100).toFixed(0).padStart(3);
|
|
227
|
+
const max = (r.maxDifficulty * 100).toFixed(0).padStart(3);
|
|
228
|
+
console.log(` ${r.protocol.padEnd(16)} ${avg}% ${max}% ${r.hardestTransition} ${icon}`);
|
|
229
|
+
}
|
|
230
|
+
console.log();
|
|
231
|
+
// Detail: hardest transitions
|
|
232
|
+
const hardTransitions = [...statsMap.values()]
|
|
233
|
+
.filter(s => s.difficulty > 0.3 && s.attempts > 0)
|
|
234
|
+
.sort((a, b) => b.difficulty - a.difficulty)
|
|
235
|
+
.slice(0, 10);
|
|
236
|
+
if (hardTransitions.length > 0) {
|
|
237
|
+
console.log("āāā Hardest Transitions (top 10) āāā");
|
|
238
|
+
for (const t of hardTransitions) {
|
|
239
|
+
const d = (t.difficulty * 100).toFixed(0);
|
|
240
|
+
console.log(` ${t.protocol.padEnd(16)} ${t.transition.padEnd(30)} diff=${d}% attempts=${t.attempts} failures=${t.failures} repairs=${t.repairs}`);
|
|
241
|
+
}
|
|
242
|
+
console.log();
|
|
243
|
+
}
|
|
244
|
+
}
|
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* P4.7: Discovery Analytics
|
|
4
|
+
*
|
|
5
|
+
* Makes DiscoveryRate a first-class KPI alongside Top-1/Top-3.
|
|
6
|
+
*
|
|
7
|
+
* Discovery Rate = fraction of benchmark cases where at least one
|
|
8
|
+
* correct candidate was found (regardless of ranking).
|
|
9
|
+
*
|
|
10
|
+
* Tracks:
|
|
11
|
+
* - Discovery Rate by Protocol
|
|
12
|
+
* - Discovery Rate by Violation Type
|
|
13
|
+
* - Discovery Rate by Goal
|
|
14
|
+
* - Trend over time (with timestamps)
|
|
15
|
+
*/
|
|
16
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
17
|
+
exports.computeDiscoveryMetrics = computeDiscoveryMetrics;
|
|
18
|
+
exports.generateFullAnalyticsReport = generateFullAnalyticsReport;
|
|
19
|
+
exports.printDiscoveryDashboard = printDiscoveryDashboard;
|
|
20
|
+
const evaluation_campaign_1 = require("./evaluation-campaign");
|
|
21
|
+
const macro_repair_1 = require("./macro-repair");
|
|
22
|
+
/**
|
|
23
|
+
* Compute discovery metrics from benchmark attributions.
|
|
24
|
+
*
|
|
25
|
+
* "Discovered" = at least one candidate was found (failureReason != "missing_candidate").
|
|
26
|
+
*/
|
|
27
|
+
function computeDiscoveryMetrics(attributed) {
|
|
28
|
+
const total = attributed.length;
|
|
29
|
+
const byProtocol = {};
|
|
30
|
+
const byViolation = {};
|
|
31
|
+
const byGoal = {};
|
|
32
|
+
for (const a of attributed) {
|
|
33
|
+
const discovered = a.failureReason !== "missing_candidate" && a.failureReason !== "bad_protocol_model";
|
|
34
|
+
// By protocol
|
|
35
|
+
const proto = a.protocol || "unknown";
|
|
36
|
+
if (!byProtocol[proto])
|
|
37
|
+
byProtocol[proto] = { total: 0, discovered: 0 };
|
|
38
|
+
byProtocol[proto].total++;
|
|
39
|
+
if (discovered)
|
|
40
|
+
byProtocol[proto].discovered++;
|
|
41
|
+
// By violation
|
|
42
|
+
const viol = a.violationType || "unknown";
|
|
43
|
+
if (!byViolation[viol])
|
|
44
|
+
byViolation[viol] = { total: 0, discovered: 0 };
|
|
45
|
+
byViolation[viol].total++;
|
|
46
|
+
if (discovered)
|
|
47
|
+
byViolation[viol].discovered++;
|
|
48
|
+
// By goal (first 2 words)
|
|
49
|
+
const goalKey = a.goal.split(" ").slice(0, 3).join(" ");
|
|
50
|
+
if (!byGoal[goalKey])
|
|
51
|
+
byGoal[goalKey] = { total: 0, discovered: 0 };
|
|
52
|
+
byGoal[goalKey].total++;
|
|
53
|
+
if (discovered)
|
|
54
|
+
byGoal[goalKey].discovered++;
|
|
55
|
+
}
|
|
56
|
+
const toRates = (rec) => Object.fromEntries(Object.entries(rec).map(([k, v]) => [k, v.total > 0 ? v.discovered / v.total : 0]));
|
|
57
|
+
return {
|
|
58
|
+
overall: total > 0 ? attributed.filter(a => a.failureReason !== "missing_candidate").length / total : 0,
|
|
59
|
+
byProtocol: toRates(byProtocol),
|
|
60
|
+
byViolation: toRates(byViolation),
|
|
61
|
+
byGoal: toRates(byGoal),
|
|
62
|
+
totalCases: total,
|
|
63
|
+
};
|
|
64
|
+
}
|
|
65
|
+
async function generateFullAnalyticsReport(telemetry) {
|
|
66
|
+
const attributed = await (0, evaluation_campaign_1.runFailureAttribution)();
|
|
67
|
+
const discovery = computeDiscoveryMetrics(attributed);
|
|
68
|
+
const macros = (0, macro_repair_1.mineMacroRepairs)(telemetry);
|
|
69
|
+
const missing = attributed.filter(a => a.failureReason === "missing_candidate").length;
|
|
70
|
+
const ranking = attributed.filter(a => a.failureReason === "bad_ranking").length;
|
|
71
|
+
const success = attributed.filter(a => a.failureReason === "success").length;
|
|
72
|
+
const total = attributed.length;
|
|
73
|
+
return {
|
|
74
|
+
timestamp: new Date().toISOString(),
|
|
75
|
+
discovery,
|
|
76
|
+
macroCount: macros.length,
|
|
77
|
+
topMacros: macros.slice(0, 5),
|
|
78
|
+
errorBudget: {
|
|
79
|
+
missingPct: total > 0 ? missing / total : 0,
|
|
80
|
+
rankingPct: total > 0 ? ranking / total : 0,
|
|
81
|
+
successPct: total > 0 ? success / total : 0,
|
|
82
|
+
},
|
|
83
|
+
};
|
|
84
|
+
}
|
|
85
|
+
function printDiscoveryDashboard(report) {
|
|
86
|
+
console.log("\nāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāā");
|
|
87
|
+
console.log("ā Discovery Analytics Dashboard ā");
|
|
88
|
+
console.log("āāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāā\n");
|
|
89
|
+
console.log(`Timestamp: ${report.timestamp}`);
|
|
90
|
+
console.log(`Total Cases: ${report.discovery.totalCases}`);
|
|
91
|
+
console.log();
|
|
92
|
+
// Discovery Rate
|
|
93
|
+
const dr = (report.discovery.overall * 100).toFixed(0);
|
|
94
|
+
console.log(`ā Discovery Rate: ${dr}%`);
|
|
95
|
+
console.log();
|
|
96
|
+
// Error Budget
|
|
97
|
+
console.log("āāā Error Budget āāā");
|
|
98
|
+
console.log(` Missing Candidate: ${(report.errorBudget.missingPct * 100).toFixed(0)}%`);
|
|
99
|
+
console.log(` Bad Ranking: ${(report.errorBudget.rankingPct * 100).toFixed(0)}%`);
|
|
100
|
+
console.log(` Success: ${(report.errorBudget.successPct * 100).toFixed(0)}%`);
|
|
101
|
+
console.log();
|
|
102
|
+
// By Protocol
|
|
103
|
+
console.log("āāā Discovery by Protocol āāā");
|
|
104
|
+
for (const [proto, rate] of Object.entries(report.discovery.byProtocol).sort((a, b) => b[1] - a[1])) {
|
|
105
|
+
const pct = (rate * 100).toFixed(0).padStart(3);
|
|
106
|
+
const bar = "ā".repeat(Math.round(rate * 20));
|
|
107
|
+
console.log(` ${proto.padEnd(16)} ${pct}% ${bar}`);
|
|
108
|
+
}
|
|
109
|
+
console.log();
|
|
110
|
+
// By Violation
|
|
111
|
+
console.log("āāā Discovery by Violation āāā");
|
|
112
|
+
for (const [viol, rate] of Object.entries(report.discovery.byViolation).sort((a, b) => b[1] - a[1])) {
|
|
113
|
+
console.log(` ${viol.padEnd(22)} ${(rate * 100).toFixed(0)}%`);
|
|
114
|
+
}
|
|
115
|
+
console.log();
|
|
116
|
+
// Macros
|
|
117
|
+
if (report.topMacros.length > 0) {
|
|
118
|
+
console.log("āāā Top Macros āāā");
|
|
119
|
+
for (const m of report.topMacros) {
|
|
120
|
+
console.log(` ${m.actions.join(" ā ")} (accept: ${(m.acceptanceRate * 100).toFixed(0)}%, freq: ${m.frequency})`);
|
|
121
|
+
}
|
|
122
|
+
console.log(` Total macros mined: ${report.macroCount}`);
|
|
123
|
+
console.log();
|
|
124
|
+
}
|
|
125
|
+
}
|