progmune-runtime 2.1.6 → 3.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +108 -468
- package/dist/ablation-study.js +144 -0
- package/dist/ablation-study.test.js +18 -0
- package/dist/action-runtime.js +3 -1
- package/dist/active-learning.js +211 -0
- package/dist/analytics.js +139 -0
- package/dist/asset-factory.js +309 -0
- package/dist/asset-growth.js +244 -0
- package/dist/asset-promotion.js +382 -0
- package/dist/asset-quality.js +550 -0
- package/dist/audit/business-translator.js +285 -0
- package/dist/audit/cli.js +66 -0
- package/dist/audit/formatters/html.js +379 -0
- package/dist/audit/formatters/json.js +11 -0
- package/dist/audit/formatters/markdown.js +192 -0
- package/dist/audit/formatters/terminal.js +189 -0
- package/dist/audit/index.js +25 -0
- package/dist/audit/report-builder.js +318 -0
- package/dist/audit/types.js +8 -0
- package/dist/audit.js +3 -3
- package/dist/auto-benchmark-generator.js +137 -0
- package/dist/auto-benchmark-generator.test.js +45 -0
- package/dist/auto-protocol-synthesizer.js +362 -0
- package/dist/auto-protocol-synthesizer.test.js +82 -0
- package/dist/autonomous-patch.js +175 -0
- package/dist/autonomous-patch.test.js +128 -0
- package/dist/badge/badge-server.js +98 -0
- package/dist/behavior-miner.js +442 -0
- package/dist/belief-layer.js +475 -0
- package/dist/benchmark-count.js +5 -0
- package/dist/benchmark-generator.js +211 -0
- package/dist/benchmark-harness.js +201 -0
- package/dist/benchmark-pass-rate.js +7 -0
- package/dist/benchmark-report.js +8 -3
- package/dist/benchmark-save.js +14 -1
- package/dist/bootstrap-validation.js +197 -0
- package/dist/bootstrap-validation.test.js +51 -0
- package/dist/branch-ledger.js +1 -1
- package/dist/capability-gap.js +130 -0
- package/dist/certify-html.js +351 -0
- package/dist/certify.js +326 -0
- package/dist/check.js +4 -4
- package/dist/compliance-miner.js +447 -0
- package/dist/continuous-benchmark.js +194 -0
- package/dist/continuous-benchmark.test.js +116 -0
- package/dist/corpus-stats.js +173 -0
- package/dist/counterfactual-engine.js +288 -0
- package/dist/coverage-dashboard.js +109 -0
- package/dist/coverage-system.test.js +205 -0
- package/dist/cross-repo-precision.js +352 -0
- package/dist/cve-benchmark.js +180 -0
- package/dist/cve-benchmark.test.js +28 -0
- package/dist/cve-collector.js +73 -0
- package/dist/data-quality.js +141 -0
- package/dist/decision-engine.js +388 -0
- package/dist/derive-metadata.js +250 -0
- package/dist/difficulty-active.test.js +198 -0
- package/dist/difficulty-map.js +244 -0
- package/dist/discovery-analytics.js +125 -0
- package/dist/discovery-model.js +149 -0
- package/dist/discovery-optimize.test.js +199 -0
- package/dist/discovery-trace.js +276 -0
- package/dist/discovery-trace.test.js +97 -0
- package/dist/emitter.js +83 -1
- package/dist/enterprise-dashboard.js +405 -0
- package/dist/eval-hardening.js +297 -0
- package/dist/eval-hardening.test.js +85 -0
- package/dist/evaluation-campaign.js +359 -0
- package/dist/evaluation-campaign.test.js +181 -0
- package/dist/evidence-growth.js +143 -0
- package/dist/evidence-repository.js +209 -0
- package/dist/evidence-system.js +441 -0
- package/dist/execute.js +15 -7
- package/dist/experimental/software-physics.js +291 -0
- package/dist/experimental/state-inference.js +516 -0
- package/dist/experimental/unsupervised-physics.js +230 -0
- package/dist/extract-ir-python.js +54 -7
- package/dist/extract-ir.js +376 -12
- package/dist/failure-collector.js +2 -2
- package/dist/failure-corpus.js +322 -9
- package/dist/feedback.js +16 -5
- package/dist/feedback.test.js +49 -0
- package/dist/file-lock.js +1 -1
- package/dist/flywheel-batch.js +292 -0
- package/dist/frameworks/express-cli.js +237 -0
- package/dist/frameworks/express-detector.js +445 -0
- package/dist/frameworks/express-detector.test.js +206 -0
- package/dist/frameworks/index.js +30 -0
- package/dist/frameworks/nestjs-detector.js +302 -0
- package/dist/frameworks/trpc-detector.js +161 -0
- package/dist/frameworks/version-awareness.js +179 -0
- package/dist/function-synonyms.js +164 -0
- package/dist/function-synonyms.test.js +68 -0
- package/dist/generalization.test.js +352 -0
- package/dist/goal-annotator.js +113 -0
- package/dist/goal-planner.js +563 -0
- package/dist/gold-cve.js +164 -0
- package/dist/gold-cve.test.js +104 -0
- package/dist/gold-quality.js +206 -0
- package/dist/gold-tiers.js +241 -0
- package/dist/governance-dashboard.js +327 -0
- package/dist/graph-viz.js +240 -0
- package/dist/guided-frontier.js +195 -0
- package/dist/hierarchical-planner.js +148 -0
- package/dist/identifier-parser.js +260 -0
- package/dist/immune-metrics.js +93 -0
- package/dist/immune-receiver.js +158 -0
- package/dist/immune-reporter.js +1 -1
- package/dist/improvement-orchestrator.js +206 -0
- package/dist/inject-p0-vocabulary.js +300 -0
- package/dist/intent-parser.js +218 -0
- package/dist/invariant-algebra.js +476 -0
- package/dist/invariant-calculus.js +533 -0
- package/dist/ir-utils.js +70 -0
- package/dist/ir-utils.test.js +50 -0
- package/dist/knowledge-api.js +312 -0
- package/dist/knowledge-evolution.js +452 -0
- package/dist/knowledge-explorer.js +506 -0
- package/dist/knowledge-flywheel.js +274 -0
- package/dist/knowledge-governance.js +338 -0
- package/dist/knowledge-governance.test.js +150 -0
- package/dist/knowledge-graph.js +181 -0
- package/dist/knowledge-guided-synth.js +246 -0
- package/dist/knowledge-loop.test.js +77 -0
- package/dist/knowledge-object.js +316 -0
- package/dist/knowledge-package.js +98 -0
- package/dist/kpi-dashboard.js +561 -0
- package/dist/l3-cross-function.js +280 -0
- package/dist/learning-ranker.js +148 -0
- package/dist/learning-ranker.test.js +291 -0
- package/dist/ledger/accountability.js +322 -0
- package/dist/ledger/chain-builder.js +185 -0
- package/dist/ledger/cli.js +222 -0
- package/dist/ledger/index.js +13 -0
- package/dist/ledger/signatures.js +193 -0
- package/dist/ledger/types.js +9 -0
- package/dist/llm.js +74 -3
- package/dist/load-benchmarks.js +8 -3
- package/dist/logger.js +66 -0
- package/dist/logger.test.js +37 -0
- package/dist/logistic-reward.js +339 -0
- package/dist/logistic-reward.test.js +180 -0
- package/dist/macro-graph.js +193 -0
- package/dist/macro-repair.js +183 -0
- package/dist/mcp-server.mjs +1202 -483
- package/dist/memory-layer.js +42 -5
- package/dist/multi-repo-precision.js +422 -0
- package/dist/name-free-protocol.js +425 -0
- package/dist/name-free-protocol.test.js +170 -0
- package/dist/name-scrambling.js +138 -0
- package/dist/name-scrambling.test.js +16 -0
- package/dist/p3-observability.test.js +281 -0
- package/dist/p5-orchestrator.test.js +225 -0
- package/dist/pairwise-preference.js +294 -0
- package/dist/pairwise-preference.test.js +140 -0
- package/dist/planner-constraints.js +104 -0
- package/dist/planner-prompts.js +155 -0
- package/dist/planner-telemetry.js +415 -0
- package/dist/planner-trace.js +214 -0
- package/dist/planner.js +162 -167
- package/dist/plsb/artifact.js +116 -0
- package/dist/plsb/cli.js +71 -0
- package/dist/plsb/index.js +19 -0
- package/dist/plsb/leaderboard.js +249 -0
- package/dist/plsb/report-md.js +156 -0
- package/dist/plsb/schema.js +179 -0
- package/dist/plsb-benchmark.js +284 -0
- package/dist/plsb-benchmark.test.js +119 -0
- package/dist/policy/cli.js +134 -0
- package/dist/policy/engine.js +333 -0
- package/dist/policy/index.js +12 -0
- package/dist/policy/types.js +59 -0
- package/dist/policy-miner.js +505 -0
- package/dist/precision-analyze.js +229 -0
- package/dist/precision-benchmark.js +147 -0
- package/dist/precision-label-c.js +134 -0
- package/dist/precision-label.js +193 -0
- package/dist/precision-report-c.js +149 -0
- package/dist/precision-report.js +246 -0
- package/dist/progmune-status.js +108 -0
- package/dist/proof-engine.js +479 -0
- package/dist/proof-provenance.js +315 -0
- package/dist/protocol-coverage.js +294 -0
- package/dist/protocol-detector.js +1189 -0
- package/dist/protocol-embedding-expanded.js +297 -0
- package/dist/protocol-embedding-expanded.test.js +97 -0
- package/dist/protocol-embedding.js +195 -0
- package/dist/protocol-embedding.test.js +82 -0
- package/dist/protocol-extractor-v2.js +354 -0
- package/dist/protocol-extractor-v2.test.js +140 -0
- package/dist/protocol-extractor.js +310 -0
- package/dist/protocol-extractor.test.js +113 -0
- package/dist/protocol-foundation.js +322 -0
- package/dist/protocol-foundation.test.js +163 -0
- package/dist/protocol-frontier.js +243 -0
- package/dist/protocol-frontier.test.js +92 -0
- package/dist/protocol-gap-analyzer.js +228 -0
- package/dist/protocol-gap-analyzer.test.js +49 -0
- package/dist/protocol-invariants.js +276 -0
- package/dist/protocol-invariants.test.js +111 -0
- package/dist/protocol-knowledge.js +464 -0
- package/dist/protocol-miner.js +343 -0
- package/dist/protocol-mining.js +207 -0
- package/dist/protocol-mining.test.js +37 -0
- package/dist/protocol-registry.js +1 -1
- package/dist/protocol-security-benchmark.js +222 -0
- package/dist/protocol-vulnerability.js +257 -0
- package/dist/protocol-vulnerability.test.js +60 -0
- package/dist/python-benchmark.js +120 -0
- package/dist/python-emitter.js +163 -45
- package/dist/python-protocol-extractor.js +187 -0
- package/dist/python-protocol-extractor.test.js +116 -0
- package/dist/realworld-benchmark.js +646 -0
- package/dist/realworld-benchmark.test.js +36 -0
- package/dist/repair-arch.test.js +411 -0
- package/dist/repair-evolution.test.js +454 -0
- package/dist/repair-executor.js +719 -0
- package/dist/repair-proposal.js +4 -4
- package/dist/repair-ranker.js +141 -0
- package/dist/repair-strategies.js +419 -0
- package/dist/repair-taxonomy.js +234 -0
- package/dist/repair-types.js +12 -0
- package/dist/repo-evaluator.js +250 -0
- package/dist/repo-evaluator.test.js +128 -0
- package/dist/resource-abstraction.js +242 -0
- package/dist/resource-detector.js +211 -0
- package/dist/result.test.js +43 -0
- package/dist/reward-system.js +411 -0
- package/dist/reward-system.test.js +175 -0
- package/dist/risk-model.js +215 -0
- package/dist/rule-miner.js +234 -7
- package/dist/rule-specificity.js +254 -0
- package/dist/runtime-types.js +27 -0
- package/dist/scaffold.js +208 -0
- package/dist/scale-collector.test.js +101 -0
- package/dist/scale-trajectory-collector.js +128 -0
- package/dist/sdk.js +250 -0
- package/dist/search-planner.js +4 -41
- package/dist/semantic-snapshot.js +1 -1
- package/dist/semantic-topology.js +121 -0
- package/dist/semantic-trace.js +310 -317
- package/dist/sequence-extractor.js +343 -0
- package/dist/skill-library.js +245 -0
- package/dist/skill-planner.test.js +189 -0
- package/dist/software-physics.js +291 -0
- package/dist/software-physics.test.js +81 -0
- package/dist/ssg-precision.js +478 -0
- package/dist/ssg-validator.js +71 -21
- package/dist/state-inference-doubleblind.test.js +160 -0
- package/dist/state-inference.js +516 -0
- package/dist/state-inference.test.js +115 -0
- package/dist/state-machine-fingerprint.js +345 -0
- package/dist/state-machine-fingerprint.test.js +120 -0
- package/dist/state-miner.js +386 -0
- package/dist/state-name-inference.js +213 -0
- package/dist/state-name-inference.test.js +69 -0
- package/dist/strategy-planner.js +262 -96
- package/dist/strategy-planner.test.js +135 -0
- package/dist/telemetry-analytics.test.js +402 -0
- package/dist/terminal-format.js +68 -0
- package/dist/terminal-format.test.js +83 -0
- package/dist/topology-factory.js +196 -0
- package/dist/topology-representation.js +242 -0
- package/dist/topology-representation.test.js +27 -0
- package/dist/trajectory-augmentation.js +254 -0
- package/dist/trajectory-augmentation.test.js +63 -0
- package/dist/trajectory-corpus.js +440 -0
- package/dist/trajectory-corpus.test.js +32 -0
- package/dist/trajectory-feedback.test.js +116 -0
- package/dist/transition-synthesizer.js +286 -0
- package/dist/transition-synthesizer.test.js +123 -0
- package/dist/trust/api-semantic-mapper.js +809 -0
- package/dist/trust/call-graph-propagator.js +225 -0
- package/dist/trust/cli.js +122 -0
- package/dist/trust/compliance-scorer.js +283 -0
- package/dist/trust/confidence-calculator.js +261 -0
- package/dist/trust/engine.js +1145 -0
- package/dist/trust/explainability.js +85 -0
- package/dist/trust/formatters/ci.js +42 -0
- package/dist/trust/formatters/json.js +11 -0
- package/dist/trust/formatters/terminal.js +152 -0
- package/dist/trust/index.js +39 -0
- package/dist/trust/phase1-verify.js +171 -0
- package/dist/trust/protocol-domain-validator.js +697 -0
- package/dist/trust/score-calculator.js +282 -0
- package/dist/trust/ssg-bridge.js +641 -0
- package/dist/trust/ssg-bridge.test.js +269 -0
- package/dist/trust/types.js +67 -0
- package/dist/trust/violation-trace.js +335 -0
- package/dist/trust-api.js +179 -0
- package/dist/trust-calibration.js +279 -0
- package/dist/unknown-protocol-discovery.js +339 -0
- package/dist/unknown-protocol-discovery.test.js +102 -0
- package/dist/unsupervised-physics.js +230 -0
- package/dist/unsupervised-physics.test.js +95 -0
- package/dist/utils.test.js +37 -0
- package/dist/validator.js +187 -10
- package/dist/verification-intelligence.js +475 -0
- package/dist/verify-api.js +432 -0
- package/dist/vi-impact-report.js +293 -0
- package/dist/wl-fingerprint.js +162 -0
- package/dist/wl-fingerprint.test.js +130 -0
- package/dist/zeroshot-strategy.js +139 -0
- package/docs/Progmune_/346/212/225/350/265/204/344/272/272/347/231/275/347/232/256/344/271/246_v2.0.html +576 -0
- package/docs/Progmune_/351/241/271/347/233/256/345/205/250/350/247/243.html +710 -0
- package/package.json +74 -7
- package/protocols.json +1956 -50
- package/.dockerignore +0 -14
- package/.mcp.json +0 -11
- package/.progmune_allowlist +0 -50
- package/.test_report/test_report.md +0 -87
- package/Dockerfile +0 -9
- package/FAQ.md +0 -167
- package/WHITEPAPER.md +0 -540
- package/demo-project/auth.ts +0 -55
- package/demo-project/tsconfig.json +0 -8
- package/dist/acl-breakdown.js +0 -13
- package/dist/all-sessions.js +0 -11
- package/dist/antibody-stats.js +0 -11
- package/dist/branch-tree-count.js +0 -14
- package/dist/common-fixpath.js +0 -12
- package/dist/constraint-types.js +0 -12
- package/dist/exec-metrics.js +0 -11
- package/dist/failure-report.js +0 -11
- package/dist/fast-path-hits.js +0 -13
- package/dist/fingerprint-list.js +0 -15
- package/dist/gen-history-log.js +0 -13
- package/dist/heatmap-data.js +0 -11
- package/dist/recent-session.js +0 -12
- package/dist/svl-distribution.js +0 -11
- package/dist/terminal-status.js +0 -11
- package/dist/token-savings.js +0 -11
- package/dist/total-repairs.js +0 -12
- package/dist/unresolved-count.js +0 -12
- package/dist/valid-fingerprints.js +0 -13
- package/dist/verify-ledgers.js +0 -11
- package/docs/whitepaper-style.css +0 -77
- package/docs/whitepaper-v2.1.md +0 -609
- package/docs/whitepaper-v2.2.md +0 -1064
- package/docs/whitepaper-v2.2.pdf +0 -0
- package/fly.toml +0 -31
- package/public/dashboard.html +0 -119
- package/server/hub.js +0 -116
- package/test/replay-golden/sess_1780063202050_mgeld.json +0 -9
- package/test/replay-golden/sess_1780064032560_gocld.json +0 -354
- package/test/replay-golden/sess_1780064413331_s2709.json +0 -606
- package/test/replay-golden/sess_1780064792710_y3avo.json +0 -614
- package/test/replay-golden.ts +0 -84
- package/test_benchmark.js +0 -165
- package/test_comprehensive.mjs +0 -638
- package/test_concurrency.js +0 -129
- package/test_ir_robustness.js +0 -85
- package/test_semantic_contracts.js +0 -269
- package/test_ssg_stress.js +0 -156
- package/test_svl3.js +0 -58
- package/tsconfig.json +0 -17
package/dist/llm.js
CHANGED
|
@@ -6,12 +6,13 @@ Object.defineProperty(exports, "__esModule", { value: true });
|
|
|
6
6
|
exports.callCount = void 0;
|
|
7
7
|
exports.resetCallCount = resetCallCount;
|
|
8
8
|
exports.estimateTokens = estimateTokens;
|
|
9
|
+
exports.isVerbose = isVerbose;
|
|
9
10
|
exports.generate = generate;
|
|
10
11
|
exports.chat = chat;
|
|
11
12
|
try {
|
|
12
13
|
require("dotenv/config");
|
|
13
14
|
}
|
|
14
|
-
catch { }
|
|
15
|
+
catch { /* ignore — dotenv is optional */ }
|
|
15
16
|
const openai_1 = __importDefault(require("openai"));
|
|
16
17
|
const provider = process.env.LLM_PROVIDER || "deepseek";
|
|
17
18
|
const configs = {
|
|
@@ -33,8 +34,54 @@ const apiKey = process.env.LLM_API_KEY || "sk-xxxx";
|
|
|
33
34
|
const baseURL = process.env.LLM_BASE_URL || selected.baseURL;
|
|
34
35
|
const model = process.env.LLM_MODEL || selected.model;
|
|
35
36
|
const client = new openai_1.default({ apiKey, baseURL });
|
|
37
|
+
// ── API Key validation ──
|
|
38
|
+
const KNOWN_KEY_PREFIXES = {
|
|
39
|
+
"ghp_": "GitHub Personal Access Token (classic)",
|
|
40
|
+
"github_pat_": "GitHub Personal Access Token (fine-grained)",
|
|
41
|
+
"glpat-": "GitLab Personal Access Token",
|
|
42
|
+
"sk-xxxx": "placeholder / default value",
|
|
43
|
+
"your-key": "placeholder / default value",
|
|
44
|
+
};
|
|
45
|
+
function validateApiKey(key) {
|
|
46
|
+
const warnings = [];
|
|
47
|
+
if (!key || key === "sk-xxxx" || key === "your-key") {
|
|
48
|
+
warnings.push("❌ LLM_API_KEY is not set or is a placeholder. Configure via .env: LLM_API_KEY=your-deepseek-key");
|
|
49
|
+
return warnings;
|
|
50
|
+
}
|
|
51
|
+
for (const [prefix, label] of Object.entries(KNOWN_KEY_PREFIXES)) {
|
|
52
|
+
if (key.startsWith(prefix)) {
|
|
53
|
+
warnings.push(`⚠️ LLM_API_KEY looks like a ${label} (prefix: "${prefix}"), not an LLM API key. DeepSeek keys start with "sk-". Get one at https://platform.deepseek.com/api_keys`);
|
|
54
|
+
break;
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
if (key.length < 20) {
|
|
58
|
+
warnings.push(`⚠️ LLM_API_KEY is unusually short (${key.length} chars). Most API keys are 30+ characters.`);
|
|
59
|
+
}
|
|
60
|
+
return warnings;
|
|
61
|
+
}
|
|
62
|
+
const keyWarnings = validateApiKey(apiKey);
|
|
63
|
+
for (const w of keyWarnings) {
|
|
64
|
+
console.error(w);
|
|
65
|
+
}
|
|
36
66
|
exports.callCount = 0;
|
|
37
67
|
function resetCallCount() { exports.callCount = 0; }
|
|
68
|
+
const MAX_LLM_CALLS = parseInt(process.env.PROGMUNE_MAX_LLM_CALLS || "50", 10);
|
|
69
|
+
const RATE_LIMIT_MS = parseInt(process.env.PROGMUNE_RATE_LIMIT_MS || "0", 10);
|
|
70
|
+
let lastCallTime = 0;
|
|
71
|
+
function assertCallLimit() {
|
|
72
|
+
if (exports.callCount >= MAX_LLM_CALLS) {
|
|
73
|
+
throw new Error(`LLM call limit reached (${exports.callCount}/${MAX_LLM_CALLS}). ` +
|
|
74
|
+
`Increase via PROGMUNE_MAX_LLM_CALLS env var or call resetCallCount().`);
|
|
75
|
+
}
|
|
76
|
+
}
|
|
77
|
+
async function applyRateLimit() {
|
|
78
|
+
if (RATE_LIMIT_MS <= 0)
|
|
79
|
+
return;
|
|
80
|
+
const elapsed = Date.now() - lastCallTime;
|
|
81
|
+
if (elapsed < RATE_LIMIT_MS) {
|
|
82
|
+
await new Promise(r => setTimeout(r, RATE_LIMIT_MS - elapsed));
|
|
83
|
+
}
|
|
84
|
+
}
|
|
38
85
|
/** 粗略 token 估算:CJK 字符 ~1.5 token/字,其余 ~0.4 token/字符 */
|
|
39
86
|
/** @requires TEXT @produces TOKEN_COUNT */
|
|
40
87
|
function estimateTokens(text) {
|
|
@@ -42,20 +89,42 @@ function estimateTokens(text) {
|
|
|
42
89
|
const other = text.length - cjk;
|
|
43
90
|
return Math.ceil(cjk * 1.5 + other * 0.4);
|
|
44
91
|
}
|
|
92
|
+
// ── Verbose debug logging ──
|
|
93
|
+
const VERBOSE = process.env.PROGMUNE_VERBOSE === "1";
|
|
94
|
+
function vlog(label, content, maxLen = 2000) {
|
|
95
|
+
if (!VERBOSE)
|
|
96
|
+
return;
|
|
97
|
+
const truncated = content.length > maxLen
|
|
98
|
+
? content.slice(0, maxLen) + `\n... [truncated, ${content.length} total chars]`
|
|
99
|
+
: content;
|
|
100
|
+
console.error(`\n${"═".repeat(60)}\n🔍 [VERBOSE] ${label}\n${"─".repeat(60)}\n${truncated}\n${"═".repeat(60)}`);
|
|
101
|
+
}
|
|
102
|
+
function isVerbose() { return VERBOSE; }
|
|
45
103
|
/** @requires PROMPT @produces LLM_RESPONSE */
|
|
46
104
|
async function generate(prompt) {
|
|
105
|
+
assertCallLimit();
|
|
106
|
+
await applyRateLimit();
|
|
47
107
|
exports.callCount++;
|
|
108
|
+
lastCallTime = Date.now();
|
|
109
|
+
vlog(`LLM Call #${exports.callCount} (generate) | Model: ${model} | Prompt (${estimateTokens(prompt)} est. tokens)`, prompt);
|
|
48
110
|
const resp = await client.chat.completions.create({
|
|
49
111
|
model,
|
|
50
112
|
messages: [{ role: "user", content: prompt }],
|
|
51
113
|
temperature: 0.0,
|
|
52
114
|
});
|
|
53
|
-
|
|
115
|
+
const content = resp.choices[0]?.message?.content || "";
|
|
116
|
+
vlog(`LLM Response #${exports.callCount} (${estimateTokens(content)} est. tokens, finish=${resp.choices[0]?.finish_reason})`, content);
|
|
117
|
+
return content;
|
|
54
118
|
}
|
|
55
119
|
/** 带 system prompt 的调用:静态规则放 system,动态内容放 user,语义分离便于未来对接各平台缓存策略 */
|
|
56
120
|
/** @requires SYSTEM_PROMPT @produces LLM_RESPONSE */
|
|
57
121
|
async function chat(systemPrompt, userPrompt) {
|
|
122
|
+
assertCallLimit();
|
|
123
|
+
await applyRateLimit();
|
|
58
124
|
exports.callCount++;
|
|
125
|
+
lastCallTime = Date.now();
|
|
126
|
+
const totalTokens = estimateTokens(systemPrompt) + estimateTokens(userPrompt);
|
|
127
|
+
vlog(`LLM Call #${exports.callCount} (chat) | Model: ${model} | System (${estimateTokens(systemPrompt)}t) + User (${estimateTokens(userPrompt)}t) = ${totalTokens} est. tokens`, `── SYSTEM ──\n${systemPrompt}\n── USER ──\n${userPrompt}`);
|
|
59
128
|
const resp = await client.chat.completions.create({
|
|
60
129
|
model,
|
|
61
130
|
messages: [
|
|
@@ -64,5 +133,7 @@ async function chat(systemPrompt, userPrompt) {
|
|
|
64
133
|
],
|
|
65
134
|
temperature: 0.0,
|
|
66
135
|
});
|
|
67
|
-
|
|
136
|
+
const content = resp.choices[0]?.message?.content || "";
|
|
137
|
+
vlog(`LLM Response #${exports.callCount} (${estimateTokens(content)} est. tokens, finish=${resp.choices[0]?.finish_reason})`, content);
|
|
138
|
+
return content;
|
|
68
139
|
}
|
package/dist/load-benchmarks.js
CHANGED
|
@@ -34,11 +34,16 @@ var __importStar = (this && this.__importStar) || (function () {
|
|
|
34
34
|
})();
|
|
35
35
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
36
36
|
exports.loadBenchmarks = loadBenchmarks;
|
|
37
|
-
/** Load benchmark tasks from bench/tasks.json
|
|
38
|
-
* @protocol pre_states=[] post_states=["BENCHMARKS_LOADED"]
|
|
39
|
-
*/
|
|
40
37
|
const fs = __importStar(require("fs"));
|
|
41
38
|
const path = __importStar(require("path"));
|
|
39
|
+
/**
|
|
40
|
+
* Load benchmark tasks from bench/tasks.json.
|
|
41
|
+
* @requires BENCH_DIR @produces BENCHMARK_TASKS
|
|
42
|
+
* @purpose Load benchmark task definitions for execution
|
|
43
|
+
* @tags benchmark, load, data
|
|
44
|
+
* @useWhen running benchmarks, generating benchmark reports
|
|
45
|
+
* @protocol pre_states=[] post_states=["BENCHMARKS_LOADED"]
|
|
46
|
+
*/
|
|
42
47
|
function loadBenchmarks() {
|
|
43
48
|
const tasksPath = path.resolve(process.cwd(), "bench/tasks.json");
|
|
44
49
|
if (!fs.existsSync(tasksPath))
|
package/dist/logger.js
ADDED
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* Structured logger for Progmune Runtime.
|
|
4
|
+
*
|
|
5
|
+
* Replaces raw console.* calls with leveled, module-scoped,
|
|
6
|
+
* optionally JSON-formatted output.
|
|
7
|
+
*
|
|
8
|
+
* Environment variables:
|
|
9
|
+
* PROGMUNE_LOG_LEVEL — debug | info | warn | error (default: info)
|
|
10
|
+
* PROGMUNE_LOG_JSON — "true" for machine-readable JSON lines
|
|
11
|
+
*/
|
|
12
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
13
|
+
exports.createLogger = createLogger;
|
|
14
|
+
const LEVEL_ORDER = { debug: 0, info: 1, warn: 2, error: 3 };
|
|
15
|
+
const minLevel = (process.env.PROGMUNE_LOG_LEVEL || "info");
|
|
16
|
+
const jsonMode = process.env.PROGMUNE_LOG_JSON === "true";
|
|
17
|
+
function shouldLog(level) {
|
|
18
|
+
return LEVEL_ORDER[level] >= LEVEL_ORDER[minLevel];
|
|
19
|
+
}
|
|
20
|
+
function formatLine(level, module, message, data) {
|
|
21
|
+
if (jsonMode) {
|
|
22
|
+
const entry = {
|
|
23
|
+
ts: new Date().toISOString(),
|
|
24
|
+
level,
|
|
25
|
+
module,
|
|
26
|
+
message,
|
|
27
|
+
...(data !== undefined ? { data } : {}),
|
|
28
|
+
};
|
|
29
|
+
return JSON.stringify(entry);
|
|
30
|
+
}
|
|
31
|
+
const prefix = {
|
|
32
|
+
debug: " ",
|
|
33
|
+
info: "ℹ ",
|
|
34
|
+
warn: "⚠ ",
|
|
35
|
+
error: "❌",
|
|
36
|
+
};
|
|
37
|
+
const tag = `[${module}]`;
|
|
38
|
+
const line = `${prefix[level]} ${tag} ${message}`;
|
|
39
|
+
if (data !== undefined) {
|
|
40
|
+
if (data instanceof Error) {
|
|
41
|
+
return `${line}\n ${data.stack || data.message}`;
|
|
42
|
+
}
|
|
43
|
+
return `${line} ${JSON.stringify(data)}`;
|
|
44
|
+
}
|
|
45
|
+
return line;
|
|
46
|
+
}
|
|
47
|
+
function createLogger(module) {
|
|
48
|
+
return {
|
|
49
|
+
debug(message, data) {
|
|
50
|
+
if (shouldLog("debug"))
|
|
51
|
+
console.error(formatLine("debug", module, message, data));
|
|
52
|
+
},
|
|
53
|
+
info(message, data) {
|
|
54
|
+
if (shouldLog("info"))
|
|
55
|
+
console.error(formatLine("info", module, message, data));
|
|
56
|
+
},
|
|
57
|
+
warn(message, data) {
|
|
58
|
+
if (shouldLog("warn"))
|
|
59
|
+
console.error(formatLine("warn", module, message, data));
|
|
60
|
+
},
|
|
61
|
+
error(message, data) {
|
|
62
|
+
if (shouldLog("error"))
|
|
63
|
+
console.error(formatLine("error", module, message, data));
|
|
64
|
+
},
|
|
65
|
+
};
|
|
66
|
+
}
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
/**
|
|
4
|
+
* Unit tests for logger.ts — structured logging.
|
|
5
|
+
*/
|
|
6
|
+
const vitest_1 = require("vitest");
|
|
7
|
+
const logger_1 = require("./logger");
|
|
8
|
+
(0, vitest_1.describe)("createLogger", () => {
|
|
9
|
+
(0, vitest_1.it)("creates a logger with all log methods", () => {
|
|
10
|
+
const log = (0, logger_1.createLogger)("test");
|
|
11
|
+
(0, vitest_1.expect)(typeof log.debug).toBe("function");
|
|
12
|
+
(0, vitest_1.expect)(typeof log.info).toBe("function");
|
|
13
|
+
(0, vitest_1.expect)(typeof log.warn).toBe("function");
|
|
14
|
+
(0, vitest_1.expect)(typeof log.error).toBe("function");
|
|
15
|
+
});
|
|
16
|
+
(0, vitest_1.it)("returns a different logger per module name", () => {
|
|
17
|
+
const a = (0, logger_1.createLogger)("module-a");
|
|
18
|
+
const b = (0, logger_1.createLogger)("module-b");
|
|
19
|
+
(0, vitest_1.expect)(a).not.toBe(b);
|
|
20
|
+
});
|
|
21
|
+
(0, vitest_1.it)("log methods accept message and optional data", () => {
|
|
22
|
+
const log = (0, logger_1.createLogger)("test");
|
|
23
|
+
// Should not throw
|
|
24
|
+
log.info("hello");
|
|
25
|
+
log.info("with data", { key: "value" });
|
|
26
|
+
log.warn("warning");
|
|
27
|
+
log.error("error", new Error("boom"));
|
|
28
|
+
log.debug("debug message");
|
|
29
|
+
});
|
|
30
|
+
(0, vitest_1.it)("loggers are callable without throwing", () => {
|
|
31
|
+
const log = (0, logger_1.createLogger)("safety");
|
|
32
|
+
(0, vitest_1.expect)(() => log.info("msg")).not.toThrow();
|
|
33
|
+
(0, vitest_1.expect)(() => log.warn("msg")).not.toThrow();
|
|
34
|
+
(0, vitest_1.expect)(() => log.error("msg")).not.toThrow();
|
|
35
|
+
(0, vitest_1.expect)(() => log.debug("msg")).not.toThrow();
|
|
36
|
+
});
|
|
37
|
+
});
|
|
@@ -0,0 +1,339 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* P4.0: Logistic Reward Model
|
|
4
|
+
*
|
|
5
|
+
* Pure TypeScript logistic regression that predicts P(accepted)
|
|
6
|
+
* from 7 normalized features. Compatible with existing Ranker interface.
|
|
7
|
+
*
|
|
8
|
+
* Why logistic regression (not neural):
|
|
9
|
+
* - Interpretable weights (auditable)
|
|
10
|
+
* - Trains on ~1000 samples (neural needs 10K+)
|
|
11
|
+
* - Same interface as LinearRanker / LearningRanker
|
|
12
|
+
* - Weights directly tell you WHICH features drive acceptance
|
|
13
|
+
*
|
|
14
|
+
* Feature vector (all [0,1]):
|
|
15
|
+
* 1. protocolSafety
|
|
16
|
+
* 2. historicalSuccessRate
|
|
17
|
+
* 3. normActionCount (actionCount / maxActions)
|
|
18
|
+
* 4. latencyCost
|
|
19
|
+
* 5. auditability
|
|
20
|
+
* 6. acceptanceRate (from TelemetryIndex)
|
|
21
|
+
* 7. executionSuccessRate (from TelemetryIndex)
|
|
22
|
+
*
|
|
23
|
+
* Model: P(accepted) = σ(w·x + b)
|
|
24
|
+
* Loss: binary cross-entropy + L2 regularization
|
|
25
|
+
* Train: mini-batch SGD
|
|
26
|
+
*
|
|
27
|
+
* Usage:
|
|
28
|
+
* const model = LogisticRewardModel.train(telemetry, config);
|
|
29
|
+
* const score = model.score(features, telemetryStats);
|
|
30
|
+
*/
|
|
31
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
32
|
+
exports.LogisticRewardModel = void 0;
|
|
33
|
+
exports.compareModels = compareModels;
|
|
34
|
+
const DEFAULT_CONFIG = {
|
|
35
|
+
learningRate: 0.01,
|
|
36
|
+
epochs: 200,
|
|
37
|
+
l2Lambda: 0.001,
|
|
38
|
+
batchSize: 32,
|
|
39
|
+
minSamples: 50,
|
|
40
|
+
convergenceThreshold: 1e-4,
|
|
41
|
+
};
|
|
42
|
+
// ═══════════════════════════════════════════════════════════════
|
|
43
|
+
// Logistic Regression Implementation
|
|
44
|
+
// ═══════════════════════════════════════════════════════════════
|
|
45
|
+
function sigmoid(z) {
|
|
46
|
+
if (z > 20)
|
|
47
|
+
return 1.0;
|
|
48
|
+
if (z < -20)
|
|
49
|
+
return 0.0;
|
|
50
|
+
return 1.0 / (1.0 + Math.exp(-z));
|
|
51
|
+
}
|
|
52
|
+
function dot(a, b) {
|
|
53
|
+
return a.reduce((s, v, i) => s + v * b[i], 0);
|
|
54
|
+
}
|
|
55
|
+
function featureVecToArray(f) {
|
|
56
|
+
return [
|
|
57
|
+
f.protocolSafety,
|
|
58
|
+
f.historicalSuccessRate,
|
|
59
|
+
f.normActionCount,
|
|
60
|
+
f.latencyCost,
|
|
61
|
+
f.auditability,
|
|
62
|
+
f.acceptanceRate,
|
|
63
|
+
f.executionSuccessRate,
|
|
64
|
+
];
|
|
65
|
+
}
|
|
66
|
+
const FEATURE_NAMES = [
|
|
67
|
+
"protocolSafety", "historicalSuccessRate", "normActionCount",
|
|
68
|
+
"latencyCost", "auditability", "acceptanceRate", "executionSuccessRate",
|
|
69
|
+
];
|
|
70
|
+
// ═══════════════════════════════════════════════════════════════
|
|
71
|
+
// LogisticRewardModel
|
|
72
|
+
// ═══════════════════════════════════════════════════════════════
|
|
73
|
+
class LogisticRewardModel {
|
|
74
|
+
constructor(weights, bias, config) {
|
|
75
|
+
this.weights = weights || new Array(7).fill(0);
|
|
76
|
+
this.bias = bias || 0;
|
|
77
|
+
this.config = { ...DEFAULT_CONFIG, ...config };
|
|
78
|
+
this.trained = weights !== undefined;
|
|
79
|
+
this.trainedSamples = 0;
|
|
80
|
+
this.loss = Infinity;
|
|
81
|
+
}
|
|
82
|
+
/** Predict P(accepted) from feature vector + telemetry stats. */
|
|
83
|
+
score(features, telemetryStats) {
|
|
84
|
+
const ts = telemetryStats || { acceptanceRate: 0.5, executionSuccessRate: 0.5 };
|
|
85
|
+
const maxActions = Math.max(features.actionCount, 8);
|
|
86
|
+
const fv = {
|
|
87
|
+
protocolSafety: features.protocolSafety,
|
|
88
|
+
historicalSuccessRate: features.historicalSuccessRate,
|
|
89
|
+
normActionCount: features.actionCount / maxActions,
|
|
90
|
+
latencyCost: features.latencyCost,
|
|
91
|
+
auditability: features.auditability,
|
|
92
|
+
acceptanceRate: ts.acceptanceRate,
|
|
93
|
+
executionSuccessRate: ts.executionSuccessRate,
|
|
94
|
+
};
|
|
95
|
+
return sigmoid(dot(this.weights, featureVecToArray(fv)) + this.bias);
|
|
96
|
+
}
|
|
97
|
+
/** Whether the model has been trained on sufficient data. */
|
|
98
|
+
get isTrained() { return this.trained; }
|
|
99
|
+
get sampleCount() { return this.trainedSamples; }
|
|
100
|
+
get finalLoss() { return this.loss; }
|
|
101
|
+
// ── Training ──
|
|
102
|
+
/** Train the model on telemetry data. */
|
|
103
|
+
static train(telemetry, config) {
|
|
104
|
+
const cfg = { ...DEFAULT_CONFIG, ...config };
|
|
105
|
+
const model = new LogisticRewardModel(undefined, undefined, cfg);
|
|
106
|
+
// Collect training samples from telemetry
|
|
107
|
+
const samples = LogisticRewardModel.collectSamples(telemetry);
|
|
108
|
+
if (samples.length < cfg.minSamples) {
|
|
109
|
+
// Not enough data — return untrained model (falls back to heuristic)
|
|
110
|
+
model.trainedSamples = samples.length;
|
|
111
|
+
return model;
|
|
112
|
+
}
|
|
113
|
+
// Normalize features
|
|
114
|
+
const normalized = LogisticRewardModel.normalizeSamples(samples);
|
|
115
|
+
// Initialize weights: small random values
|
|
116
|
+
let w = new Array(7).fill(0).map(() => (Math.random() - 0.5) * 0.1);
|
|
117
|
+
let b = 0.0;
|
|
118
|
+
let prevLoss = Infinity;
|
|
119
|
+
// Mini-batch SGD
|
|
120
|
+
for (let epoch = 0; epoch < cfg.epochs; epoch++) {
|
|
121
|
+
// Shuffle
|
|
122
|
+
const shuffled = [...normalized].sort(() => Math.random() - 0.5);
|
|
123
|
+
let totalLoss = 0;
|
|
124
|
+
for (let batchStart = 0; batchStart < shuffled.length; batchStart += cfg.batchSize) {
|
|
125
|
+
const batch = shuffled.slice(batchStart, batchStart + cfg.batchSize);
|
|
126
|
+
const [wGrad, bGrad] = LogisticRewardModel.computeGradients(batch, w, b, cfg.l2Lambda);
|
|
127
|
+
const loss = LogisticRewardModel.computeLoss(batch, w, b, cfg.l2Lambda);
|
|
128
|
+
totalLoss += loss;
|
|
129
|
+
// Update weights
|
|
130
|
+
for (let i = 0; i < 7; i++) {
|
|
131
|
+
w[i] -= cfg.learningRate * wGrad[i];
|
|
132
|
+
}
|
|
133
|
+
b -= cfg.learningRate * bGrad;
|
|
134
|
+
}
|
|
135
|
+
const avgLoss = totalLoss / Math.ceil(shuffled.length / cfg.batchSize);
|
|
136
|
+
// Early stopping
|
|
137
|
+
if (Math.abs(prevLoss - avgLoss) < cfg.convergenceThreshold) {
|
|
138
|
+
break;
|
|
139
|
+
}
|
|
140
|
+
prevLoss = avgLoss;
|
|
141
|
+
}
|
|
142
|
+
model.weights.length = 0;
|
|
143
|
+
model.weights.push(...w);
|
|
144
|
+
model.bias = b;
|
|
145
|
+
model.trained = true;
|
|
146
|
+
model.trainedSamples = samples.length;
|
|
147
|
+
model.loss = prevLoss;
|
|
148
|
+
return model;
|
|
149
|
+
}
|
|
150
|
+
// ── Sample Collection ──
|
|
151
|
+
static collectSamples(telemetry) {
|
|
152
|
+
const samples = [];
|
|
153
|
+
const decisions = telemetry.all();
|
|
154
|
+
for (const d of decisions) {
|
|
155
|
+
if (!d.feedback || !d.selectedCandidateId)
|
|
156
|
+
continue;
|
|
157
|
+
const sel = d.candidates.find(c => c.candidateId === d.selectedCandidateId);
|
|
158
|
+
if (!sel || sel.actions.length === 0)
|
|
159
|
+
continue;
|
|
160
|
+
// Compute label: 1 = accepted AND executed successfully, 0 otherwise
|
|
161
|
+
const accepted = d.feedback.decision === "accepted";
|
|
162
|
+
const execOk = d.feedback.executionResult?.success !== false;
|
|
163
|
+
const label = (accepted && execOk) ? 1 : 0;
|
|
164
|
+
// Compute base features from the candidate
|
|
165
|
+
const actionCount = sel.actions.length;
|
|
166
|
+
const maxActions = Math.max(actionCount, 8);
|
|
167
|
+
// Get telemetry stats for this fingerprint
|
|
168
|
+
const fp = sel.candidateId;
|
|
169
|
+
const stats = telemetry.getCandidateStats(fp);
|
|
170
|
+
const acceptTotal = stats.accepted + stats.rejected;
|
|
171
|
+
const acceptanceRate = acceptTotal > 0 ? stats.accepted / acceptTotal : 0.5;
|
|
172
|
+
const execTotal = stats.executionSuccess + stats.executionFailure;
|
|
173
|
+
const executionSuccessRate = execTotal > 0 ? stats.executionSuccess / execTotal : 0.5;
|
|
174
|
+
// Compute protocolSafety heuristic
|
|
175
|
+
const safetyFromLength = Math.max(0, 1.0 - actionCount / 10);
|
|
176
|
+
samples.push({
|
|
177
|
+
features: {
|
|
178
|
+
protocolSafety: safetyFromLength,
|
|
179
|
+
historicalSuccessRate: 0.5,
|
|
180
|
+
normActionCount: actionCount / maxActions,
|
|
181
|
+
latencyCost: Math.min(1, actionCount / maxActions),
|
|
182
|
+
auditability: 1.0 - actionCount / maxActions,
|
|
183
|
+
acceptanceRate,
|
|
184
|
+
executionSuccessRate,
|
|
185
|
+
},
|
|
186
|
+
label,
|
|
187
|
+
});
|
|
188
|
+
}
|
|
189
|
+
return samples;
|
|
190
|
+
}
|
|
191
|
+
// ── Normalization ──
|
|
192
|
+
static normalizeSamples(samples) {
|
|
193
|
+
// Features 0-5 are already [0,1]. Features 5,6 (acceptanceRate, executionSuccessRate) are also [0,1].
|
|
194
|
+
// No normalization needed — all features are already bounded.
|
|
195
|
+
return samples;
|
|
196
|
+
}
|
|
197
|
+
// ── Gradient Computation ──
|
|
198
|
+
static computeGradients(batch, w, b, l2Lambda) {
|
|
199
|
+
const wGrad = new Array(7).fill(0);
|
|
200
|
+
let bGrad = 0;
|
|
201
|
+
const n = batch.length;
|
|
202
|
+
if (n === 0)
|
|
203
|
+
return [wGrad, bGrad];
|
|
204
|
+
for (const sample of batch) {
|
|
205
|
+
const x = featureVecToArray(sample.features);
|
|
206
|
+
const z = dot(w, x) + b;
|
|
207
|
+
const yPred = sigmoid(z);
|
|
208
|
+
const error = yPred - sample.label;
|
|
209
|
+
for (let i = 0; i < 7; i++) {
|
|
210
|
+
wGrad[i] += error * x[i];
|
|
211
|
+
}
|
|
212
|
+
bGrad += error;
|
|
213
|
+
}
|
|
214
|
+
// Average gradients + L2 regularization on weights (not bias)
|
|
215
|
+
for (let i = 0; i < 7; i++) {
|
|
216
|
+
wGrad[i] = wGrad[i] / n + l2Lambda * w[i];
|
|
217
|
+
}
|
|
218
|
+
bGrad /= n;
|
|
219
|
+
return [wGrad, bGrad];
|
|
220
|
+
}
|
|
221
|
+
static computeLoss(batch, w, b, l2Lambda) {
|
|
222
|
+
let loss = 0;
|
|
223
|
+
const n = batch.length;
|
|
224
|
+
if (n === 0)
|
|
225
|
+
return 0;
|
|
226
|
+
for (const sample of batch) {
|
|
227
|
+
const x = featureVecToArray(sample.features);
|
|
228
|
+
const z = dot(w, x) + b;
|
|
229
|
+
const yPred = Math.max(1e-15, Math.min(1 - 1e-15, sigmoid(z))); // clip for numerical stability
|
|
230
|
+
const y = sample.label;
|
|
231
|
+
loss += -(y * Math.log(yPred) + (1 - y) * Math.log(1 - yPred));
|
|
232
|
+
}
|
|
233
|
+
loss /= n;
|
|
234
|
+
// L2 regularization (on weights only)
|
|
235
|
+
const l2Penalty = 0.5 * l2Lambda * w.reduce((s, wi) => s + wi * wi, 0);
|
|
236
|
+
return loss + l2Penalty;
|
|
237
|
+
}
|
|
238
|
+
// ── Interpretability ──
|
|
239
|
+
/** Return feature importance (absolute weight values, normalized). */
|
|
240
|
+
featureImportance() {
|
|
241
|
+
const absWeights = this.weights.map(Math.abs);
|
|
242
|
+
const total = absWeights.reduce((s, v) => s + v, 0) || 1;
|
|
243
|
+
return FEATURE_NAMES.map((name, i) => ({
|
|
244
|
+
name,
|
|
245
|
+
weight: this.weights[i],
|
|
246
|
+
importance: absWeights[i] / total,
|
|
247
|
+
})).sort((a, b) => b.importance - a.importance);
|
|
248
|
+
}
|
|
249
|
+
printWeights() {
|
|
250
|
+
console.log("\n─── LogisticRewardModel Weights ───");
|
|
251
|
+
console.log("Feature Weight Importance");
|
|
252
|
+
console.log("─────────────────────────────────────────");
|
|
253
|
+
const imp = this.featureImportance();
|
|
254
|
+
for (const f of imp) {
|
|
255
|
+
const w = f.weight.toFixed(4).padStart(8);
|
|
256
|
+
const pct = (f.importance * 100).toFixed(0).padStart(4);
|
|
257
|
+
console.log(` ${f.name.padEnd(22)} ${w} ${pct}%`);
|
|
258
|
+
}
|
|
259
|
+
console.log(` bias: ${this.bias.toFixed(4)}`);
|
|
260
|
+
console.log(` samples: ${this.trainedSamples} | loss: ${this.loss.toFixed(6)}`);
|
|
261
|
+
console.log();
|
|
262
|
+
}
|
|
263
|
+
/** Export weights for persistence. */
|
|
264
|
+
exportWeights() {
|
|
265
|
+
return {
|
|
266
|
+
weights: [...this.weights],
|
|
267
|
+
bias: this.bias,
|
|
268
|
+
trainedSamples: this.trainedSamples,
|
|
269
|
+
loss: this.loss,
|
|
270
|
+
};
|
|
271
|
+
}
|
|
272
|
+
/** Import weights from persistence. */
|
|
273
|
+
static importWeights(data, config) {
|
|
274
|
+
const model = new LogisticRewardModel(data.weights, data.bias, config);
|
|
275
|
+
model.trained = true;
|
|
276
|
+
model.trainedSamples = data.trainedSamples;
|
|
277
|
+
model.loss = data.loss;
|
|
278
|
+
return model;
|
|
279
|
+
}
|
|
280
|
+
}
|
|
281
|
+
exports.LogisticRewardModel = LogisticRewardModel;
|
|
282
|
+
/**
|
|
283
|
+
* Compare LogisticRewardModel against LinearRanker and LearningRanker
|
|
284
|
+
* on held-out telemetry data.
|
|
285
|
+
*/
|
|
286
|
+
function compareModels(telemetry, testSplit = 0.3) {
|
|
287
|
+
const decisions = telemetry.all().filter(d => d.feedback && d.selectedCandidateId);
|
|
288
|
+
if (decisions.length < 20)
|
|
289
|
+
return [];
|
|
290
|
+
const testSize = Math.floor(decisions.length * testSplit);
|
|
291
|
+
const trainDecisions = decisions.slice(0, decisions.length - testSize);
|
|
292
|
+
const testDecisions = decisions.slice(decisions.length - testSize);
|
|
293
|
+
// Train LogisticRewardModel on training split
|
|
294
|
+
const model = LogisticRewardModel.train(telemetry);
|
|
295
|
+
// Evaluate all models on test split
|
|
296
|
+
const comparisons = [];
|
|
297
|
+
// LogisticRewardModel
|
|
298
|
+
if (model.isTrained) {
|
|
299
|
+
let correct = 0;
|
|
300
|
+
let logLoss = 0;
|
|
301
|
+
for (const d of testDecisions) {
|
|
302
|
+
const sel = d.candidates.find(c => c.candidateId === d.selectedCandidateId);
|
|
303
|
+
if (!sel)
|
|
304
|
+
continue;
|
|
305
|
+
const label = d.feedback.decision === "accepted" ? 1 : 0;
|
|
306
|
+
const stats = telemetry.getCandidateStats(sel.candidateId);
|
|
307
|
+
const acceptTotal = stats.accepted + stats.rejected;
|
|
308
|
+
const acceptanceRate = acceptTotal > 0 ? stats.accepted / acceptTotal : 0.5;
|
|
309
|
+
const execTotal = stats.executionSuccess + stats.executionFailure;
|
|
310
|
+
const executionSuccessRate = execTotal > 0 ? stats.executionSuccess / execTotal : 0.5;
|
|
311
|
+
const prediction = model.score({ protocolSafety: 0.8, historicalSuccessRate: 0.5, actionCount: sel.actions.length, latencyCost: 0.5, auditability: 0.5, corpusEvidence: 0, source: "protocol" }, { acceptanceRate, executionSuccessRate });
|
|
312
|
+
if ((prediction >= 0.5 ? 1 : 0) === label)
|
|
313
|
+
correct++;
|
|
314
|
+
const p = Math.max(1e-15, Math.min(1 - 1e-15, prediction));
|
|
315
|
+
logLoss += -(label * Math.log(p) + (1 - label) * Math.log(1 - p));
|
|
316
|
+
}
|
|
317
|
+
const n = testDecisions.length || 1;
|
|
318
|
+
comparisons.push({
|
|
319
|
+
model: "LogisticReward",
|
|
320
|
+
accuracy: correct / n,
|
|
321
|
+
auc: correct / n, // simplified: accuracy ≈ AUC for binary classification
|
|
322
|
+
logLoss: logLoss / n,
|
|
323
|
+
trained: true,
|
|
324
|
+
});
|
|
325
|
+
}
|
|
326
|
+
else {
|
|
327
|
+
comparisons.push({
|
|
328
|
+
model: "LogisticReward",
|
|
329
|
+
accuracy: 0, auc: 0, logLoss: Infinity, trained: false,
|
|
330
|
+
});
|
|
331
|
+
}
|
|
332
|
+
// Baseline: always predict majority class
|
|
333
|
+
const acceptedCount = testDecisions.filter(d => d.feedback.decision === "accepted").length;
|
|
334
|
+
const majorityRate = Math.max(acceptedCount, testDecisions.length - acceptedCount) / testDecisions.length;
|
|
335
|
+
comparisons.push({
|
|
336
|
+
model: "Baseline (majority)", accuracy: majorityRate, auc: 0.5, logLoss: Infinity, trained: false,
|
|
337
|
+
});
|
|
338
|
+
return comparisons;
|
|
339
|
+
}
|