progmune-runtime 2.1.6 → 3.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +108 -468
- package/dist/ablation-study.js +144 -0
- package/dist/ablation-study.test.js +18 -0
- package/dist/action-runtime.js +3 -1
- package/dist/active-learning.js +211 -0
- package/dist/analytics.js +139 -0
- package/dist/asset-factory.js +309 -0
- package/dist/asset-growth.js +244 -0
- package/dist/asset-promotion.js +382 -0
- package/dist/asset-quality.js +550 -0
- package/dist/audit/business-translator.js +285 -0
- package/dist/audit/cli.js +66 -0
- package/dist/audit/formatters/html.js +379 -0
- package/dist/audit/formatters/json.js +11 -0
- package/dist/audit/formatters/markdown.js +192 -0
- package/dist/audit/formatters/terminal.js +189 -0
- package/dist/audit/index.js +25 -0
- package/dist/audit/report-builder.js +318 -0
- package/dist/audit/types.js +8 -0
- package/dist/audit.js +3 -3
- package/dist/auto-benchmark-generator.js +137 -0
- package/dist/auto-benchmark-generator.test.js +45 -0
- package/dist/auto-protocol-synthesizer.js +362 -0
- package/dist/auto-protocol-synthesizer.test.js +82 -0
- package/dist/autonomous-patch.js +175 -0
- package/dist/autonomous-patch.test.js +128 -0
- package/dist/badge/badge-server.js +98 -0
- package/dist/behavior-miner.js +442 -0
- package/dist/belief-layer.js +475 -0
- package/dist/benchmark-count.js +5 -0
- package/dist/benchmark-generator.js +211 -0
- package/dist/benchmark-harness.js +201 -0
- package/dist/benchmark-pass-rate.js +7 -0
- package/dist/benchmark-report.js +8 -3
- package/dist/benchmark-save.js +14 -1
- package/dist/bootstrap-validation.js +197 -0
- package/dist/bootstrap-validation.test.js +51 -0
- package/dist/branch-ledger.js +1 -1
- package/dist/capability-gap.js +130 -0
- package/dist/certify-html.js +351 -0
- package/dist/certify.js +326 -0
- package/dist/check.js +4 -4
- package/dist/compliance-miner.js +447 -0
- package/dist/continuous-benchmark.js +194 -0
- package/dist/continuous-benchmark.test.js +116 -0
- package/dist/corpus-stats.js +173 -0
- package/dist/counterfactual-engine.js +288 -0
- package/dist/coverage-dashboard.js +109 -0
- package/dist/coverage-system.test.js +205 -0
- package/dist/cross-repo-precision.js +352 -0
- package/dist/cve-benchmark.js +180 -0
- package/dist/cve-benchmark.test.js +28 -0
- package/dist/cve-collector.js +73 -0
- package/dist/data-quality.js +141 -0
- package/dist/decision-engine.js +388 -0
- package/dist/derive-metadata.js +250 -0
- package/dist/difficulty-active.test.js +198 -0
- package/dist/difficulty-map.js +244 -0
- package/dist/discovery-analytics.js +125 -0
- package/dist/discovery-model.js +149 -0
- package/dist/discovery-optimize.test.js +199 -0
- package/dist/discovery-trace.js +276 -0
- package/dist/discovery-trace.test.js +97 -0
- package/dist/emitter.js +83 -1
- package/dist/enterprise-dashboard.js +405 -0
- package/dist/eval-hardening.js +297 -0
- package/dist/eval-hardening.test.js +85 -0
- package/dist/evaluation-campaign.js +359 -0
- package/dist/evaluation-campaign.test.js +181 -0
- package/dist/evidence-growth.js +143 -0
- package/dist/evidence-repository.js +209 -0
- package/dist/evidence-system.js +441 -0
- package/dist/execute.js +15 -7
- package/dist/experimental/software-physics.js +291 -0
- package/dist/experimental/state-inference.js +516 -0
- package/dist/experimental/unsupervised-physics.js +230 -0
- package/dist/extract-ir-python.js +54 -7
- package/dist/extract-ir.js +376 -12
- package/dist/failure-collector.js +2 -2
- package/dist/failure-corpus.js +322 -9
- package/dist/feedback.js +16 -5
- package/dist/feedback.test.js +49 -0
- package/dist/file-lock.js +1 -1
- package/dist/flywheel-batch.js +292 -0
- package/dist/frameworks/express-cli.js +237 -0
- package/dist/frameworks/express-detector.js +445 -0
- package/dist/frameworks/express-detector.test.js +206 -0
- package/dist/frameworks/index.js +30 -0
- package/dist/frameworks/nestjs-detector.js +302 -0
- package/dist/frameworks/trpc-detector.js +161 -0
- package/dist/frameworks/version-awareness.js +179 -0
- package/dist/function-synonyms.js +164 -0
- package/dist/function-synonyms.test.js +68 -0
- package/dist/generalization.test.js +352 -0
- package/dist/goal-annotator.js +113 -0
- package/dist/goal-planner.js +563 -0
- package/dist/gold-cve.js +164 -0
- package/dist/gold-cve.test.js +104 -0
- package/dist/gold-quality.js +206 -0
- package/dist/gold-tiers.js +241 -0
- package/dist/governance-dashboard.js +327 -0
- package/dist/graph-viz.js +240 -0
- package/dist/guided-frontier.js +195 -0
- package/dist/hierarchical-planner.js +148 -0
- package/dist/identifier-parser.js +260 -0
- package/dist/immune-metrics.js +93 -0
- package/dist/immune-receiver.js +158 -0
- package/dist/immune-reporter.js +1 -1
- package/dist/improvement-orchestrator.js +206 -0
- package/dist/inject-p0-vocabulary.js +300 -0
- package/dist/intent-parser.js +218 -0
- package/dist/invariant-algebra.js +476 -0
- package/dist/invariant-calculus.js +533 -0
- package/dist/ir-utils.js +70 -0
- package/dist/ir-utils.test.js +50 -0
- package/dist/knowledge-api.js +312 -0
- package/dist/knowledge-evolution.js +452 -0
- package/dist/knowledge-explorer.js +506 -0
- package/dist/knowledge-flywheel.js +274 -0
- package/dist/knowledge-governance.js +338 -0
- package/dist/knowledge-governance.test.js +150 -0
- package/dist/knowledge-graph.js +181 -0
- package/dist/knowledge-guided-synth.js +246 -0
- package/dist/knowledge-loop.test.js +77 -0
- package/dist/knowledge-object.js +316 -0
- package/dist/knowledge-package.js +98 -0
- package/dist/kpi-dashboard.js +561 -0
- package/dist/l3-cross-function.js +280 -0
- package/dist/learning-ranker.js +148 -0
- package/dist/learning-ranker.test.js +291 -0
- package/dist/ledger/accountability.js +322 -0
- package/dist/ledger/chain-builder.js +185 -0
- package/dist/ledger/cli.js +222 -0
- package/dist/ledger/index.js +13 -0
- package/dist/ledger/signatures.js +193 -0
- package/dist/ledger/types.js +9 -0
- package/dist/llm.js +74 -3
- package/dist/load-benchmarks.js +8 -3
- package/dist/logger.js +66 -0
- package/dist/logger.test.js +37 -0
- package/dist/logistic-reward.js +339 -0
- package/dist/logistic-reward.test.js +180 -0
- package/dist/macro-graph.js +193 -0
- package/dist/macro-repair.js +183 -0
- package/dist/mcp-server.mjs +1202 -483
- package/dist/memory-layer.js +42 -5
- package/dist/multi-repo-precision.js +422 -0
- package/dist/name-free-protocol.js +425 -0
- package/dist/name-free-protocol.test.js +170 -0
- package/dist/name-scrambling.js +138 -0
- package/dist/name-scrambling.test.js +16 -0
- package/dist/p3-observability.test.js +281 -0
- package/dist/p5-orchestrator.test.js +225 -0
- package/dist/pairwise-preference.js +294 -0
- package/dist/pairwise-preference.test.js +140 -0
- package/dist/planner-constraints.js +104 -0
- package/dist/planner-prompts.js +155 -0
- package/dist/planner-telemetry.js +415 -0
- package/dist/planner-trace.js +214 -0
- package/dist/planner.js +162 -167
- package/dist/plsb/artifact.js +116 -0
- package/dist/plsb/cli.js +71 -0
- package/dist/plsb/index.js +19 -0
- package/dist/plsb/leaderboard.js +249 -0
- package/dist/plsb/report-md.js +156 -0
- package/dist/plsb/schema.js +179 -0
- package/dist/plsb-benchmark.js +284 -0
- package/dist/plsb-benchmark.test.js +119 -0
- package/dist/policy/cli.js +134 -0
- package/dist/policy/engine.js +333 -0
- package/dist/policy/index.js +12 -0
- package/dist/policy/types.js +59 -0
- package/dist/policy-miner.js +505 -0
- package/dist/precision-analyze.js +229 -0
- package/dist/precision-benchmark.js +147 -0
- package/dist/precision-label-c.js +134 -0
- package/dist/precision-label.js +193 -0
- package/dist/precision-report-c.js +149 -0
- package/dist/precision-report.js +246 -0
- package/dist/progmune-status.js +108 -0
- package/dist/proof-engine.js +479 -0
- package/dist/proof-provenance.js +315 -0
- package/dist/protocol-coverage.js +294 -0
- package/dist/protocol-detector.js +1189 -0
- package/dist/protocol-embedding-expanded.js +297 -0
- package/dist/protocol-embedding-expanded.test.js +97 -0
- package/dist/protocol-embedding.js +195 -0
- package/dist/protocol-embedding.test.js +82 -0
- package/dist/protocol-extractor-v2.js +354 -0
- package/dist/protocol-extractor-v2.test.js +140 -0
- package/dist/protocol-extractor.js +310 -0
- package/dist/protocol-extractor.test.js +113 -0
- package/dist/protocol-foundation.js +322 -0
- package/dist/protocol-foundation.test.js +163 -0
- package/dist/protocol-frontier.js +243 -0
- package/dist/protocol-frontier.test.js +92 -0
- package/dist/protocol-gap-analyzer.js +228 -0
- package/dist/protocol-gap-analyzer.test.js +49 -0
- package/dist/protocol-invariants.js +276 -0
- package/dist/protocol-invariants.test.js +111 -0
- package/dist/protocol-knowledge.js +464 -0
- package/dist/protocol-miner.js +343 -0
- package/dist/protocol-mining.js +207 -0
- package/dist/protocol-mining.test.js +37 -0
- package/dist/protocol-registry.js +1 -1
- package/dist/protocol-security-benchmark.js +222 -0
- package/dist/protocol-vulnerability.js +257 -0
- package/dist/protocol-vulnerability.test.js +60 -0
- package/dist/python-benchmark.js +120 -0
- package/dist/python-emitter.js +163 -45
- package/dist/python-protocol-extractor.js +187 -0
- package/dist/python-protocol-extractor.test.js +116 -0
- package/dist/realworld-benchmark.js +646 -0
- package/dist/realworld-benchmark.test.js +36 -0
- package/dist/repair-arch.test.js +411 -0
- package/dist/repair-evolution.test.js +454 -0
- package/dist/repair-executor.js +719 -0
- package/dist/repair-proposal.js +4 -4
- package/dist/repair-ranker.js +141 -0
- package/dist/repair-strategies.js +419 -0
- package/dist/repair-taxonomy.js +234 -0
- package/dist/repair-types.js +12 -0
- package/dist/repo-evaluator.js +250 -0
- package/dist/repo-evaluator.test.js +128 -0
- package/dist/resource-abstraction.js +242 -0
- package/dist/resource-detector.js +211 -0
- package/dist/result.test.js +43 -0
- package/dist/reward-system.js +411 -0
- package/dist/reward-system.test.js +175 -0
- package/dist/risk-model.js +215 -0
- package/dist/rule-miner.js +234 -7
- package/dist/rule-specificity.js +254 -0
- package/dist/runtime-types.js +27 -0
- package/dist/scaffold.js +208 -0
- package/dist/scale-collector.test.js +101 -0
- package/dist/scale-trajectory-collector.js +128 -0
- package/dist/sdk.js +250 -0
- package/dist/search-planner.js +4 -41
- package/dist/semantic-snapshot.js +1 -1
- package/dist/semantic-topology.js +121 -0
- package/dist/semantic-trace.js +310 -317
- package/dist/sequence-extractor.js +343 -0
- package/dist/skill-library.js +245 -0
- package/dist/skill-planner.test.js +189 -0
- package/dist/software-physics.js +291 -0
- package/dist/software-physics.test.js +81 -0
- package/dist/ssg-precision.js +478 -0
- package/dist/ssg-validator.js +71 -21
- package/dist/state-inference-doubleblind.test.js +160 -0
- package/dist/state-inference.js +516 -0
- package/dist/state-inference.test.js +115 -0
- package/dist/state-machine-fingerprint.js +345 -0
- package/dist/state-machine-fingerprint.test.js +120 -0
- package/dist/state-miner.js +386 -0
- package/dist/state-name-inference.js +213 -0
- package/dist/state-name-inference.test.js +69 -0
- package/dist/strategy-planner.js +262 -96
- package/dist/strategy-planner.test.js +135 -0
- package/dist/telemetry-analytics.test.js +402 -0
- package/dist/terminal-format.js +68 -0
- package/dist/terminal-format.test.js +83 -0
- package/dist/topology-factory.js +196 -0
- package/dist/topology-representation.js +242 -0
- package/dist/topology-representation.test.js +27 -0
- package/dist/trajectory-augmentation.js +254 -0
- package/dist/trajectory-augmentation.test.js +63 -0
- package/dist/trajectory-corpus.js +440 -0
- package/dist/trajectory-corpus.test.js +32 -0
- package/dist/trajectory-feedback.test.js +116 -0
- package/dist/transition-synthesizer.js +286 -0
- package/dist/transition-synthesizer.test.js +123 -0
- package/dist/trust/api-semantic-mapper.js +809 -0
- package/dist/trust/call-graph-propagator.js +225 -0
- package/dist/trust/cli.js +122 -0
- package/dist/trust/compliance-scorer.js +283 -0
- package/dist/trust/confidence-calculator.js +261 -0
- package/dist/trust/engine.js +1145 -0
- package/dist/trust/explainability.js +85 -0
- package/dist/trust/formatters/ci.js +42 -0
- package/dist/trust/formatters/json.js +11 -0
- package/dist/trust/formatters/terminal.js +152 -0
- package/dist/trust/index.js +39 -0
- package/dist/trust/phase1-verify.js +171 -0
- package/dist/trust/protocol-domain-validator.js +697 -0
- package/dist/trust/score-calculator.js +282 -0
- package/dist/trust/ssg-bridge.js +641 -0
- package/dist/trust/ssg-bridge.test.js +269 -0
- package/dist/trust/types.js +67 -0
- package/dist/trust/violation-trace.js +335 -0
- package/dist/trust-api.js +179 -0
- package/dist/trust-calibration.js +279 -0
- package/dist/unknown-protocol-discovery.js +339 -0
- package/dist/unknown-protocol-discovery.test.js +102 -0
- package/dist/unsupervised-physics.js +230 -0
- package/dist/unsupervised-physics.test.js +95 -0
- package/dist/utils.test.js +37 -0
- package/dist/validator.js +187 -10
- package/dist/verification-intelligence.js +475 -0
- package/dist/verify-api.js +432 -0
- package/dist/vi-impact-report.js +293 -0
- package/dist/wl-fingerprint.js +162 -0
- package/dist/wl-fingerprint.test.js +130 -0
- package/dist/zeroshot-strategy.js +139 -0
- package/docs/Progmune_/346/212/225/350/265/204/344/272/272/347/231/275/347/232/256/344/271/246_v2.0.html +576 -0
- package/docs/Progmune_/351/241/271/347/233/256/345/205/250/350/247/243.html +710 -0
- package/package.json +74 -7
- package/protocols.json +1956 -50
- package/.dockerignore +0 -14
- package/.mcp.json +0 -11
- package/.progmune_allowlist +0 -50
- package/.test_report/test_report.md +0 -87
- package/Dockerfile +0 -9
- package/FAQ.md +0 -167
- package/WHITEPAPER.md +0 -540
- package/demo-project/auth.ts +0 -55
- package/demo-project/tsconfig.json +0 -8
- package/dist/acl-breakdown.js +0 -13
- package/dist/all-sessions.js +0 -11
- package/dist/antibody-stats.js +0 -11
- package/dist/branch-tree-count.js +0 -14
- package/dist/common-fixpath.js +0 -12
- package/dist/constraint-types.js +0 -12
- package/dist/exec-metrics.js +0 -11
- package/dist/failure-report.js +0 -11
- package/dist/fast-path-hits.js +0 -13
- package/dist/fingerprint-list.js +0 -15
- package/dist/gen-history-log.js +0 -13
- package/dist/heatmap-data.js +0 -11
- package/dist/recent-session.js +0 -12
- package/dist/svl-distribution.js +0 -11
- package/dist/terminal-status.js +0 -11
- package/dist/token-savings.js +0 -11
- package/dist/total-repairs.js +0 -12
- package/dist/unresolved-count.js +0 -12
- package/dist/valid-fingerprints.js +0 -13
- package/dist/verify-ledgers.js +0 -11
- package/docs/whitepaper-style.css +0 -77
- package/docs/whitepaper-v2.1.md +0 -609
- package/docs/whitepaper-v2.2.md +0 -1064
- package/docs/whitepaper-v2.2.pdf +0 -0
- package/fly.toml +0 -31
- package/public/dashboard.html +0 -119
- package/server/hub.js +0 -116
- package/test/replay-golden/sess_1780063202050_mgeld.json +0 -9
- package/test/replay-golden/sess_1780064032560_gocld.json +0 -354
- package/test/replay-golden/sess_1780064413331_s2709.json +0 -606
- package/test/replay-golden/sess_1780064792710_y3avo.json +0 -614
- package/test/replay-golden.ts +0 -84
- package/test_benchmark.js +0 -165
- package/test_comprehensive.mjs +0 -638
- package/test_concurrency.js +0 -129
- package/test_ir_robustness.js +0 -85
- package/test_semantic_contracts.js +0 -269
- package/test_ssg_stress.js +0 -156
- package/test_svl3.js +0 -58
- package/tsconfig.json +0 -17
|
@@ -0,0 +1,447 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* Progmune V11 — Compliance Knowledge Layer
|
|
4
|
+
* ===========================================
|
|
5
|
+
* 将 SOC 2、ISO 27001、EU AI Act、NIST、OWASP 等合规标准
|
|
6
|
+
* 编码为 Protocol Invariant,纳入统一的 Knowledge Object 体系。
|
|
7
|
+
*
|
|
8
|
+
* 核心理念:
|
|
9
|
+
* 合规要求 = 另一种来源的 Protocol
|
|
10
|
+
* SOC 2 CC7.2 "所有高权限操作必须记录审计日志"
|
|
11
|
+
* → Invariant: PrivilegeChange ⇒ AuditRecorded
|
|
12
|
+
*
|
|
13
|
+
* 与 Business Claim 使用完全相同的 Claim/Proof/Belief 框架。
|
|
14
|
+
* 只是 origin 不同。
|
|
15
|
+
*/
|
|
16
|
+
var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
|
|
17
|
+
if (k2 === undefined) k2 = k;
|
|
18
|
+
var desc = Object.getOwnPropertyDescriptor(m, k);
|
|
19
|
+
if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
|
|
20
|
+
desc = { enumerable: true, get: function() { return m[k]; } };
|
|
21
|
+
}
|
|
22
|
+
Object.defineProperty(o, k2, desc);
|
|
23
|
+
}) : (function(o, m, k, k2) {
|
|
24
|
+
if (k2 === undefined) k2 = k;
|
|
25
|
+
o[k2] = m[k];
|
|
26
|
+
}));
|
|
27
|
+
var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
|
|
28
|
+
Object.defineProperty(o, "default", { enumerable: true, value: v });
|
|
29
|
+
}) : function(o, v) {
|
|
30
|
+
o["default"] = v;
|
|
31
|
+
});
|
|
32
|
+
var __importStar = (this && this.__importStar) || (function () {
|
|
33
|
+
var ownKeys = function(o) {
|
|
34
|
+
ownKeys = Object.getOwnPropertyNames || function (o) {
|
|
35
|
+
var ar = [];
|
|
36
|
+
for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
|
|
37
|
+
return ar;
|
|
38
|
+
};
|
|
39
|
+
return ownKeys(o);
|
|
40
|
+
};
|
|
41
|
+
return function (mod) {
|
|
42
|
+
if (mod && mod.__esModule) return mod;
|
|
43
|
+
var result = {};
|
|
44
|
+
if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
|
|
45
|
+
__setModuleDefault(result, mod);
|
|
46
|
+
return result;
|
|
47
|
+
};
|
|
48
|
+
})();
|
|
49
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
50
|
+
exports.buildComplianceLayer = buildComplianceLayer;
|
|
51
|
+
const fs = __importStar(require("fs"));
|
|
52
|
+
const path = __importStar(require("path"));
|
|
53
|
+
// ══════════════════════════════════════════════
|
|
54
|
+
// COMPLIANCE KNOWLEDGE BASE
|
|
55
|
+
// ══════════════════════════════════════════════
|
|
56
|
+
/**
|
|
57
|
+
* Key regulatory requirements encoded as Protocol Invariants.
|
|
58
|
+
* Each is a Claim that can be verified, refuted, and believed —
|
|
59
|
+
* using the same framework as Business Claims.
|
|
60
|
+
*/
|
|
61
|
+
const COMPLIANCE_REQUIREMENTS = [
|
|
62
|
+
// ── SOC 2 ──
|
|
63
|
+
{
|
|
64
|
+
id: "SOC2-CC6.1",
|
|
65
|
+
standard: "SOC2",
|
|
66
|
+
clause: "CC6.1",
|
|
67
|
+
description: "逻辑和物理访问控制——只有授权实体可以访问系统资源",
|
|
68
|
+
invariant: "ResourceAccess ⇒ Authenticated",
|
|
69
|
+
severity: "critical",
|
|
70
|
+
category: "access_control",
|
|
71
|
+
evidenceRequired: "每次资源访问前有认证检查的代码证据",
|
|
72
|
+
},
|
|
73
|
+
{
|
|
74
|
+
id: "SOC2-CC6.3",
|
|
75
|
+
standard: "SOC2",
|
|
76
|
+
clause: "CC6.3",
|
|
77
|
+
description: "职责分离——高权限操作需要独立审批",
|
|
78
|
+
invariant: "AdminAction ⇒ IndependentReview",
|
|
79
|
+
severity: "critical",
|
|
80
|
+
category: "access_control",
|
|
81
|
+
evidenceRequired: "管理操作触发审批流程的代码证据",
|
|
82
|
+
},
|
|
83
|
+
{
|
|
84
|
+
id: "SOC2-CC7.1",
|
|
85
|
+
standard: "SOC2",
|
|
86
|
+
clause: "CC7.1",
|
|
87
|
+
description: "审计日志——所有高权限操作必须记录",
|
|
88
|
+
invariant: "PrivilegeChange ⇒ AuditRecorded",
|
|
89
|
+
severity: "critical",
|
|
90
|
+
category: "audit",
|
|
91
|
+
evidenceRequired: "权限变更操作写入审计日志的代码证据",
|
|
92
|
+
},
|
|
93
|
+
{
|
|
94
|
+
id: "SOC2-CC7.2",
|
|
95
|
+
standard: "SOC2",
|
|
96
|
+
clause: "CC7.2",
|
|
97
|
+
description: "系统事件监控——异常行为检测和告警",
|
|
98
|
+
invariant: "AnomalyDetected ⇒ AlertTriggered",
|
|
99
|
+
severity: "high",
|
|
100
|
+
category: "monitoring",
|
|
101
|
+
evidenceRequired: "异常检测到告警触发的完整链路",
|
|
102
|
+
},
|
|
103
|
+
{
|
|
104
|
+
id: "SOC2-CC8.1",
|
|
105
|
+
standard: "SOC2",
|
|
106
|
+
clause: "CC8.1",
|
|
107
|
+
description: "变更管理——所有系统变更需要授权和记录",
|
|
108
|
+
invariant: "SystemChange ⇒ Authorized",
|
|
109
|
+
severity: "high",
|
|
110
|
+
category: "change_management",
|
|
111
|
+
evidenceRequired: "变更授权和记录的代码/流程证据",
|
|
112
|
+
},
|
|
113
|
+
// ── ISO 27001 ──
|
|
114
|
+
{
|
|
115
|
+
id: "ISO-A.9.2",
|
|
116
|
+
standard: "ISO27001",
|
|
117
|
+
clause: "A.9.2",
|
|
118
|
+
description: "用户注册和注销——正式的用户访问配置流程",
|
|
119
|
+
invariant: "UserCreated ⇒ AccessPolicy",
|
|
120
|
+
severity: "critical",
|
|
121
|
+
category: "access_control",
|
|
122
|
+
evidenceRequired: "用户创建时自动配置访问策略的代码证据",
|
|
123
|
+
},
|
|
124
|
+
{
|
|
125
|
+
id: "ISO-A.9.4",
|
|
126
|
+
standard: "ISO27001",
|
|
127
|
+
clause: "A.9.4",
|
|
128
|
+
description: "秘密认证信息——密码和令牌的安全管理",
|
|
129
|
+
invariant: "CredentialStored ⇒ HashedOrEncrypted",
|
|
130
|
+
severity: "critical",
|
|
131
|
+
category: "secrets",
|
|
132
|
+
evidenceRequired: "密码哈希或令牌加密存储的代码证据",
|
|
133
|
+
},
|
|
134
|
+
{
|
|
135
|
+
id: "ISO-A.12.4",
|
|
136
|
+
standard: "ISO27001",
|
|
137
|
+
clause: "A.12.4",
|
|
138
|
+
description: "事件日志——管理员和操作员活动需记录",
|
|
139
|
+
invariant: "OperatorAction ⇒ Logged",
|
|
140
|
+
severity: "high",
|
|
141
|
+
category: "audit",
|
|
142
|
+
evidenceRequired: "操作活动被日志记录的代码证据",
|
|
143
|
+
},
|
|
144
|
+
{
|
|
145
|
+
id: "ISO-A.14.2",
|
|
146
|
+
standard: "ISO27001",
|
|
147
|
+
clause: "A.14.2",
|
|
148
|
+
description: "安全开发——输入数据需验证,输出数据需清理",
|
|
149
|
+
invariant: "ExternalInput ⇒ Validated",
|
|
150
|
+
severity: "critical",
|
|
151
|
+
category: "data_protection",
|
|
152
|
+
evidenceRequired: "外部输入经过验证的代码证据(如 Zod schema)",
|
|
153
|
+
},
|
|
154
|
+
// ── EU AI Act ──
|
|
155
|
+
{
|
|
156
|
+
id: "AIACT-Art9",
|
|
157
|
+
standard: "EU_AI_ACT",
|
|
158
|
+
clause: "Article 9",
|
|
159
|
+
description: "风险管理——AI 系统需持续进行风险评估",
|
|
160
|
+
invariant: "AIModelUpdate ⇒ RiskAssessment",
|
|
161
|
+
severity: "critical",
|
|
162
|
+
category: "ai_governance",
|
|
163
|
+
evidenceRequired: "AI 模型更新前触发风险评估的证据",
|
|
164
|
+
},
|
|
165
|
+
{
|
|
166
|
+
id: "AIACT-Art14",
|
|
167
|
+
standard: "EU_AI_ACT",
|
|
168
|
+
clause: "Article 14",
|
|
169
|
+
description: "人工监督——高风险 AI 决策需人工复核",
|
|
170
|
+
invariant: "HighRiskDecision ⇒ HumanReview",
|
|
171
|
+
severity: "critical",
|
|
172
|
+
category: "ai_governance",
|
|
173
|
+
evidenceRequired: "高风险决策触发人工复核流程的证据",
|
|
174
|
+
},
|
|
175
|
+
{
|
|
176
|
+
id: "AIACT-Art12",
|
|
177
|
+
standard: "EU_AI_ACT",
|
|
178
|
+
clause: "Article 12",
|
|
179
|
+
description: "记录保存——AI 系统操作需自动记录日志",
|
|
180
|
+
invariant: "AISystemOperation ⇒ Logged",
|
|
181
|
+
severity: "high",
|
|
182
|
+
category: "ai_governance",
|
|
183
|
+
evidenceRequired: "AI 操作自动记录日志的证据",
|
|
184
|
+
},
|
|
185
|
+
{
|
|
186
|
+
id: "AIACT-Art15",
|
|
187
|
+
standard: "EU_AI_ACT",
|
|
188
|
+
clause: "Article 15",
|
|
189
|
+
description: "透明度和可追溯性——AI 生成内容需可追溯到来源",
|
|
190
|
+
invariant: "AIGeneratedCode ⇒ TraceableToPrompt",
|
|
191
|
+
severity: "critical",
|
|
192
|
+
category: "ai_governance",
|
|
193
|
+
evidenceRequired: "AI 生成代码可追溯到 Prompt 的完整链路",
|
|
194
|
+
},
|
|
195
|
+
// ── NIST 800-53 ──
|
|
196
|
+
{
|
|
197
|
+
id: "NIST-AC-2",
|
|
198
|
+
standard: "NIST",
|
|
199
|
+
clause: "AC-2",
|
|
200
|
+
description: "账户管理——所有账户的生命周期需受控",
|
|
201
|
+
invariant: "AccountCreated ⇒ Authorized",
|
|
202
|
+
severity: "critical",
|
|
203
|
+
category: "access_control",
|
|
204
|
+
evidenceRequired: "账户创建需授权的代码证据",
|
|
205
|
+
},
|
|
206
|
+
{
|
|
207
|
+
id: "NIST-IA-5",
|
|
208
|
+
standard: "NIST",
|
|
209
|
+
clause: "IA-5",
|
|
210
|
+
description: "认证器管理——密码强度、重置、多因素认证",
|
|
211
|
+
invariant: "PasswordReset ⇒ MFAVerified",
|
|
212
|
+
severity: "critical",
|
|
213
|
+
category: "secrets",
|
|
214
|
+
evidenceRequired: "密码重置需多因素认证的代码证据",
|
|
215
|
+
},
|
|
216
|
+
{
|
|
217
|
+
id: "NIST-AU-3",
|
|
218
|
+
standard: "NIST",
|
|
219
|
+
clause: "AU-3",
|
|
220
|
+
description: "审计记录内容——日志需包含足够信息以追溯事件",
|
|
221
|
+
invariant: "SecurityEvent ⇒ LoggedWithContext",
|
|
222
|
+
severity: "high",
|
|
223
|
+
category: "audit",
|
|
224
|
+
evidenceRequired: "安全事件日志包含完整上下文信息的证据",
|
|
225
|
+
},
|
|
226
|
+
// ── OWASP ASVS ──
|
|
227
|
+
{
|
|
228
|
+
id: "OWASP-V2.1",
|
|
229
|
+
standard: "OWASP",
|
|
230
|
+
clause: "V2.1",
|
|
231
|
+
description: "密码安全——所有密码需使用认可的哈希算法存储",
|
|
232
|
+
invariant: "PasswordStored ⇒ BcryptOrArgon2",
|
|
233
|
+
severity: "critical",
|
|
234
|
+
category: "secrets",
|
|
235
|
+
evidenceRequired: "密码使用 bcrypt/argon2 哈希存储的代码证据",
|
|
236
|
+
},
|
|
237
|
+
{
|
|
238
|
+
id: "OWASP-V4.1",
|
|
239
|
+
standard: "OWASP",
|
|
240
|
+
clause: "V4.1",
|
|
241
|
+
description: "访问控制——每次资源访问需验证权限",
|
|
242
|
+
invariant: "ResourceAccess ⇒ AuthorizationChecked",
|
|
243
|
+
severity: "critical",
|
|
244
|
+
category: "access_control",
|
|
245
|
+
evidenceRequired: "每次资源访问前检查权限的代码证据",
|
|
246
|
+
},
|
|
247
|
+
{
|
|
248
|
+
id: "OWASP-V5.2",
|
|
249
|
+
standard: "OWASP",
|
|
250
|
+
clause: "V5.2",
|
|
251
|
+
description: "输入验证——所有外部输入需经过服务端验证",
|
|
252
|
+
invariant: "ExternalInput ⇒ ServerSideValidated",
|
|
253
|
+
severity: "critical",
|
|
254
|
+
category: "data_protection",
|
|
255
|
+
evidenceRequired: "服务端验证外部输入的代码证据(Zod/Yup schema)",
|
|
256
|
+
},
|
|
257
|
+
];
|
|
258
|
+
// ══════════════════════════════════════════════
|
|
259
|
+
// COVERAGE ANALYSIS
|
|
260
|
+
// ══════════════════════════════════════════════
|
|
261
|
+
/**
|
|
262
|
+
* Analyze how existing business claims map to compliance requirements.
|
|
263
|
+
*
|
|
264
|
+
* Mapping logic:
|
|
265
|
+
* - If a business claim's predicate overlaps with a compliance invariant,
|
|
266
|
+
* the business claim partially satisfies the compliance requirement.
|
|
267
|
+
* - Scans actual code for evidence patterns (auth middleware, bcrypt, Zod, etc.)
|
|
268
|
+
* - Maps compliance concepts to real code artifacts
|
|
269
|
+
*/
|
|
270
|
+
function analyzeCoverage(businessClaims, complianceReqs, projectPath) {
|
|
271
|
+
const standards = [...new Set(complianceReqs.map(r => r.standard))];
|
|
272
|
+
const coverage = [];
|
|
273
|
+
const evidence = collectCodeEvidence(projectPath);
|
|
274
|
+
for (const standard of standards) {
|
|
275
|
+
const reqs = complianceReqs.filter(r => r.standard === standard);
|
|
276
|
+
const satisfied = [];
|
|
277
|
+
const unsatisfied = [];
|
|
278
|
+
for (const req of reqs) {
|
|
279
|
+
if (checkRequirement(req, evidence, businessClaims)) {
|
|
280
|
+
satisfied.push(req.id);
|
|
281
|
+
}
|
|
282
|
+
else {
|
|
283
|
+
unsatisfied.push(req.id);
|
|
284
|
+
}
|
|
285
|
+
}
|
|
286
|
+
coverage.push({
|
|
287
|
+
standard,
|
|
288
|
+
totalRequirements: reqs.length,
|
|
289
|
+
satisfiedByBusiness: satisfied.length,
|
|
290
|
+
satisfiedByEvidence: satisfied.length,
|
|
291
|
+
unsatisfied: unsatisfied.length,
|
|
292
|
+
claims: satisfied,
|
|
293
|
+
});
|
|
294
|
+
}
|
|
295
|
+
return coverage;
|
|
296
|
+
}
|
|
297
|
+
function collectCodeEvidence(projectPath) {
|
|
298
|
+
const ev = {
|
|
299
|
+
files: new Set(),
|
|
300
|
+
imports: new Set(),
|
|
301
|
+
dependencies: new Set(),
|
|
302
|
+
patterns: new Map(),
|
|
303
|
+
};
|
|
304
|
+
const serverDir = path.join(projectPath, "server");
|
|
305
|
+
if (!fs.existsSync(serverDir))
|
|
306
|
+
return ev;
|
|
307
|
+
function scanDir(dir) {
|
|
308
|
+
for (const entry of fs.readdirSync(dir, { withFileTypes: true })) {
|
|
309
|
+
const full = path.join(dir, entry.name);
|
|
310
|
+
if (entry.name.startsWith(".") || entry.name === "node_modules")
|
|
311
|
+
continue;
|
|
312
|
+
if (entry.isDirectory()) {
|
|
313
|
+
scanDir(full);
|
|
314
|
+
continue;
|
|
315
|
+
}
|
|
316
|
+
if (!entry.name.endsWith(".ts"))
|
|
317
|
+
continue;
|
|
318
|
+
ev.files.add(path.relative(projectPath, full));
|
|
319
|
+
const content = fs.readFileSync(full, "utf-8");
|
|
320
|
+
const importMatches = content.match(/import\s+.*from\s+['"]([^'"]+)['"]/g) || [];
|
|
321
|
+
for (const m of importMatches) {
|
|
322
|
+
const pkg = m.match(/from\s+['"]([^'"]+)['"]/)?.[1] || "";
|
|
323
|
+
if (pkg && !pkg.startsWith("."))
|
|
324
|
+
ev.imports.add(pkg);
|
|
325
|
+
}
|
|
326
|
+
const patternChecks = [
|
|
327
|
+
["auth_middleware", /authenticate|authRequired|requireAuth|verifyToken|auth\.utils/i],
|
|
328
|
+
["bcrypt_usage", /bcrypt|passwordHash|hashPassword|argon2|compareSync/i],
|
|
329
|
+
["zod_validation", /z\.(string|number|object|enum|array)\(|\.parse\(|\.safeParse\(/i],
|
|
330
|
+
["audit_logging", /audit|log.*activity|recordSession|pointsLog|notification.*log/i],
|
|
331
|
+
["rate_limiting", /rate.?limit|throttle|express-rate-limit|verificationAttempts/i],
|
|
332
|
+
["verification_code", /verifyCode|verificationCode|verifyPhone|sendVerificationCode/i],
|
|
333
|
+
["db_transaction", /db\.transaction|\.transaction\(/i],
|
|
334
|
+
["session_management", /session|createSession|refreshSession|revokeSession/i],
|
|
335
|
+
["provenance_tracking", /provenance|fingerprint|ledger|@progmune/i],
|
|
336
|
+
["mfa_pattern", /mfa|2fa|two.factor|verification/i],
|
|
337
|
+
];
|
|
338
|
+
for (const [pattern, regex] of patternChecks) {
|
|
339
|
+
if (regex.test(content)) {
|
|
340
|
+
if (!ev.patterns.has(pattern))
|
|
341
|
+
ev.patterns.set(pattern, []);
|
|
342
|
+
ev.patterns.get(pattern).push(path.relative(projectPath, full));
|
|
343
|
+
}
|
|
344
|
+
}
|
|
345
|
+
}
|
|
346
|
+
}
|
|
347
|
+
scanDir(serverDir);
|
|
348
|
+
const pkgPath = path.join(projectPath, "package.json");
|
|
349
|
+
if (fs.existsSync(pkgPath)) {
|
|
350
|
+
const pkg = JSON.parse(fs.readFileSync(pkgPath, "utf-8"));
|
|
351
|
+
for (const dep of Object.keys({ ...pkg.dependencies, ...pkg.devDependencies })) {
|
|
352
|
+
ev.dependencies.add(dep);
|
|
353
|
+
}
|
|
354
|
+
}
|
|
355
|
+
return ev;
|
|
356
|
+
}
|
|
357
|
+
// ── Requirement Checking ──
|
|
358
|
+
function checkRequirement(req, evidence, _businessClaims) {
|
|
359
|
+
const categoryEvidence = {
|
|
360
|
+
"access_control": ["auth_middleware", "session_management"],
|
|
361
|
+
"secrets": ["bcrypt_usage"],
|
|
362
|
+
"audit": ["audit_logging", "provenance_tracking"],
|
|
363
|
+
"data_protection": ["zod_validation"],
|
|
364
|
+
"monitoring": ["rate_limiting"],
|
|
365
|
+
"ai_governance": ["provenance_tracking", "session_management"],
|
|
366
|
+
"change_management": ["db_transaction", "audit_logging"],
|
|
367
|
+
};
|
|
368
|
+
const required = categoryEvidence[req.category] || [];
|
|
369
|
+
const matched = required.filter(p => (evidence.patterns.get(p)?.length || 0) > 0);
|
|
370
|
+
return matched.length > 0;
|
|
371
|
+
}
|
|
372
|
+
// ══════════════════════════════════════════════
|
|
373
|
+
// MAIN
|
|
374
|
+
// ══════════════════════════════════════════════
|
|
375
|
+
function buildComplianceLayer(projectPath) {
|
|
376
|
+
console.log("🔬 Progmune Compliance Knowledge Layer — V11");
|
|
377
|
+
console.log(" Project:", projectPath);
|
|
378
|
+
const kbPath = path.join(projectPath, ".progmune_knowledge.json");
|
|
379
|
+
const businessClaims = fs.existsSync(kbPath)
|
|
380
|
+
? JSON.parse(fs.readFileSync(kbPath, "utf-8")).claims || []
|
|
381
|
+
: [];
|
|
382
|
+
console.log(` Loaded ${businessClaims.length} business claims from V9`);
|
|
383
|
+
console.log(` Compliance library: ${COMPLIANCE_REQUIREMENTS.length} requirements`);
|
|
384
|
+
console.log(` Standards: ${[...new Set(COMPLIANCE_REQUIREMENTS.map(r => r.standard))].join(", ")}`);
|
|
385
|
+
// ── Coverage Analysis ──
|
|
386
|
+
console.log("\n── Compliance Coverage by Standard ──");
|
|
387
|
+
const coverage = analyzeCoverage(businessClaims, COMPLIANCE_REQUIREMENTS, projectPath);
|
|
388
|
+
for (const cov of coverage) {
|
|
389
|
+
const pct = Math.round((cov.satisfiedByBusiness / cov.totalRequirements) * 100);
|
|
390
|
+
const bar = "█".repeat(Math.round(pct / 10)) + "░".repeat(10 - Math.round(pct / 10));
|
|
391
|
+
console.log(`\n ${cov.standard}: ${bar} ${pct}%`);
|
|
392
|
+
console.log(` ${cov.satisfiedByBusiness}/${cov.totalRequirements} requirements satisfied by existing business claims`);
|
|
393
|
+
if (cov.satisfiedByBusiness > 0) {
|
|
394
|
+
console.log(` Satisfied by:`);
|
|
395
|
+
const reqs = COMPLIANCE_REQUIREMENTS.filter(r => cov.claims.includes(r.id));
|
|
396
|
+
for (const req of reqs.slice(0, 3)) {
|
|
397
|
+
console.log(` ✅ ${req.id}: ${req.description}`);
|
|
398
|
+
console.log(` → ${req.invariant}`);
|
|
399
|
+
}
|
|
400
|
+
}
|
|
401
|
+
if (cov.unsatisfied > 0) {
|
|
402
|
+
console.log(` Not yet satisfied:`);
|
|
403
|
+
const reqs = COMPLIANCE_REQUIREMENTS.filter(r => !cov.claims.includes(r.id));
|
|
404
|
+
for (const req of reqs.slice(0, 2)) {
|
|
405
|
+
console.log(` ❌ ${req.id}: ${req.description}`);
|
|
406
|
+
console.log(` → ${req.invariant} [需要: ${req.evidenceRequired}]`);
|
|
407
|
+
}
|
|
408
|
+
}
|
|
409
|
+
}
|
|
410
|
+
// ── Summary ──
|
|
411
|
+
const totalSatisfied = coverage.reduce((s, c) => s + c.satisfiedByBusiness, 0);
|
|
412
|
+
const totalReqs = COMPLIANCE_REQUIREMENTS.length;
|
|
413
|
+
const overallCoverage = Math.round((totalSatisfied / totalReqs) * 100);
|
|
414
|
+
console.log("\n═══ Compliance Knowledge Summary ═══");
|
|
415
|
+
console.log(` Business claims: ${businessClaims.length}`);
|
|
416
|
+
console.log(` Compliance claims: ${totalReqs}`);
|
|
417
|
+
console.log(` Satisfied by business: ${totalSatisfied}/${totalReqs} (${overallCoverage}%)`);
|
|
418
|
+
console.log(` Standards: ${coverage.length}`);
|
|
419
|
+
console.log();
|
|
420
|
+
const report = {
|
|
421
|
+
timestamp: new Date().toISOString(),
|
|
422
|
+
businessClaims,
|
|
423
|
+
complianceClaims: COMPLIANCE_REQUIREMENTS,
|
|
424
|
+
coverage,
|
|
425
|
+
summary: {
|
|
426
|
+
totalBusinessClaims: businessClaims.length,
|
|
427
|
+
totalComplianceClaims: totalReqs,
|
|
428
|
+
overallCoverage,
|
|
429
|
+
standardsCovered: coverage.map(c => c.standard),
|
|
430
|
+
},
|
|
431
|
+
};
|
|
432
|
+
return report;
|
|
433
|
+
}
|
|
434
|
+
// ══════════════════════════════════════════════
|
|
435
|
+
// CLI
|
|
436
|
+
// ══════════════════════════════════════════════
|
|
437
|
+
if (require.main === module) {
|
|
438
|
+
const targetProject = process.argv[2];
|
|
439
|
+
if (!targetProject) {
|
|
440
|
+
console.error("Usage: npx ts-node src/compliance-miner.ts <project-path>");
|
|
441
|
+
process.exit(1);
|
|
442
|
+
}
|
|
443
|
+
const report = buildComplianceLayer(targetProject);
|
|
444
|
+
const outputPath = path.join(targetProject, ".progmune_compliance.json");
|
|
445
|
+
fs.writeFileSync(outputPath, JSON.stringify(report, null, 2));
|
|
446
|
+
console.log(`✅ Compliance report saved to: ${outputPath}`);
|
|
447
|
+
}
|
|
@@ -0,0 +1,194 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* P5.4: Continuous Benchmark Expansion
|
|
4
|
+
*
|
|
5
|
+
* Auto-generates benchmark cases from auto-approved patches
|
|
6
|
+
* and skill library entries, feeding the coverage flywheel.
|
|
7
|
+
*
|
|
8
|
+
* Loop: Skills/Patches → Benchmark Generation → Run Suite
|
|
9
|
+
* → Updated Baseline → Coverage Improvement
|
|
10
|
+
*
|
|
11
|
+
* This ensures the system never stops measuring itself,
|
|
12
|
+
* even as it autonomously extends its own knowledge.
|
|
13
|
+
*/
|
|
14
|
+
var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
|
|
15
|
+
if (k2 === undefined) k2 = k;
|
|
16
|
+
var desc = Object.getOwnPropertyDescriptor(m, k);
|
|
17
|
+
if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
|
|
18
|
+
desc = { enumerable: true, get: function() { return m[k]; } };
|
|
19
|
+
}
|
|
20
|
+
Object.defineProperty(o, k2, desc);
|
|
21
|
+
}) : (function(o, m, k, k2) {
|
|
22
|
+
if (k2 === undefined) k2 = k;
|
|
23
|
+
o[k2] = m[k];
|
|
24
|
+
}));
|
|
25
|
+
var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
|
|
26
|
+
Object.defineProperty(o, "default", { enumerable: true, value: v });
|
|
27
|
+
}) : function(o, v) {
|
|
28
|
+
o["default"] = v;
|
|
29
|
+
});
|
|
30
|
+
var __importStar = (this && this.__importStar) || (function () {
|
|
31
|
+
var ownKeys = function(o) {
|
|
32
|
+
ownKeys = Object.getOwnPropertyNames || function (o) {
|
|
33
|
+
var ar = [];
|
|
34
|
+
for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
|
|
35
|
+
return ar;
|
|
36
|
+
};
|
|
37
|
+
return ownKeys(o);
|
|
38
|
+
};
|
|
39
|
+
return function (mod) {
|
|
40
|
+
if (mod && mod.__esModule) return mod;
|
|
41
|
+
var result = {};
|
|
42
|
+
if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
|
|
43
|
+
__setModuleDefault(result, mod);
|
|
44
|
+
return result;
|
|
45
|
+
};
|
|
46
|
+
})();
|
|
47
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
48
|
+
exports.generateBenchmarksFromPatches = generateBenchmarksFromPatches;
|
|
49
|
+
exports.generateBenchmarksFromSkills = generateBenchmarksFromSkills;
|
|
50
|
+
exports.runContinuousBenchmark = runContinuousBenchmark;
|
|
51
|
+
exports.printContinuousBenchmarkReport = printContinuousBenchmarkReport;
|
|
52
|
+
const fs = __importStar(require("fs"));
|
|
53
|
+
const path = __importStar(require("path"));
|
|
54
|
+
const benchmark_harness_1 = require("./benchmark-harness");
|
|
55
|
+
/**
|
|
56
|
+
* Generate benchmark cases from approved knowledge patches.
|
|
57
|
+
* Each approved patch that adds a virtual rule gets a test case.
|
|
58
|
+
*/
|
|
59
|
+
function generateBenchmarksFromPatches(patchStore) {
|
|
60
|
+
const cases = [];
|
|
61
|
+
for (const patch of patchStore.approved) {
|
|
62
|
+
const [fnA, fnB] = patch.change.split(" → ");
|
|
63
|
+
// Case 1: missing the bridge (broken = just fnA, expected = fnA → fnB)
|
|
64
|
+
cases.push({
|
|
65
|
+
goal: `cover patch: ${patch.change}`,
|
|
66
|
+
protocol: "_global",
|
|
67
|
+
broken: [fnA],
|
|
68
|
+
expected: fnB ? [fnA, fnB] : [fnA],
|
|
69
|
+
violationType: "missing_prerequisite",
|
|
70
|
+
source: "patch",
|
|
71
|
+
sourceId: patch.id,
|
|
72
|
+
});
|
|
73
|
+
// Case 2: resource cleanup variant if applicable
|
|
74
|
+
if (patch.toState === "∅") {
|
|
75
|
+
cases.push({
|
|
76
|
+
goal: `cover cleanup: ${patch.change}`,
|
|
77
|
+
protocol: "_global",
|
|
78
|
+
broken: [fnA, fnB].filter(Boolean),
|
|
79
|
+
expected: [fnA, fnB].filter(Boolean),
|
|
80
|
+
violationType: "resource_leak",
|
|
81
|
+
source: "patch",
|
|
82
|
+
sourceId: patch.id,
|
|
83
|
+
});
|
|
84
|
+
}
|
|
85
|
+
}
|
|
86
|
+
return cases;
|
|
87
|
+
}
|
|
88
|
+
/**
|
|
89
|
+
* Generate benchmark cases from skill library entries.
|
|
90
|
+
* Each skill with high confidence gets a test case.
|
|
91
|
+
*/
|
|
92
|
+
function generateBenchmarksFromSkills(library) {
|
|
93
|
+
const cases = [];
|
|
94
|
+
for (const skill of library.all()) {
|
|
95
|
+
// Only generate for skills with strong evidence
|
|
96
|
+
if (skill.frequency < 5 || skill.successRate < 0.8)
|
|
97
|
+
continue;
|
|
98
|
+
// Case 1: full skill as expected repair
|
|
99
|
+
cases.push({
|
|
100
|
+
goal: `cover skill: ${skill.goal}`,
|
|
101
|
+
protocol: "_global",
|
|
102
|
+
broken: skill.macro.slice(0, Math.max(1, skill.macro.length - 1)), // remove last action
|
|
103
|
+
expected: skill.macro,
|
|
104
|
+
violationType: skill.effects.length > 0 && skill.preconditions.length === 0
|
|
105
|
+
? "missing_prerequisite"
|
|
106
|
+
: "resource_leak",
|
|
107
|
+
source: "skill",
|
|
108
|
+
sourceId: skill.id,
|
|
109
|
+
});
|
|
110
|
+
// Case 2: resource cleanup (just the last cleanup action)
|
|
111
|
+
if (skill.macro.length >= 2) {
|
|
112
|
+
const cleanupAction = skill.macro[skill.macro.length - 1];
|
|
113
|
+
cases.push({
|
|
114
|
+
goal: `cover cleanup: ${cleanupAction}`,
|
|
115
|
+
protocol: "_global",
|
|
116
|
+
broken: skill.macro.slice(0, skill.macro.length - 1),
|
|
117
|
+
expected: skill.macro,
|
|
118
|
+
violationType: "resource_leak",
|
|
119
|
+
source: "skill",
|
|
120
|
+
sourceId: skill.id,
|
|
121
|
+
});
|
|
122
|
+
}
|
|
123
|
+
}
|
|
124
|
+
return cases;
|
|
125
|
+
}
|
|
126
|
+
/**
|
|
127
|
+
* Run continuous benchmark expansion: generate → write → measure.
|
|
128
|
+
*
|
|
129
|
+
* 1. Generate new cases from patches + skills
|
|
130
|
+
* 2. Write to benchmarks/expanded/ directory
|
|
131
|
+
* 3. Run full benchmark suite
|
|
132
|
+
* 4. Report updated baseline
|
|
133
|
+
*/
|
|
134
|
+
async function runContinuousBenchmark(patchStore, library, outputDir) {
|
|
135
|
+
// 1. Generate
|
|
136
|
+
const patchCases = generateBenchmarksFromPatches(patchStore);
|
|
137
|
+
const skillCases = generateBenchmarksFromSkills(library);
|
|
138
|
+
const allCases = [...patchCases, ...skillCases];
|
|
139
|
+
// 2. Write
|
|
140
|
+
const outDir = outputDir || path.resolve(__dirname, "..", "benchmarks", "expanded");
|
|
141
|
+
if (!fs.existsSync(outDir))
|
|
142
|
+
fs.mkdirSync(outDir, { recursive: true });
|
|
143
|
+
const timestamp = new Date().toISOString().slice(0, 10);
|
|
144
|
+
const filepath = path.join(outDir, `auto_generated_${timestamp}.json`);
|
|
145
|
+
fs.writeFileSync(filepath, JSON.stringify({
|
|
146
|
+
generatedAt: new Date().toISOString(),
|
|
147
|
+
source: "continuous-benchmark",
|
|
148
|
+
skillsCount: library.size,
|
|
149
|
+
patchesCount: patchStore.approved.length,
|
|
150
|
+
cases: allCases,
|
|
151
|
+
}, null, 2));
|
|
152
|
+
// 3. Run benchmark
|
|
153
|
+
let benchmark;
|
|
154
|
+
try {
|
|
155
|
+
benchmark = await (0, benchmark_harness_1.runBenchmark)(path.resolve(__dirname, "..", "benchmarks"));
|
|
156
|
+
}
|
|
157
|
+
catch {
|
|
158
|
+
// benchmark run can fail if no cases; gracefully skip
|
|
159
|
+
}
|
|
160
|
+
// 4. Report
|
|
161
|
+
return {
|
|
162
|
+
timestamp: new Date().toISOString(),
|
|
163
|
+
existingCases: benchmark?.cases || 0,
|
|
164
|
+
generatedCases: allCases.length,
|
|
165
|
+
sourceBreakdown: {
|
|
166
|
+
skills: skillCases.length,
|
|
167
|
+
patches: patchCases.length,
|
|
168
|
+
},
|
|
169
|
+
writtenFiles: [filepath],
|
|
170
|
+
benchmark,
|
|
171
|
+
summary: allCases.length > 0
|
|
172
|
+
? `Generated ${allCases.length} new benchmark cases (${skillCases.length} from skills, ${patchCases.length} from patches). Total suite: ${benchmark?.cases || 0} cases.`
|
|
173
|
+
: "No new cases generated. Skills and patches may need more evidence.",
|
|
174
|
+
};
|
|
175
|
+
}
|
|
176
|
+
function printContinuousBenchmarkReport(report) {
|
|
177
|
+
console.log("\n╔════════════════════════════════════════════════════╗");
|
|
178
|
+
console.log("║ P5.4 Continuous Benchmark Expansion ║");
|
|
179
|
+
console.log("╚════════════════════════════════════════════════════╝\n");
|
|
180
|
+
console.log(`Timestamp: ${report.timestamp}`);
|
|
181
|
+
console.log(`Generated Cases: ${report.generatedCases}`);
|
|
182
|
+
console.log(` From Skills: ${report.sourceBreakdown.skills}`);
|
|
183
|
+
console.log(` From Patches: ${report.sourceBreakdown.patches}`);
|
|
184
|
+
console.log(`Total Suite: ${report.existingCases}`);
|
|
185
|
+
console.log();
|
|
186
|
+
console.log(`Summary: ${report.summary}`);
|
|
187
|
+
console.log();
|
|
188
|
+
if (report.benchmark) {
|
|
189
|
+
console.log("─── Updated Benchmark Baseline ───");
|
|
190
|
+
console.log(` Top-1: ${(report.benchmark.top1Rate * 100).toFixed(0)}%`);
|
|
191
|
+
console.log(` Top-3: ${(report.benchmark.top3Rate * 100).toFixed(0)}%`);
|
|
192
|
+
console.log();
|
|
193
|
+
}
|
|
194
|
+
}
|