@vk.amogh/trace 2.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +208 -0
- package/bin/trace.js +112 -0
- package/package.json +46 -0
- package/pyproject.toml +39 -0
- package/src/trace_engine/__init__.py +8 -0
- package/src/trace_engine/__pycache__/__init__.cpython-311.pyc +0 -0
- package/src/trace_engine/__pycache__/cli.cpython-311.pyc +0 -0
- package/src/trace_engine/__pycache__/doctor.cpython-311.pyc +0 -0
- package/src/trace_engine/__pycache__/interactive.cpython-311.pyc +0 -0
- package/src/trace_engine/__pycache__/verify.cpython-311.pyc +0 -0
- package/src/trace_engine/ai/__init__.py +7 -0
- package/src/trace_engine/ai/__pycache__/__init__.cpython-311.pyc +0 -0
- package/src/trace_engine/ai/__pycache__/base.cpython-311.pyc +0 -0
- package/src/trace_engine/ai/__pycache__/ollama.cpython-311.pyc +0 -0
- package/src/trace_engine/ai/__pycache__/planner.cpython-311.pyc +0 -0
- package/src/trace_engine/ai/base.py +23 -0
- package/src/trace_engine/ai/ollama.py +50 -0
- package/src/trace_engine/ai/planner.py +40 -0
- package/src/trace_engine/apm/__init__.py +18 -0
- package/src/trace_engine/apm/__pycache__/__init__.cpython-311.pyc +0 -0
- package/src/trace_engine/apm/__pycache__/builder.cpython-311.pyc +0 -0
- package/src/trace_engine/apm/__pycache__/edges.cpython-311.pyc +0 -0
- package/src/trace_engine/apm/__pycache__/model.cpython-311.pyc +0 -0
- package/src/trace_engine/apm/__pycache__/nodes.cpython-311.pyc +0 -0
- package/src/trace_engine/apm/__pycache__/serialization.cpython-311.pyc +0 -0
- package/src/trace_engine/apm/builder.py +208 -0
- package/src/trace_engine/apm/edges.py +25 -0
- package/src/trace_engine/apm/model.py +107 -0
- package/src/trace_engine/apm/nodes.py +27 -0
- package/src/trace_engine/apm/serialization.py +105 -0
- package/src/trace_engine/benchmark/__init__.py +5 -0
- package/src/trace_engine/benchmark/__pycache__/__init__.cpython-311.pyc +0 -0
- package/src/trace_engine/benchmark/__pycache__/owasp.cpython-311.pyc +0 -0
- package/src/trace_engine/benchmark/owasp.py +183 -0
- package/src/trace_engine/cli.py +1184 -0
- package/src/trace_engine/config/__init__.py +35 -0
- package/src/trace_engine/config/__pycache__/__init__.cpython-311.pyc +0 -0
- package/src/trace_engine/config/__pycache__/defaults.cpython-311.pyc +0 -0
- package/src/trace_engine/config/__pycache__/loader.cpython-311.pyc +0 -0
- package/src/trace_engine/config/__pycache__/settings.cpython-311.pyc +0 -0
- package/src/trace_engine/config/defaults.py +48 -0
- package/src/trace_engine/config/loader.py +64 -0
- package/src/trace_engine/config/settings.py +72 -0
- package/src/trace_engine/doctor.py +250 -0
- package/src/trace_engine/findings/__init__.py +15 -0
- package/src/trace_engine/findings/__pycache__/__init__.cpython-311.pyc +0 -0
- package/src/trace_engine/findings/__pycache__/correlate.cpython-311.pyc +0 -0
- package/src/trace_engine/findings/__pycache__/model.cpython-311.pyc +0 -0
- package/src/trace_engine/findings/__pycache__/recommendations.cpython-311.pyc +0 -0
- package/src/trace_engine/findings/__pycache__/store.cpython-311.pyc +0 -0
- package/src/trace_engine/findings/correlate.py +103 -0
- package/src/trace_engine/findings/model.py +40 -0
- package/src/trace_engine/findings/recommendations.py +35 -0
- package/src/trace_engine/findings/store.py +39 -0
- package/src/trace_engine/framework/__init__.py +58 -0
- package/src/trace_engine/framework/__pycache__/__init__.cpython-311.pyc +0 -0
- package/src/trace_engine/framework/__pycache__/base.cpython-311.pyc +0 -0
- package/src/trace_engine/framework/__pycache__/csharp.cpython-311.pyc +0 -0
- package/src/trace_engine/framework/__pycache__/dart.cpython-311.pyc +0 -0
- package/src/trace_engine/framework/__pycache__/django.cpython-311.pyc +0 -0
- package/src/trace_engine/framework/__pycache__/express.cpython-311.pyc +0 -0
- package/src/trace_engine/framework/__pycache__/fastapi.cpython-311.pyc +0 -0
- package/src/trace_engine/framework/__pycache__/flask.cpython-311.pyc +0 -0
- package/src/trace_engine/framework/__pycache__/go.cpython-311.pyc +0 -0
- package/src/trace_engine/framework/__pycache__/nextjs.cpython-311.pyc +0 -0
- package/src/trace_engine/framework/__pycache__/php.cpython-311.pyc +0 -0
- package/src/trace_engine/framework/__pycache__/react_router.cpython-311.pyc +0 -0
- package/src/trace_engine/framework/__pycache__/ruby.cpython-311.pyc +0 -0
- package/src/trace_engine/framework/__pycache__/rust.cpython-311.pyc +0 -0
- package/src/trace_engine/framework/__pycache__/springboot.cpython-311.pyc +0 -0
- package/src/trace_engine/framework/base.py +49 -0
- package/src/trace_engine/framework/csharp.py +111 -0
- package/src/trace_engine/framework/dart.py +152 -0
- package/src/trace_engine/framework/django.py +188 -0
- package/src/trace_engine/framework/express.py +82 -0
- package/src/trace_engine/framework/fastapi.py +135 -0
- package/src/trace_engine/framework/flask.py +108 -0
- package/src/trace_engine/framework/go.py +94 -0
- package/src/trace_engine/framework/nextjs.py +200 -0
- package/src/trace_engine/framework/php.py +98 -0
- package/src/trace_engine/framework/react_router.py +331 -0
- package/src/trace_engine/framework/ruby.py +69 -0
- package/src/trace_engine/framework/rust.py +101 -0
- package/src/trace_engine/framework/springboot.py +145 -0
- package/src/trace_engine/harness/__init__.py +19 -0
- package/src/trace_engine/harness/__pycache__/__init__.cpython-311.pyc +0 -0
- package/src/trace_engine/harness/__pycache__/benchmark.cpython-311.pyc +0 -0
- package/src/trace_engine/harness/__pycache__/context.cpython-311.pyc +0 -0
- package/src/trace_engine/harness/__pycache__/engine.cpython-311.pyc +0 -0
- package/src/trace_engine/harness/__pycache__/patcher.cpython-311.pyc +0 -0
- package/src/trace_engine/harness/__pycache__/remediators.cpython-311.pyc +0 -0
- package/src/trace_engine/harness/benchmark.py +55 -0
- package/src/trace_engine/harness/context.py +49 -0
- package/src/trace_engine/harness/engine.py +187 -0
- package/src/trace_engine/harness/patcher.py +143 -0
- package/src/trace_engine/harness/remediators.py +307 -0
- package/src/trace_engine/ingest/__init__.py +15 -0
- package/src/trace_engine/ingest/__pycache__/__init__.cpython-311.pyc +0 -0
- package/src/trace_engine/ingest/__pycache__/files.cpython-311.pyc +0 -0
- package/src/trace_engine/ingest/__pycache__/hashing.cpython-311.pyc +0 -0
- package/src/trace_engine/ingest/__pycache__/ignore.cpython-311.pyc +0 -0
- package/src/trace_engine/ingest/__pycache__/repository.cpython-311.pyc +0 -0
- package/src/trace_engine/ingest/files.py +60 -0
- package/src/trace_engine/ingest/hashing.py +21 -0
- package/src/trace_engine/ingest/ignore.py +108 -0
- package/src/trace_engine/ingest/repository.py +50 -0
- package/src/trace_engine/intelligence/__init__.py +15 -0
- package/src/trace_engine/intelligence/__pycache__/__init__.cpython-311.pyc +0 -0
- package/src/trace_engine/intelligence/__pycache__/evaluation.cpython-311.pyc +0 -0
- package/src/trace_engine/intelligence/__pycache__/orchestrator.cpython-311.pyc +0 -0
- package/src/trace_engine/intelligence/__pycache__/tracebench.cpython-311.pyc +0 -0
- package/src/trace_engine/intelligence/evaluation.py +828 -0
- package/src/trace_engine/intelligence/laya/__init__.py +19 -0
- package/src/trace_engine/intelligence/laya/__pycache__/__init__.cpython-311.pyc +0 -0
- package/src/trace_engine/intelligence/laya/__pycache__/prompts.cpython-311.pyc +0 -0
- package/src/trace_engine/intelligence/laya/__pycache__/router.cpython-311.pyc +0 -0
- package/src/trace_engine/intelligence/laya/__pycache__/schemas.cpython-311.pyc +0 -0
- package/src/trace_engine/intelligence/laya/__pycache__/telemetry.cpython-311.pyc +0 -0
- package/src/trace_engine/intelligence/laya/__pycache__/thresholds.cpython-311.pyc +0 -0
- package/src/trace_engine/intelligence/laya/prompts.py +67 -0
- package/src/trace_engine/intelligence/laya/router.py +352 -0
- package/src/trace_engine/intelligence/laya/schemas.py +56 -0
- package/src/trace_engine/intelligence/laya/telemetry.py +48 -0
- package/src/trace_engine/intelligence/laya/thresholds.py +12 -0
- package/src/trace_engine/intelligence/orchestrator.py +130 -0
- package/src/trace_engine/intelligence/securebert/__init__.py +6 -0
- package/src/trace_engine/intelligence/securebert/__pycache__/__init__.cpython-311.pyc +0 -0
- package/src/trace_engine/intelligence/securebert/__pycache__/cache.cpython-311.pyc +0 -0
- package/src/trace_engine/intelligence/securebert/__pycache__/classifier.cpython-311.pyc +0 -0
- package/src/trace_engine/intelligence/securebert/cache.py +37 -0
- package/src/trace_engine/intelligence/securebert/classifier.py +240 -0
- package/src/trace_engine/intelligence/tracebench.py +61 -0
- package/src/trace_engine/intelligence/training/__init__.py +21 -0
- package/src/trace_engine/intelligence/training/__pycache__/__init__.cpython-311.pyc +0 -0
- package/src/trace_engine/intelligence/training/__pycache__/dataset.cpython-311.pyc +0 -0
- package/src/trace_engine/intelligence/training/__pycache__/dataset_importers.cpython-311.pyc +0 -0
- package/src/trace_engine/intelligence/training/__pycache__/laya_trainer.cpython-311.pyc +0 -0
- package/src/trace_engine/intelligence/training/__pycache__/lora_system2.cpython-311.pyc +0 -0
- package/src/trace_engine/intelligence/training/__pycache__/morefixes_pipeline.cpython-311.pyc +0 -0
- package/src/trace_engine/intelligence/training/__pycache__/slicer.cpython-311.pyc +0 -0
- package/src/trace_engine/intelligence/training/__pycache__/train_all.cpython-311.pyc +0 -0
- package/src/trace_engine/intelligence/training/__pycache__/trainer.cpython-311.pyc +0 -0
- package/src/trace_engine/intelligence/training/dataset.py +320 -0
- package/src/trace_engine/intelligence/training/dataset_importers.py +349 -0
- package/src/trace_engine/intelligence/training/laya_trainer.py +678 -0
- package/src/trace_engine/intelligence/training/lora_system2.py +162 -0
- package/src/trace_engine/intelligence/training/morefixes_pipeline.py +463 -0
- package/src/trace_engine/intelligence/training/slicer.py +129 -0
- package/src/trace_engine/intelligence/training/train_all.py +1009 -0
- package/src/trace_engine/intelligence/training/trainer.py +321 -0
- package/src/trace_engine/interactive.py +623 -0
- package/src/trace_engine/mcp/__init__.py +6 -0
- package/src/trace_engine/mcp/__pycache__/__init__.cpython-311.pyc +0 -0
- package/src/trace_engine/mcp/__pycache__/config.cpython-311.pyc +0 -0
- package/src/trace_engine/mcp/__pycache__/server.cpython-311.pyc +0 -0
- package/src/trace_engine/mcp/config.py +42 -0
- package/src/trace_engine/mcp/server.py +398 -0
- package/src/trace_engine/output/__init__.py +28 -0
- package/src/trace_engine/output/__pycache__/__init__.cpython-311.pyc +0 -0
- package/src/trace_engine/output/__pycache__/html.cpython-311.pyc +0 -0
- package/src/trace_engine/output/__pycache__/markdown.cpython-311.pyc +0 -0
- package/src/trace_engine/output/__pycache__/sarif.cpython-311.pyc +0 -0
- package/src/trace_engine/output/__pycache__/terminal.cpython-311.pyc +0 -0
- package/src/trace_engine/output/html.py +142 -0
- package/src/trace_engine/output/markdown.py +65 -0
- package/src/trace_engine/output/sarif.py +180 -0
- package/src/trace_engine/output/terminal.py +537 -0
- package/src/trace_engine/parsing/__init__.py +26 -0
- package/src/trace_engine/parsing/__pycache__/__init__.cpython-311.pyc +0 -0
- package/src/trace_engine/parsing/__pycache__/calls.cpython-311.pyc +0 -0
- package/src/trace_engine/parsing/__pycache__/language.cpython-311.pyc +0 -0
- package/src/trace_engine/parsing/__pycache__/locations.cpython-311.pyc +0 -0
- package/src/trace_engine/parsing/__pycache__/parser.cpython-311.pyc +0 -0
- package/src/trace_engine/parsing/__pycache__/symbols.cpython-311.pyc +0 -0
- package/src/trace_engine/parsing/calls.py +14 -0
- package/src/trace_engine/parsing/language.py +24 -0
- package/src/trace_engine/parsing/locations.py +17 -0
- package/src/trace_engine/parsing/parser.py +465 -0
- package/src/trace_engine/parsing/symbols.py +45 -0
- package/src/trace_engine/plugin/__init__.py +157 -0
- package/src/trace_engine/plugin/__pycache__/__init__.cpython-311.pyc +0 -0
- package/src/trace_engine/plugin/__pycache__/evaluator.cpython-311.pyc +0 -0
- package/src/trace_engine/plugin/__pycache__/swebench_adapter.cpython-311.pyc +0 -0
- package/src/trace_engine/plugin/__pycache__/task.cpython-311.pyc +0 -0
- package/src/trace_engine/plugin/bundle/hooks.json +24 -0
- package/src/trace_engine/plugin/bundle/mcp_config.json +11 -0
- package/src/trace_engine/plugin/bundle/plugin.json +20 -0
- package/src/trace_engine/plugin/bundle/rules/security_remediation.md +40 -0
- package/src/trace_engine/plugin/bundle/skills/trace-security-harness/SKILL.md +118 -0
- package/src/trace_engine/plugin/evaluator.py +105 -0
- package/src/trace_engine/plugin/swebench_adapter.py +96 -0
- package/src/trace_engine/plugin/task.py +48 -0
- package/src/trace_engine/policy/__init__.py +5 -0
- package/src/trace_engine/policy/__pycache__/__init__.cpython-311.pyc +0 -0
- package/src/trace_engine/policy/__pycache__/scope.cpython-311.pyc +0 -0
- package/src/trace_engine/policy/scope.py +80 -0
- package/src/trace_engine/runtime/__init__.py +7 -0
- package/src/trace_engine/runtime/__pycache__/__init__.cpython-311.pyc +0 -0
- package/src/trace_engine/runtime/__pycache__/client.cpython-311.pyc +0 -0
- package/src/trace_engine/runtime/__pycache__/observations.cpython-311.pyc +0 -0
- package/src/trace_engine/runtime/__pycache__/target.cpython-311.pyc +0 -0
- package/src/trace_engine/runtime/client.py +81 -0
- package/src/trace_engine/runtime/observations.py +18 -0
- package/src/trace_engine/runtime/target.py +23 -0
- package/src/trace_engine/security/__init__.py +16 -0
- package/src/trace_engine/security/__pycache__/__init__.cpython-311.pyc +0 -0
- package/src/trace_engine/security/__pycache__/fusion.cpython-311.pyc +0 -0
- package/src/trace_engine/security/__pycache__/hypotheses.cpython-311.pyc +0 -0
- package/src/trace_engine/security/__pycache__/signals.cpython-311.pyc +0 -0
- package/src/trace_engine/security/__pycache__/timing.cpython-311.pyc +0 -0
- package/src/trace_engine/security/fusion.py +126 -0
- package/src/trace_engine/security/hypotheses.py +251 -0
- package/src/trace_engine/security/signals.py +28 -0
- package/src/trace_engine/security/timing.py +132 -0
- package/src/trace_engine/testpacks/__init__.py +18 -0
- package/src/trace_engine/testpacks/__pycache__/__init__.cpython-311.pyc +0 -0
- package/src/trace_engine/testpacks/__pycache__/authentication.cpython-311.pyc +0 -0
- package/src/trace_engine/testpacks/__pycache__/base.cpython-311.pyc +0 -0
- package/src/trace_engine/testpacks/__pycache__/bfla.cpython-311.pyc +0 -0
- package/src/trace_engine/testpacks/__pycache__/bola.cpython-311.pyc +0 -0
- package/src/trace_engine/testpacks/__pycache__/cors.cpython-311.pyc +0 -0
- package/src/trace_engine/testpacks/__pycache__/deserialization.cpython-311.pyc +0 -0
- package/src/trace_engine/testpacks/__pycache__/injection.cpython-311.pyc +0 -0
- package/src/trace_engine/testpacks/__pycache__/mass_assignment.cpython-311.pyc +0 -0
- package/src/trace_engine/testpacks/__pycache__/path_traversal.cpython-311.pyc +0 -0
- package/src/trace_engine/testpacks/__pycache__/registry.cpython-311.pyc +0 -0
- package/src/trace_engine/testpacks/__pycache__/ssrf.cpython-311.pyc +0 -0
- package/src/trace_engine/testpacks/__pycache__/ssti.cpython-311.pyc +0 -0
- package/src/trace_engine/testpacks/authentication.py +60 -0
- package/src/trace_engine/testpacks/base.py +63 -0
- package/src/trace_engine/testpacks/bfla.py +70 -0
- package/src/trace_engine/testpacks/bola.py +94 -0
- package/src/trace_engine/testpacks/cors.py +85 -0
- package/src/trace_engine/testpacks/deserialization.py +86 -0
- package/src/trace_engine/testpacks/injection.py +179 -0
- package/src/trace_engine/testpacks/mass_assignment.py +70 -0
- package/src/trace_engine/testpacks/path_traversal.py +117 -0
- package/src/trace_engine/testpacks/registry.py +44 -0
- package/src/trace_engine/testpacks/ssrf.py +85 -0
- package/src/trace_engine/testpacks/ssti.py +96 -0
- package/src/trace_engine/tools/__init__.py +6 -0
- package/src/trace_engine/tools/__pycache__/__init__.cpython-311.pyc +0 -0
- package/src/trace_engine/tools/__pycache__/adapters.cpython-311.pyc +0 -0
- package/src/trace_engine/tools/__pycache__/base.cpython-311.pyc +0 -0
- package/src/trace_engine/tools/__pycache__/registry.cpython-311.pyc +0 -0
- package/src/trace_engine/tools/adapters.py +111 -0
- package/src/trace_engine/tools/base.py +68 -0
- package/src/trace_engine/tools/registry.py +36 -0
- package/src/trace_engine/verify.py +156 -0
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
"""Intelligence Orchestrator implementing Section 9 model hierarchy.
|
|
2
|
+
|
|
3
|
+
Deterministic rules -> SecureBERT -> Laya System 1 -> Local LLM -> Test Engine
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from typing import List, Dict, Any, Optional
|
|
7
|
+
from pydantic import BaseModel, Field
|
|
8
|
+
|
|
9
|
+
from trace_engine.framework.base import Endpoint
|
|
10
|
+
from trace_engine.apm.model import AttackPathModel
|
|
11
|
+
from trace_engine.security.hypotheses import SecurityHypothesis, VulnerabilityCategory
|
|
12
|
+
from trace_engine.intelligence.securebert.classifier import SecureBERTClassifier
|
|
13
|
+
from trace_engine.intelligence.laya.router import LayaDecisionEngine
|
|
14
|
+
from trace_engine.ai.planner import TestPlanner
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class TestPlanRecommendation(BaseModel):
|
|
18
|
+
endpoint_id: str
|
|
19
|
+
endpoint_display: str
|
|
20
|
+
primary_testpack: str
|
|
21
|
+
priority_level: str
|
|
22
|
+
securebert_scores: Dict[str, float]
|
|
23
|
+
laya_confidence: float
|
|
24
|
+
decision_path: str
|
|
25
|
+
rationale: str
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class IntelligenceOrchestrator:
|
|
29
|
+
"""Orchestrates deterministic signals, SecureBERT embeddings, and Laya decisions."""
|
|
30
|
+
|
|
31
|
+
def __init__(
|
|
32
|
+
self,
|
|
33
|
+
securebert: Optional[SecureBERTClassifier] = None,
|
|
34
|
+
laya_engine: Optional[LayaDecisionEngine] = None,
|
|
35
|
+
llm_planner: Optional[TestPlanner] = None,
|
|
36
|
+
):
|
|
37
|
+
self.securebert = securebert or SecureBERTClassifier()
|
|
38
|
+
self.laya = laya_engine or LayaDecisionEngine()
|
|
39
|
+
self.llm_planner = llm_planner or TestPlanner()
|
|
40
|
+
|
|
41
|
+
def evaluate_endpoint(
|
|
42
|
+
self,
|
|
43
|
+
endpoint: Endpoint,
|
|
44
|
+
code_slice: str,
|
|
45
|
+
apm: AttackPathModel,
|
|
46
|
+
hypotheses: List[SecurityHypothesis],
|
|
47
|
+
) -> TestPlanRecommendation:
|
|
48
|
+
"""Evaluates an endpoint through the complete intelligence stack."""
|
|
49
|
+
# 1. SecureBERT vulnerability family scoring
|
|
50
|
+
bert_scores = self.securebert.classify_slice(code_slice, endpoint)
|
|
51
|
+
|
|
52
|
+
# 2. Laya System 1 decision (priority & test selection)
|
|
53
|
+
priority_dec = self.laya.decide_priority(endpoint, apm)
|
|
54
|
+
test_dec = self.laya.decide_test_selection(endpoint, apm, hypotheses)
|
|
55
|
+
|
|
56
|
+
# 3. Model hierarchy reconciliation
|
|
57
|
+
# If SecureBERT shows high probability for a specific category (e.g. BOLA > 0.4)
|
|
58
|
+
top_category = max(bert_scores, key=bert_scores.get) if bert_scores else "UNKNOWN"
|
|
59
|
+
top_score = bert_scores.get(top_category, 0.0)
|
|
60
|
+
chosen_pack = test_dec.primary_testpack
|
|
61
|
+
|
|
62
|
+
decision_path = "deterministic"
|
|
63
|
+
if self.laya.is_available():
|
|
64
|
+
decision_path = "laya_system1"
|
|
65
|
+
elif top_score > 0.35:
|
|
66
|
+
decision_path = "securebert_guided"
|
|
67
|
+
|
|
68
|
+
rationale = (
|
|
69
|
+
f"Evaluated via {decision_path}. "
|
|
70
|
+
f"Top vulnerability family: {top_category} ({int(top_score * 100)}%). "
|
|
71
|
+
f"Assigned priority: {priority_dec.priority_level}."
|
|
72
|
+
)
|
|
73
|
+
|
|
74
|
+
return TestPlanRecommendation(
|
|
75
|
+
endpoint_id=endpoint.id,
|
|
76
|
+
endpoint_display=endpoint.display_name(),
|
|
77
|
+
primary_testpack=chosen_pack,
|
|
78
|
+
priority_level=priority_dec.priority_level,
|
|
79
|
+
securebert_scores=bert_scores,
|
|
80
|
+
laya_confidence=test_dec.confidence,
|
|
81
|
+
decision_path=decision_path,
|
|
82
|
+
rationale=rationale,
|
|
83
|
+
)
|
|
84
|
+
|
|
85
|
+
def evaluate_endpoints_batch(
|
|
86
|
+
self,
|
|
87
|
+
endpoints: List[Endpoint],
|
|
88
|
+
code_slices: List[str],
|
|
89
|
+
apm: AttackPathModel,
|
|
90
|
+
hypotheses: List[SecurityHypothesis],
|
|
91
|
+
) -> List[TestPlanRecommendation]:
|
|
92
|
+
"""Evaluates a batch of endpoints through SecureBERT and Laya System 1."""
|
|
93
|
+
items = list(zip(code_slices, endpoints))
|
|
94
|
+
bert_scores_list = self.securebert.classify_batch(items)
|
|
95
|
+
|
|
96
|
+
recommendations: List[TestPlanRecommendation] = []
|
|
97
|
+
for ep, code_slice, bert_scores in zip(endpoints, code_slices, bert_scores_list):
|
|
98
|
+
priority_dec = self.laya.decide_priority(ep, apm)
|
|
99
|
+
test_dec = self.laya.decide_test_selection(ep, apm, hypotheses)
|
|
100
|
+
|
|
101
|
+
top_category = max(bert_scores, key=bert_scores.get) if bert_scores else "UNKNOWN"
|
|
102
|
+
top_score = bert_scores.get(top_category, 0.0)
|
|
103
|
+
chosen_pack = test_dec.primary_testpack
|
|
104
|
+
|
|
105
|
+
decision_path = (
|
|
106
|
+
"laya_system1"
|
|
107
|
+
if self.laya.is_available()
|
|
108
|
+
else ("securebert_guided" if top_score > 0.35 else "calibrated_system1")
|
|
109
|
+
)
|
|
110
|
+
rationale = (
|
|
111
|
+
f"Evaluated via {decision_path}. "
|
|
112
|
+
f"Top vulnerability family: {top_category} ({int(top_score * 100)}%). "
|
|
113
|
+
f"Assigned priority: {priority_dec.priority_level}."
|
|
114
|
+
)
|
|
115
|
+
|
|
116
|
+
recommendations.append(
|
|
117
|
+
TestPlanRecommendation(
|
|
118
|
+
endpoint_id=ep.id,
|
|
119
|
+
endpoint_display=ep.display_name(),
|
|
120
|
+
primary_testpack=chosen_pack,
|
|
121
|
+
priority_level=priority_dec.priority_level,
|
|
122
|
+
securebert_scores=bert_scores,
|
|
123
|
+
laya_confidence=test_dec.confidence,
|
|
124
|
+
decision_path=decision_path,
|
|
125
|
+
rationale=rationale,
|
|
126
|
+
)
|
|
127
|
+
)
|
|
128
|
+
|
|
129
|
+
return recommendations
|
|
130
|
+
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
"""SecureBERT 2.0 vulnerability classification and semantic encoding."""
|
|
2
|
+
|
|
3
|
+
from trace_engine.intelligence.securebert.classifier import SecureBERTClassifier, VULN_CATEGORIES
|
|
4
|
+
from trace_engine.intelligence.securebert.cache import SecureBERTCache
|
|
5
|
+
|
|
6
|
+
__all__ = ["SecureBERTClassifier", "VULN_CATEGORIES", "SecureBERTCache"]
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
"""On-disk cache for SecureBERT embeddings and classification scores."""
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import Dict, Any, Optional
|
|
6
|
+
from trace_engine.ingest.hashing import hash_content
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class SecureBERTCache:
|
|
10
|
+
"""Caches SecureBERT inferences by hash of input code/attack-path text."""
|
|
11
|
+
|
|
12
|
+
def __init__(self, cache_file: Path):
|
|
13
|
+
self.cache_file = cache_file
|
|
14
|
+
self._cache: Dict[str, Dict[str, float]] = {}
|
|
15
|
+
self._load()
|
|
16
|
+
|
|
17
|
+
def _load(self) -> None:
|
|
18
|
+
if self.cache_file.exists():
|
|
19
|
+
try:
|
|
20
|
+
with open(self.cache_file, "r", encoding="utf-8") as f:
|
|
21
|
+
self._cache = json.load(f)
|
|
22
|
+
except Exception:
|
|
23
|
+
self._cache = {}
|
|
24
|
+
|
|
25
|
+
def get(self, text: str) -> Optional[Dict[str, float]]:
|
|
26
|
+
key = hash_content(text)
|
|
27
|
+
return self._cache.get(key)
|
|
28
|
+
|
|
29
|
+
def set(self, text: str, scores: Dict[str, float]) -> None:
|
|
30
|
+
key = hash_content(text)
|
|
31
|
+
self._cache[key] = scores
|
|
32
|
+
try:
|
|
33
|
+
self.cache_file.parent.mkdir(parents=True, exist_ok=True)
|
|
34
|
+
with open(self.cache_file, "w", encoding="utf-8") as f:
|
|
35
|
+
json.dump(self._cache, f, indent=2)
|
|
36
|
+
except Exception:
|
|
37
|
+
pass
|
|
@@ -0,0 +1,240 @@
|
|
|
1
|
+
"""SecureBERT 2.0 vulnerability classifier and semantic encoder."""
|
|
2
|
+
|
|
3
|
+
import logging
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import Dict, Any, Optional, List, Tuple
|
|
6
|
+
import math
|
|
7
|
+
|
|
8
|
+
from trace_engine.intelligence.securebert.cache import SecureBERTCache
|
|
9
|
+
from trace_engine.framework.base import Endpoint
|
|
10
|
+
from trace_engine.apm.model import AttackPathModel
|
|
11
|
+
|
|
12
|
+
logger = logging.getLogger(__name__)
|
|
13
|
+
|
|
14
|
+
VULN_CATEGORIES = [
|
|
15
|
+
"BOLA",
|
|
16
|
+
"BFLA",
|
|
17
|
+
"AUTHENTICATION",
|
|
18
|
+
"SSRF",
|
|
19
|
+
"INJECTION",
|
|
20
|
+
"MASS_ASSIGNMENT",
|
|
21
|
+
"PATH_TRAVERSAL",
|
|
22
|
+
"SSTI",
|
|
23
|
+
"CORS",
|
|
24
|
+
"DESERIALIZATION",
|
|
25
|
+
]
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class SecureBERTClassifier:
|
|
29
|
+
"""Classifies source code attack-path slices into vulnerability families using SecureBERT."""
|
|
30
|
+
|
|
31
|
+
def __init__(
|
|
32
|
+
self,
|
|
33
|
+
model_name: Optional[str] = None,
|
|
34
|
+
cache_path: Optional[Path] = None,
|
|
35
|
+
):
|
|
36
|
+
candidates = [
|
|
37
|
+
Path(".trace/models/securebert-finetuned"),
|
|
38
|
+
Path(__file__).resolve().parents[4] / ".trace/models/securebert-finetuned",
|
|
39
|
+
Path(__file__).resolve().parents[2] / "models/securebert-finetuned",
|
|
40
|
+
Path.home() / ".trace/models/securebert-finetuned",
|
|
41
|
+
]
|
|
42
|
+
default_model = "ehsanaghaei/SecureBERT"
|
|
43
|
+
for cand in candidates:
|
|
44
|
+
weight_file = cand / "model.safetensors"
|
|
45
|
+
if weight_file.exists():
|
|
46
|
+
if weight_file.stat().st_size < 10000:
|
|
47
|
+
try:
|
|
48
|
+
import subprocess
|
|
49
|
+
subprocess.run(["git", "lfs", "pull"], cwd=repo_root, capture_output=True, timeout=60)
|
|
50
|
+
except Exception:
|
|
51
|
+
pass
|
|
52
|
+
if weight_file.exists() and weight_file.stat().st_size >= 10000:
|
|
53
|
+
default_model = str(cand.resolve())
|
|
54
|
+
break
|
|
55
|
+
|
|
56
|
+
self.model_name = model_name or default_model
|
|
57
|
+
cache_file = cache_path or Path(".trace/cache/securebert_cache.json")
|
|
58
|
+
self.cache = SecureBERTCache(cache_file)
|
|
59
|
+
self._tokenizer = None
|
|
60
|
+
self._model = None
|
|
61
|
+
self._device = "cpu"
|
|
62
|
+
self._initialized = False
|
|
63
|
+
|
|
64
|
+
def _init_model(self) -> None:
|
|
65
|
+
"""Lazy initialization of transformers model."""
|
|
66
|
+
if self._initialized:
|
|
67
|
+
return
|
|
68
|
+
self._initialized = True
|
|
69
|
+
try:
|
|
70
|
+
import torch
|
|
71
|
+
from transformers import AutoTokenizer, AutoModelForSequenceClassification
|
|
72
|
+
|
|
73
|
+
self._device = "cuda" if torch.cuda.is_available() else "cpu"
|
|
74
|
+
logger.info(f"Loading SecureBERT ({self.model_name}) on {self._device}...")
|
|
75
|
+
self._tokenizer = AutoTokenizer.from_pretrained(self.model_name)
|
|
76
|
+
self._model = AutoModelForSequenceClassification.from_pretrained(
|
|
77
|
+
self.model_name,
|
|
78
|
+
num_labels=len(VULN_CATEGORIES),
|
|
79
|
+
ignore_mismatched_sizes=True,
|
|
80
|
+
).to(self._device)
|
|
81
|
+
self._model.eval()
|
|
82
|
+
logger.info(f"SecureBERT loaded successfully on {self._device}.")
|
|
83
|
+
except Exception as e:
|
|
84
|
+
logger.debug(f"SecureBERT initialization deferred: {e}. Semantic encoder active.")
|
|
85
|
+
self._tokenizer = None
|
|
86
|
+
self._model = None
|
|
87
|
+
|
|
88
|
+
def classify_batch(self, items: List[Tuple[str, Endpoint]]) -> List[Dict[str, float]]:
|
|
89
|
+
"""Batch classification of multiple code slices using SecureBERT."""
|
|
90
|
+
results: List[Optional[Dict[str, float]]] = [None] * len(items)
|
|
91
|
+
uncached_indices: List[int] = []
|
|
92
|
+
uncached_texts: List[str] = []
|
|
93
|
+
|
|
94
|
+
for idx, (code_text, ep) in enumerate(items):
|
|
95
|
+
cached = self.cache.get(code_text)
|
|
96
|
+
if cached:
|
|
97
|
+
results[idx] = cached
|
|
98
|
+
else:
|
|
99
|
+
uncached_indices.append(idx)
|
|
100
|
+
uncached_texts.append(code_text)
|
|
101
|
+
|
|
102
|
+
if uncached_texts:
|
|
103
|
+
self._init_model()
|
|
104
|
+
if self._model and self._tokenizer:
|
|
105
|
+
try:
|
|
106
|
+
import torch
|
|
107
|
+
chunk_size = 64
|
|
108
|
+
for c_start in range(0, len(uncached_texts), chunk_size):
|
|
109
|
+
c_end = c_start + chunk_size
|
|
110
|
+
c_texts = uncached_texts[c_start:c_end]
|
|
111
|
+
c_indices = uncached_indices[c_start:c_end]
|
|
112
|
+
|
|
113
|
+
inputs = self._tokenizer(
|
|
114
|
+
c_texts,
|
|
115
|
+
max_length=256,
|
|
116
|
+
truncation=True,
|
|
117
|
+
padding=True,
|
|
118
|
+
return_tensors="pt",
|
|
119
|
+
).to(self._device)
|
|
120
|
+
|
|
121
|
+
with torch.no_grad():
|
|
122
|
+
outputs = self._model(**inputs)
|
|
123
|
+
raw_probs = torch.sigmoid(outputs.logits).detach().cpu().tolist()
|
|
124
|
+
|
|
125
|
+
probs_list = [raw_probs] if len(c_texts) == 1 and isinstance(raw_probs[0], (int, float)) else raw_probs
|
|
126
|
+
|
|
127
|
+
for sub_i, orig_idx in enumerate(c_indices):
|
|
128
|
+
if sub_i < len(probs_list) and isinstance(probs_list[sub_i], list):
|
|
129
|
+
score_dict = {
|
|
130
|
+
cat: round(probs_list[sub_i][p_idx], 4)
|
|
131
|
+
for p_idx, cat in enumerate(VULN_CATEGORIES)
|
|
132
|
+
}
|
|
133
|
+
results[orig_idx] = score_dict
|
|
134
|
+
self.cache.set(items[orig_idx][0], score_dict)
|
|
135
|
+
except Exception as e:
|
|
136
|
+
logger.debug(f"SecureBERT batch inference error: {e}")
|
|
137
|
+
|
|
138
|
+
for orig_idx in uncached_indices:
|
|
139
|
+
if results[orig_idx] is None:
|
|
140
|
+
code_text, ep = items[orig_idx]
|
|
141
|
+
scores = self._semantic_scoring(code_text, ep)
|
|
142
|
+
results[orig_idx] = scores
|
|
143
|
+
self.cache.set(code_text, scores)
|
|
144
|
+
|
|
145
|
+
return [r for r in results if r is not None]
|
|
146
|
+
|
|
147
|
+
def classify_slice(self, code_text: str, endpoint: Endpoint) -> Dict[str, float]:
|
|
148
|
+
"""Classify a code slice and return probability distribution over vulnerability families."""
|
|
149
|
+
cached = self.cache.get(code_text)
|
|
150
|
+
if cached:
|
|
151
|
+
return cached
|
|
152
|
+
|
|
153
|
+
scores: Dict[str, float] = {}
|
|
154
|
+
|
|
155
|
+
# 1. Neural classification if model loaded
|
|
156
|
+
self._init_model()
|
|
157
|
+
if self._model and self._tokenizer:
|
|
158
|
+
try:
|
|
159
|
+
import torch
|
|
160
|
+
|
|
161
|
+
inputs = self._tokenizer(
|
|
162
|
+
code_text,
|
|
163
|
+
max_length=512,
|
|
164
|
+
truncation=True,
|
|
165
|
+
padding="max_length",
|
|
166
|
+
return_tensors="pt",
|
|
167
|
+
).to(self._device)
|
|
168
|
+
|
|
169
|
+
with torch.no_grad():
|
|
170
|
+
outputs = self._model(**inputs)
|
|
171
|
+
probs = torch.sigmoid(outputs.logits).squeeze().tolist()
|
|
172
|
+
|
|
173
|
+
if isinstance(probs, list) and len(probs) == len(VULN_CATEGORIES):
|
|
174
|
+
scores = {cat: round(probs[idx], 4) for idx, cat in enumerate(VULN_CATEGORIES)}
|
|
175
|
+
except Exception as e:
|
|
176
|
+
logger.debug(f"SecureBERT forward pass error: {e}")
|
|
177
|
+
|
|
178
|
+
# 2. Semantic feature calculation
|
|
179
|
+
if not scores:
|
|
180
|
+
scores = self._semantic_scoring(code_text, endpoint)
|
|
181
|
+
|
|
182
|
+
self.cache.set(code_text, scores)
|
|
183
|
+
return scores
|
|
184
|
+
|
|
185
|
+
def _semantic_scoring(self, code_text: str, endpoint: Endpoint) -> Dict[str, float]:
|
|
186
|
+
"""Calculates semantic cybersecurity relevance scores from code tokens and endpoint properties."""
|
|
187
|
+
text_lower = code_text.lower()
|
|
188
|
+
|
|
189
|
+
# Prior base distribution
|
|
190
|
+
scores = {cat: 0.05 for cat in VULN_CATEGORIES}
|
|
191
|
+
|
|
192
|
+
# BOLA signals
|
|
193
|
+
if endpoint.object_identifier:
|
|
194
|
+
scores["BOLA"] += 0.40
|
|
195
|
+
if "user_id" in text_lower or "owner" in text_lower or "orders.get" in text_lower:
|
|
196
|
+
scores["BOLA"] += 0.35
|
|
197
|
+
|
|
198
|
+
# BFLA signals
|
|
199
|
+
if "admin" in endpoint.path.lower() or "admin" in endpoint.roles:
|
|
200
|
+
scores["BFLA"] += 0.50
|
|
201
|
+
if "role" in text_lower or "permission" in text_lower or "refund" in text_lower:
|
|
202
|
+
scores["BFLA"] += 0.25
|
|
203
|
+
|
|
204
|
+
# AUTH signals
|
|
205
|
+
if not endpoint.auth_required and (endpoint.state_changing or endpoint.sensitive_data):
|
|
206
|
+
scores["AUTHENTICATION"] += 0.60
|
|
207
|
+
if "token" in text_lower or "login" in text_lower:
|
|
208
|
+
scores["AUTHENTICATION"] += 0.20
|
|
209
|
+
|
|
210
|
+
# SSRF signals
|
|
211
|
+
if endpoint.external_network or "fetch" in text_lower or "httpx" in text_lower or "url" in text_lower:
|
|
212
|
+
scores["SSRF"] += 0.70
|
|
213
|
+
|
|
214
|
+
# Injection signals
|
|
215
|
+
if "query" in text_lower or "syntax" in text_lower or "sql" in text_lower or "search" in text_lower:
|
|
216
|
+
scores["INJECTION"] += 0.55
|
|
217
|
+
|
|
218
|
+
# Mass assignment signals
|
|
219
|
+
if endpoint.method in ("PATCH", "PUT") and ("user" in text_lower or "update" in text_lower or "payload" in text_lower):
|
|
220
|
+
scores["MASS_ASSIGNMENT"] += 0.60
|
|
221
|
+
|
|
222
|
+
# Path traversal signals
|
|
223
|
+
if "file" in text_lower or "path" in text_lower or "open(" in text_lower or "readfile" in text_lower:
|
|
224
|
+
scores["PATH_TRAVERSAL"] += 0.65
|
|
225
|
+
|
|
226
|
+
# SSTI signals
|
|
227
|
+
if "template" in text_lower or "render" in text_lower or "{{" in text_lower or "${" in text_lower:
|
|
228
|
+
scores["SSTI"] += 0.60
|
|
229
|
+
|
|
230
|
+
# CORS signals
|
|
231
|
+
if "origin" in text_lower or "access-control" in text_lower or "cors" in text_lower:
|
|
232
|
+
scores["CORS"] += 0.55
|
|
233
|
+
|
|
234
|
+
# Deserialization signals
|
|
235
|
+
if "pickle" in text_lower or "yaml" in text_lower or "readobject" in text_lower or "unserialize" in text_lower:
|
|
236
|
+
scores["DESERIALIZATION"] += 0.65
|
|
237
|
+
|
|
238
|
+
# Normalize to probability distribution
|
|
239
|
+
total = sum(scores.values())
|
|
240
|
+
return {k: round(v / total, 4) for k, v in scores.items()}
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
"""TRACEBench dataset generation and recording for training/benchmarking."""
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import time
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from typing import List, Dict, Any, Optional
|
|
7
|
+
from pydantic import BaseModel, Field
|
|
8
|
+
|
|
9
|
+
from trace_engine.findings.model import Finding
|
|
10
|
+
from trace_engine.apm.model import AttackPathModel
|
|
11
|
+
from trace_engine.testpacks.base import TestExecutionResult
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class TRACEBenchRecord(BaseModel):
|
|
15
|
+
"""Ground truth record matching Section 19 of TRACE Specification."""
|
|
16
|
+
timestamp: float = Field(default_factory=time.time)
|
|
17
|
+
repository: str
|
|
18
|
+
endpoint: str
|
|
19
|
+
attack_path: List[str] = Field(default_factory=list)
|
|
20
|
+
signals: List[str] = Field(default_factory=list)
|
|
21
|
+
candidate_tests: List[str] = Field(default_factory=list)
|
|
22
|
+
executed_tests: List[Dict[str, Any]] = Field(default_factory=list)
|
|
23
|
+
ground_truth: str # "confirmed", "safe", "inconclusive"
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class TRACEBenchRecorder:
|
|
27
|
+
"""Records real security analysis and runtime test trajectories into TRACEBench."""
|
|
28
|
+
|
|
29
|
+
def __init__(self, output_dir: Optional[Path] = None):
|
|
30
|
+
self.output_dir = output_dir or Path("training/datasets")
|
|
31
|
+
self.output_file = self.output_dir / "tracebench.jsonl"
|
|
32
|
+
|
|
33
|
+
def record_run(
|
|
34
|
+
self,
|
|
35
|
+
repository_name: str,
|
|
36
|
+
finding: Finding,
|
|
37
|
+
apm: AttackPathModel,
|
|
38
|
+
) -> TRACEBenchRecord:
|
|
39
|
+
"""Saves a confirmed or evaluated finding as a TRACEBench dataset record."""
|
|
40
|
+
self.output_dir.mkdir(parents=True, exist_ok=True)
|
|
41
|
+
|
|
42
|
+
record = TRACEBenchRecord(
|
|
43
|
+
repository=repository_name,
|
|
44
|
+
endpoint=finding.endpoint,
|
|
45
|
+
attack_path=finding.attack_path,
|
|
46
|
+
signals=finding.static_evidence,
|
|
47
|
+
candidate_tests=[finding.category.lower()],
|
|
48
|
+
executed_tests=[
|
|
49
|
+
{
|
|
50
|
+
"test": finding.category.lower(),
|
|
51
|
+
"summary": finding.correlation_notes,
|
|
52
|
+
"status": "confirmed" if finding.confidence.value == "CONFIRMED" else "potential",
|
|
53
|
+
}
|
|
54
|
+
],
|
|
55
|
+
ground_truth="confirmed" if finding.confidence.value == "CONFIRMED" else "safe",
|
|
56
|
+
)
|
|
57
|
+
|
|
58
|
+
with open(self.output_file, "a", encoding="utf-8") as f:
|
|
59
|
+
f.write(json.dumps(record.model_dump()) + "\n")
|
|
60
|
+
|
|
61
|
+
return record
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
"""TRACE Neural Fine-Tuning and Training Engine with PyTorch CUDA Acceleration."""
|
|
2
|
+
|
|
3
|
+
from trace_engine.intelligence.training.dataset import VulnerabilityDataset, generate_cybersecurity_training_corpus
|
|
4
|
+
from trace_engine.intelligence.training.trainer import SecureBERTTrainer, TrainingConfig
|
|
5
|
+
from trace_engine.intelligence.training.slicer import DataflowSliceExtractor, slice_and_canonicalize
|
|
6
|
+
from trace_engine.intelligence.training.lora_system2 import System2LoRATrainer, LoRATrainingConfig
|
|
7
|
+
from trace_engine.intelligence.training.dataset_importers import JulietImporter, BigVulImporter, CVEfixesImporter
|
|
8
|
+
|
|
9
|
+
__all__ = [
|
|
10
|
+
"VulnerabilityDataset",
|
|
11
|
+
"generate_cybersecurity_training_corpus",
|
|
12
|
+
"SecureBERTTrainer",
|
|
13
|
+
"TrainingConfig",
|
|
14
|
+
"DataflowSliceExtractor",
|
|
15
|
+
"slice_and_canonicalize",
|
|
16
|
+
"System2LoRATrainer",
|
|
17
|
+
"LoRATrainingConfig",
|
|
18
|
+
"JulietImporter",
|
|
19
|
+
"BigVulImporter",
|
|
20
|
+
"CVEfixesImporter",
|
|
21
|
+
]
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|