praxis-sec 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +170 -0
- package/ai-defense/cost-protection.md +292 -0
- package/ai-defense/llm-security-checklist.md +324 -0
- package/ai-defense/prompt-injection-patterns.js +283 -0
- package/ai-defense/system-prompt-armor.md +327 -0
- package/checklists/launch-day.md +168 -0
- package/cli/agents/abom-generator.js +225 -0
- package/cli/agents/agent-attestation-agent.js +318 -0
- package/cli/agents/agent-config-scanner.js +787 -0
- package/cli/agents/agent-telemetry-agent.js +415 -0
- package/cli/agents/agentic-security-agent.js +296 -0
- package/cli/agents/agentic-supply-chain-agent.js +463 -0
- package/cli/agents/ai-infra-inventory-agent.js +449 -0
- package/cli/agents/api-fuzzer.js +345 -0
- package/cli/agents/auth-bypass-agent.js +348 -0
- package/cli/agents/base-agent.js +280 -0
- package/cli/agents/cicd-scanner.js +300 -0
- package/cli/agents/config-auditor.js +757 -0
- package/cli/agents/deep-analyzer.js +776 -0
- package/cli/agents/endpoint-agent-abuse-agent.js +404 -0
- package/cli/agents/exception-handler-agent.js +187 -0
- package/cli/agents/git-history-scanner.js +169 -0
- package/cli/agents/governance-audits.js +138 -0
- package/cli/agents/hermes-security-agent.js +536 -0
- package/cli/agents/html-reporter.js +1125 -0
- package/cli/agents/index.js +147 -0
- package/cli/agents/injection-tester.js +502 -0
- package/cli/agents/legal-risk-agent.js +328 -0
- package/cli/agents/llm-redteam.js +199 -0
- package/cli/agents/managed-agent-scanner.js +333 -0
- package/cli/agents/mcp-security-agent.js +588 -0
- package/cli/agents/memory-poisoning-agent.js +305 -0
- package/cli/agents/mobile-scanner.js +231 -0
- package/cli/agents/model-file-scanner.js +259 -0
- package/cli/agents/orchestrator.js +355 -0
- package/cli/agents/pii-compliance-agent.js +301 -0
- package/cli/agents/policy-engine.js +229 -0
- package/cli/agents/prompt-injection-prober.js +224 -0
- package/cli/agents/rag-security-agent.js +204 -0
- package/cli/agents/recon-agent.js +207 -0
- package/cli/agents/sbom-generator.js +265 -0
- package/cli/agents/scoring-engine.js +273 -0
- package/cli/agents/ssrf-prober.js +130 -0
- package/cli/agents/stateful-watcher.js +238 -0
- package/cli/agents/supabase-rls-agent.js +154 -0
- package/cli/agents/supply-chain-agent.js +857 -0
- package/cli/agents/swarm-orchestrator.js +200 -0
- package/cli/agents/verifier-agent.js +303 -0
- package/cli/agents/vibe-coding-agent.js +250 -0
- package/cli/bin/praxis.js +866 -0
- package/cli/commands/abom.js +73 -0
- package/cli/commands/agent-fix.js +1245 -0
- package/cli/commands/audit.js +1180 -0
- package/cli/commands/autofix.js +383 -0
- package/cli/commands/baseline.js +193 -0
- package/cli/commands/benchmark.js +327 -0
- package/cli/commands/checklist.js +223 -0
- package/cli/commands/ci.js +403 -0
- package/cli/commands/deps.js +516 -0
- package/cli/commands/diff.js +200 -0
- package/cli/commands/doctor.js +195 -0
- package/cli/commands/env-audit.js +349 -0
- package/cli/commands/fix.js +218 -0
- package/cli/commands/guard.js +396 -0
- package/cli/commands/hooks.js +278 -0
- package/cli/commands/init.js +514 -0
- package/cli/commands/legal.js +158 -0
- package/cli/commands/live-advisories.js +241 -0
- package/cli/commands/mcp.js +660 -0
- package/cli/commands/openclaw.js +386 -0
- package/cli/commands/red-team.js +350 -0
- package/cli/commands/redteam.js +78 -0
- package/cli/commands/remediate.js +797 -0
- package/cli/commands/rotate.js +768 -0
- package/cli/commands/rules.js +196 -0
- package/cli/commands/scan-mcp.js +534 -0
- package/cli/commands/scan-skill.js +588 -0
- package/cli/commands/scan-standard.js +251 -0
- package/cli/commands/scan.js +524 -0
- package/cli/commands/score.js +449 -0
- package/cli/commands/shell.js +514 -0
- package/cli/commands/team-report.js +398 -0
- package/cli/commands/undo.js +161 -0
- package/cli/commands/update-intel.js +126 -0
- package/cli/commands/vibe-check.js +276 -0
- package/cli/commands/watch.js +757 -0
- package/cli/commands/web.js +63 -0
- package/cli/core/ast/guardrail-detector.js +141 -0
- package/cli/core/ast/index.js +22 -0
- package/cli/core/ast/parser.js +676 -0
- package/cli/core/ast/scope-tree.js +287 -0
- package/cli/core/ast/taint-tracker.js +158 -0
- package/cli/core/branding.js +37 -0
- package/cli/core/env.js +38 -0
- package/cli/core/errors.js +61 -0
- package/cli/core/fs.js +62 -0
- package/cli/core/output/compliance.js +90 -0
- package/cli/core/output/html-theme.js +158 -0
- package/cli/core/output/index.js +57 -0
- package/cli/core/output/json.js +48 -0
- package/cli/core/output/sarif.js +240 -0
- package/cli/core/version.js +67 -0
- package/cli/core/web/jobs.js +183 -0
- package/cli/core/web/projects.js +146 -0
- package/cli/core/web/server.js +439 -0
- package/cli/data/atlas-knowledge.json +5640 -0
- package/cli/data/eaa-catalog.json +39 -0
- package/cli/data/known-mcps.json +26 -0
- package/cli/data/probes/prompt-injection-corpus.json +271 -0
- package/cli/data/threat-intel.json +85 -0
- package/cli/data/threatpacks/latest.json +41 -0
- package/cli/hooks/patterns.js +313 -0
- package/cli/hooks/post-tool-use.js +140 -0
- package/cli/hooks/pre-tool-use.js +186 -0
- package/cli/index.js +90 -0
- package/cli/providers/llm-provider.js +766 -0
- package/cli/utils/autofix-rules.js +74 -0
- package/cli/utils/cache-manager.js +310 -0
- package/cli/utils/compliance-map.js +191 -0
- package/cli/utils/entropy.js +132 -0
- package/cli/utils/fix-ledger.js +127 -0
- package/cli/utils/hermes-tool-registry.js +252 -0
- package/cli/utils/intel/cache.js +61 -0
- package/cli/utils/intel/http.js +88 -0
- package/cli/utils/intel/index.js +235 -0
- package/cli/utils/intel/merge.js +229 -0
- package/cli/utils/intel/sources/epss.js +54 -0
- package/cli/utils/intel/sources/ghsa.js +81 -0
- package/cli/utils/intel/sources/gitguardian.js +40 -0
- package/cli/utils/intel/sources/gitleaks.js +101 -0
- package/cli/utils/intel/sources/kev.js +38 -0
- package/cli/utils/intel/sources/nvd.js +84 -0
- package/cli/utils/intel/sources/osv.js +132 -0
- package/cli/utils/intel/sources/phylum.js +44 -0
- package/cli/utils/intel/sources/snyk.js +46 -0
- package/cli/utils/intel/sources/socket.js +69 -0
- package/cli/utils/intel/sources/sonatype.js +84 -0
- package/cli/utils/intel/sources/threatpack.js +69 -0
- package/cli/utils/mcp-trust.js +60 -0
- package/cli/utils/output.js +251 -0
- package/cli/utils/patterns.js +1130 -0
- package/cli/utils/pdf-generator.js +94 -0
- package/cli/utils/plugin-loader.js +364 -0
- package/cli/utils/rule-import.js +228 -0
- package/cli/utils/rule-registry.js +426 -0
- package/cli/utils/scan-fingerprint.js +109 -0
- package/cli/utils/scan-playbook.js +312 -0
- package/cli/utils/score-history.js +119 -0
- package/cli/utils/secrets-verifier.js +247 -0
- package/cli/utils/security-memory.js +296 -0
- package/cli/utils/standards/atlas-knowledge.js +87 -0
- package/cli/utils/standards/index.js +127 -0
- package/cli/utils/standards/sources/avid.js +45 -0
- package/cli/utils/standards/sources/eu-ai-act.js +89 -0
- package/cli/utils/standards/sources/google-saif.js +39 -0
- package/cli/utils/standards/sources/iso-42001.js +94 -0
- package/cli/utils/standards/sources/mitre-atlas.js +54 -0
- package/cli/utils/standards/sources/nist-ai-600-1.js +45 -0
- package/cli/utils/standards/sources/owasp-llm.js +45 -0
- package/cli/utils/standards/sources/owasp-ml.js +45 -0
- package/cli/utils/threat-intel.js +265 -0
- package/configs/firebase/firestore-rules.txt +215 -0
- package/configs/firebase/security-checklist.md +236 -0
- package/configs/firebase/storage-rules.txt +206 -0
- package/configs/gitignore-template +258 -0
- package/configs/nextjs-security-headers.js +220 -0
- package/configs/praxisignore-template +50 -0
- package/configs/supabase/secure-client.ts +225 -0
- package/configs/supabase/security-checklist.md +278 -0
- package/docs/THIRD_PARTY_NOTICES.md +26 -0
- package/docs/THREAT_INTEL.md +292 -0
- package/docs/USAGE.md +1205 -0
- package/docs/design/WEB-UI.md +82 -0
- package/package.json +71 -0
- package/scripts/check-determinism.mjs +119 -0
- package/snippets/README.md +122 -0
- package/snippets/api-security/api-security-checklist.md +412 -0
- package/snippets/api-security/cors-config.ts +322 -0
- package/snippets/api-security/input-validation.ts +430 -0
- package/snippets/auth/jwt-checklist.md +322 -0
- package/snippets/rate-limiting/nextjs-middleware.ts +211 -0
- package/snippets/rate-limiting/upstash-ratelimit.ts +229 -0
|
@@ -0,0 +1,776 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* DeepAnalyzer — Multi-Tier LLM-Powered Taint Analysis
|
|
3
|
+
* ======================================================
|
|
4
|
+
*
|
|
5
|
+
* Hermes-inspired three-tier analysis pipeline:
|
|
6
|
+
*
|
|
7
|
+
* Tier 1 (Haiku / cheap model) — Fast triage of all critical+high findings.
|
|
8
|
+
* Labels each finding: "skip" | "review" | "escalate".
|
|
9
|
+
* Skips obvious false-positives early and cheaply.
|
|
10
|
+
*
|
|
11
|
+
* Tier 2 (Sonnet / mid model) — Deep taint analysis of "review" findings.
|
|
12
|
+
* Full file context, sanitization checking, exploitability rating.
|
|
13
|
+
*
|
|
14
|
+
* Tier 3 (Opus / frontier model) — Full exploit-chain reasoning for "escalate"
|
|
15
|
+
* findings (confirmed critical severity with untrusted input path).
|
|
16
|
+
* Returns attack vector, business impact, and exact fix.
|
|
17
|
+
*
|
|
18
|
+
* When only one provider/model is configured (non-Anthropic), the pipeline falls
|
|
19
|
+
* back gracefully to a single-tier analysis identical to the previous behavior.
|
|
20
|
+
*
|
|
21
|
+
* Structured output:
|
|
22
|
+
* When the provider is AnthropicProvider, all LLM calls use the tool-use API
|
|
23
|
+
* (tool_choice: forced) which guarantees JSON matching the schema — no regex
|
|
24
|
+
* cleanup, no silent dropped findings.
|
|
25
|
+
*
|
|
26
|
+
* Supports:
|
|
27
|
+
* - Anthropic API (ANTHROPIC_API_KEY) — full multi-tier + structured output
|
|
28
|
+
* - OpenAI API (OPENAI_API_KEY) — single-tier, text parsing
|
|
29
|
+
* - Google Gemini (GOOGLE_API_KEY) — single-tier, text parsing
|
|
30
|
+
* - Ollama / Gemma4 (--local) — large context, schema-enforced output
|
|
31
|
+
*
|
|
32
|
+
* USAGE:
|
|
33
|
+
* const analyzer = new DeepAnalyzer({ provider, budgetCents: 50 });
|
|
34
|
+
* const enrichedFindings = await analyzer.analyze(findings, context);
|
|
35
|
+
*/
|
|
36
|
+
|
|
37
|
+
import fs from 'fs';
|
|
38
|
+
import path from 'path';
|
|
39
|
+
import { createProvider, autoDetectProvider } from '../providers/llm-provider.js';
|
|
40
|
+
import { ASTParser, ScopeTree, TaintTracker, GuardrailDetector } from '../core/ast/index.js';
|
|
41
|
+
|
|
42
|
+
// Lazy-import ScanPlaybook to avoid circular dep; only used when rootPath is known
|
|
43
|
+
let _ScanPlaybook = null;
|
|
44
|
+
async function getScanPlaybook() {
|
|
45
|
+
if (!_ScanPlaybook) {
|
|
46
|
+
const mod = await import('../utils/scan-playbook.js');
|
|
47
|
+
_ScanPlaybook = mod.ScanPlaybook;
|
|
48
|
+
}
|
|
49
|
+
return _ScanPlaybook;
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
// =============================================================================
|
|
53
|
+
// CONSTANTS
|
|
54
|
+
// =============================================================================
|
|
55
|
+
|
|
56
|
+
/** Max file content per finding for standard providers */
|
|
57
|
+
const MAX_FILE_CHARS_DEFAULT = 4000;
|
|
58
|
+
|
|
59
|
+
/** Max file content per finding for large-context providers (Gemma 4 128K–256K) */
|
|
60
|
+
const MAX_FILE_CHARS_LARGE_CTX = 80000;
|
|
61
|
+
|
|
62
|
+
/** Max findings to analyze per run (cost control) */
|
|
63
|
+
const MAX_FINDINGS = 30;
|
|
64
|
+
|
|
65
|
+
// Approximate cost per 1K tokens (Haiku pricing used as baseline)
|
|
66
|
+
const COST_PER_1K_INPUT = 0.08; // cents
|
|
67
|
+
const COST_PER_1K_OUTPUT = 0.4; // cents
|
|
68
|
+
|
|
69
|
+
const EST_INPUT_TOKENS_PER_FINDING = 1500;
|
|
70
|
+
const EST_OUTPUT_TOKENS_PER_FINDING = 300;
|
|
71
|
+
|
|
72
|
+
// Multi-tier Anthropic model IDs
|
|
73
|
+
const TIER1_MODEL = 'claude-haiku-4-5-20251001'; // fast triage
|
|
74
|
+
const TIER2_MODEL = 'claude-sonnet-4-6'; // deep analysis
|
|
75
|
+
const TIER3_MODEL = 'claude-opus-4-6'; // exploit chain
|
|
76
|
+
|
|
77
|
+
// =============================================================================
|
|
78
|
+
// JSON SCHEMAS — used with Anthropic tool-use for guaranteed output
|
|
79
|
+
// =============================================================================
|
|
80
|
+
|
|
81
|
+
/** Tier 1: quick triage schema */
|
|
82
|
+
const TRIAGE_SCHEMA = {
|
|
83
|
+
type: 'object',
|
|
84
|
+
properties: {
|
|
85
|
+
results: {
|
|
86
|
+
type: 'array',
|
|
87
|
+
items: {
|
|
88
|
+
type: 'object',
|
|
89
|
+
properties: {
|
|
90
|
+
findingId: { type: 'string' },
|
|
91
|
+
tier: { type: 'string', enum: ['skip', 'review', 'escalate'] },
|
|
92
|
+
reason: { type: 'string' },
|
|
93
|
+
},
|
|
94
|
+
required: ['findingId', 'tier', 'reason'],
|
|
95
|
+
additionalProperties: false,
|
|
96
|
+
},
|
|
97
|
+
},
|
|
98
|
+
},
|
|
99
|
+
required: ['results'],
|
|
100
|
+
};
|
|
101
|
+
|
|
102
|
+
/** Tier 2: deep analysis schema */
|
|
103
|
+
const DEEP_ANALYSIS_SCHEMA = {
|
|
104
|
+
type: 'object',
|
|
105
|
+
properties: {
|
|
106
|
+
results: {
|
|
107
|
+
type: 'array',
|
|
108
|
+
items: {
|
|
109
|
+
type: 'object',
|
|
110
|
+
properties: {
|
|
111
|
+
findingId: { type: 'string' },
|
|
112
|
+
tainted: { type: 'boolean' },
|
|
113
|
+
sanitized: { type: 'boolean' },
|
|
114
|
+
exploitability: { type: 'string', enum: ['confirmed', 'likely', 'unlikely', 'false_positive'] },
|
|
115
|
+
reasoning: { type: 'string' },
|
|
116
|
+
},
|
|
117
|
+
required: ['findingId', 'tainted', 'sanitized', 'exploitability', 'reasoning'],
|
|
118
|
+
additionalProperties: false,
|
|
119
|
+
},
|
|
120
|
+
},
|
|
121
|
+
},
|
|
122
|
+
required: ['results'],
|
|
123
|
+
};
|
|
124
|
+
|
|
125
|
+
/** Tier 3: exploit-chain schema */
|
|
126
|
+
const EXPLOIT_SCHEMA = {
|
|
127
|
+
type: 'object',
|
|
128
|
+
properties: {
|
|
129
|
+
results: {
|
|
130
|
+
type: 'array',
|
|
131
|
+
items: {
|
|
132
|
+
type: 'object',
|
|
133
|
+
properties: {
|
|
134
|
+
findingId: { type: 'string' },
|
|
135
|
+
tainted: { type: 'boolean' },
|
|
136
|
+
sanitized: { type: 'boolean' },
|
|
137
|
+
exploitability: { type: 'string', enum: ['confirmed', 'likely', 'unlikely', 'false_positive'] },
|
|
138
|
+
reasoning: { type: 'string' },
|
|
139
|
+
attackVector: { type: 'string' },
|
|
140
|
+
businessImpact: { type: 'string' },
|
|
141
|
+
fix: { type: 'string' },
|
|
142
|
+
},
|
|
143
|
+
required: ['findingId', 'tainted', 'sanitized', 'exploitability', 'reasoning', 'attackVector', 'businessImpact', 'fix'],
|
|
144
|
+
additionalProperties: false,
|
|
145
|
+
},
|
|
146
|
+
},
|
|
147
|
+
},
|
|
148
|
+
required: ['results'],
|
|
149
|
+
};
|
|
150
|
+
|
|
151
|
+
// =============================================================================
|
|
152
|
+
// SYSTEM PROMPTS
|
|
153
|
+
// =============================================================================
|
|
154
|
+
|
|
155
|
+
const TRIAGE_SYSTEM = `You are a fast security triage assistant. For each finding, quickly decide:
|
|
156
|
+
- "skip" — Obvious false positive (hardcoded literal, test file, sanitized value, documentation).
|
|
157
|
+
- "review" — Needs deeper analysis. Possibly tainted user input reaching a dangerous sink.
|
|
158
|
+
- "escalate" — Clear, unsanitized path from user-controlled input to a critical sink (SQL query, shell exec, file write, deserialization). Escalate only when confident.
|
|
159
|
+
|
|
160
|
+
Be conservative: prefer "review" over "escalate" when unsure.`;
|
|
161
|
+
|
|
162
|
+
const DEEP_SYSTEM = `You are a security code auditor performing taint analysis. For each finding, determine:
|
|
163
|
+
|
|
164
|
+
1. Tainted: Is the flagged value controllable by an external user (HTTP request, file upload, CLI args, env vars, DB read)?
|
|
165
|
+
2. Sanitized: Is there sanitization, validation, or encoding between source and sink that neutralizes the risk?
|
|
166
|
+
3. Exploitability: "confirmed" | "likely" | "unlikely" | "false_positive"
|
|
167
|
+
4. Reasoning: One concise sentence explaining your verdict.
|
|
168
|
+
|
|
169
|
+
Rules:
|
|
170
|
+
- Hardcoded string literals with no user input path → NOT tainted.
|
|
171
|
+
- Validation library (zod, joi, yup, ajv) or sanitize function between input and sink → sanitized=true.
|
|
172
|
+
- Test/example/documentation file → false_positive.
|
|
173
|
+
- Cannot determine taint flow from provided context → "unlikely".
|
|
174
|
+
- Only "confirmed" when there is a clear, unsanitized path from user input to dangerous sink.`;
|
|
175
|
+
|
|
176
|
+
const EXPLOIT_SYSTEM = `You are an expert security researcher performing full exploit-chain analysis. For each confirmed critical finding:
|
|
177
|
+
|
|
178
|
+
1. Trace the complete attack vector from attacker-controlled input to dangerous sink.
|
|
179
|
+
2. Assess the real-world business impact (data breach, account takeover, RCE, etc.).
|
|
180
|
+
3. Write a precise, actionable fix (code change, library call, or config update).
|
|
181
|
+
4. Rate exploitability as "confirmed" only if the path is fully unsanitized; otherwise "likely".
|
|
182
|
+
|
|
183
|
+
Be specific. Code references, line numbers, and exact fix suggestions are expected.`;
|
|
184
|
+
|
|
185
|
+
// Fallback system prompt for non-tiered (single provider) analysis
|
|
186
|
+
const SINGLE_TIER_SYSTEM = `You are a security code auditor performing taint analysis. For each finding, determine:
|
|
187
|
+
|
|
188
|
+
1. **Tainted**: Is the flagged value controllable by an external user (via HTTP request, file upload, CLI args, env vars, database read, etc.)?
|
|
189
|
+
2. **Sanitized**: Is there sanitization, validation, or encoding between the source and sink that neutralizes the risk?
|
|
190
|
+
3. **Exploitability**: Rate as "confirmed", "likely", "unlikely", or "false_positive".
|
|
191
|
+
4. **Reasoning**: One sentence explaining your verdict.
|
|
192
|
+
|
|
193
|
+
Respond with a JSON array ONLY. No markdown, no explanation outside JSON.
|
|
194
|
+
|
|
195
|
+
[{
|
|
196
|
+
"findingId": "<id>",
|
|
197
|
+
"tainted": true|false,
|
|
198
|
+
"sanitized": true|false,
|
|
199
|
+
"exploitability": "confirmed"|"likely"|"unlikely"|"false_positive",
|
|
200
|
+
"reasoning": "<one sentence>"
|
|
201
|
+
}]
|
|
202
|
+
|
|
203
|
+
Rules:
|
|
204
|
+
- If the value is a hardcoded string literal with no user input path, it is NOT tainted.
|
|
205
|
+
- If there is a validation library (zod, joi, yup, ajv) or sanitization function between input and sink, mark sanitized=true.
|
|
206
|
+
- If the code is in a test file, example, or documentation, mark as false_positive.
|
|
207
|
+
- If you cannot determine taint flow from the provided context, mark exploitability as "unlikely" rather than guessing.
|
|
208
|
+
- Be conservative: only mark "confirmed" when there is a clear, unsanitized path from user input to dangerous sink.`;
|
|
209
|
+
|
|
210
|
+
// =============================================================================
|
|
211
|
+
// DEEP ANALYZER
|
|
212
|
+
// =============================================================================
|
|
213
|
+
|
|
214
|
+
export class DeepAnalyzer {
|
|
215
|
+
/**
|
|
216
|
+
* @param {object} options
|
|
217
|
+
* @param {object} options.provider — LLM provider instance (from createProvider)
|
|
218
|
+
* @param {number} options.budgetCents — Max spend in cents (default: 50)
|
|
219
|
+
* @param {boolean} options.verbose — Log analysis progress
|
|
220
|
+
*/
|
|
221
|
+
constructor(options = {}) {
|
|
222
|
+
this.provider = options.provider || null;
|
|
223
|
+
this.budgetCents = options.budgetCents ?? 50;
|
|
224
|
+
this.verbose = options.verbose || false;
|
|
225
|
+
this.spentCents = 0;
|
|
226
|
+
this.analyzedCount = 0;
|
|
227
|
+
this._tier2Count = 0;
|
|
228
|
+
this._tier3Count = 0;
|
|
229
|
+
this._skippedCount = 0;
|
|
230
|
+
|
|
231
|
+
// Large-context mode for local models (Gemma 4, etc.)
|
|
232
|
+
const ctxWindow = this.provider?.contextWindow ?? 0;
|
|
233
|
+
this.largeContext = ctxWindow >= 65536;
|
|
234
|
+
this.maxFileChars = this.largeContext ? MAX_FILE_CHARS_LARGE_CTX : MAX_FILE_CHARS_DEFAULT;
|
|
235
|
+
this.batchSize = this.largeContext ? 15 : 5;
|
|
236
|
+
|
|
237
|
+
// Whether we can use multi-tier structured output routing
|
|
238
|
+
this._isAnthropic = this.provider?.name === 'Anthropic';
|
|
239
|
+
this._supportsTools = this._isAnthropic || this.provider?.supportsStructuredOutput === true;
|
|
240
|
+
}
|
|
241
|
+
|
|
242
|
+
/**
|
|
243
|
+
* Create a DeepAnalyzer with auto-detected provider.
|
|
244
|
+
* Returns null if no provider is available.
|
|
245
|
+
*/
|
|
246
|
+
static create(rootPath, options = {}) {
|
|
247
|
+
if (options.local) {
|
|
248
|
+
const provider = createProvider('gemma4', null, {
|
|
249
|
+
model: options.model,
|
|
250
|
+
baseUrl: options.ollamaUrl,
|
|
251
|
+
});
|
|
252
|
+
return new DeepAnalyzer({ provider, ...options });
|
|
253
|
+
}
|
|
254
|
+
|
|
255
|
+
const provider = autoDetectProvider(rootPath, {
|
|
256
|
+
provider: options.provider,
|
|
257
|
+
baseUrl: options.baseUrl,
|
|
258
|
+
model: options.model,
|
|
259
|
+
});
|
|
260
|
+
if (!provider) return null;
|
|
261
|
+
|
|
262
|
+
return new DeepAnalyzer({ provider, ...options });
|
|
263
|
+
}
|
|
264
|
+
|
|
265
|
+
/**
|
|
266
|
+
* Analyze findings with LLM-powered taint analysis.
|
|
267
|
+
* Uses multi-tier pipeline when Anthropic is detected; single-tier otherwise.
|
|
268
|
+
*
|
|
269
|
+
* @param {object[]} findings — All findings from agents
|
|
270
|
+
* @param {object} context — { rootPath, recon }
|
|
271
|
+
* @returns {Promise<object[]>} — Findings with deepAnalysis attached
|
|
272
|
+
*/
|
|
273
|
+
async analyze(findings, context = {}) {
|
|
274
|
+
if (!this.provider) return findings;
|
|
275
|
+
|
|
276
|
+
// Load playbook context once — injected into all LLM calls for this run
|
|
277
|
+
if (context.rootPath) {
|
|
278
|
+
try {
|
|
279
|
+
const PlaybookClass = await getScanPlaybook();
|
|
280
|
+
const playbook = new PlaybookClass(context.rootPath);
|
|
281
|
+
this._playbookContext = playbook.getPromptContext();
|
|
282
|
+
} catch { this._playbookContext = ''; }
|
|
283
|
+
}
|
|
284
|
+
|
|
285
|
+
// Only analyze critical/high findings
|
|
286
|
+
const candidates = findings.filter(
|
|
287
|
+
f => f.severity === 'critical' || f.severity === 'high'
|
|
288
|
+
);
|
|
289
|
+
if (candidates.length === 0) return findings;
|
|
290
|
+
|
|
291
|
+
// Cap at MAX_FINDINGS with budget scaling
|
|
292
|
+
const toAnalyze = candidates.slice(0, MAX_FINDINGS);
|
|
293
|
+
const estimatedCost = this._estimateCost(toAnalyze.length);
|
|
294
|
+
if (estimatedCost > this.budgetCents) {
|
|
295
|
+
const affordable = Math.floor(this.budgetCents / (estimatedCost / toAnalyze.length));
|
|
296
|
+
toAnalyze.length = Math.max(1, affordable);
|
|
297
|
+
}
|
|
298
|
+
|
|
299
|
+
const results = this._supportsTools
|
|
300
|
+
? await this._analyzeTiered(toAnalyze, context)
|
|
301
|
+
: await this._analyzeSingleTier(toAnalyze, context);
|
|
302
|
+
|
|
303
|
+
// Attach deep analysis to findings
|
|
304
|
+
for (const finding of findings) {
|
|
305
|
+
const id = this._findingId(finding);
|
|
306
|
+
const analysis = results.get(id);
|
|
307
|
+
if (analysis) {
|
|
308
|
+
finding.deepAnalysis = {
|
|
309
|
+
tainted: analysis.tainted,
|
|
310
|
+
sanitized: analysis.sanitized,
|
|
311
|
+
exploitability: analysis.exploitability,
|
|
312
|
+
reasoning: analysis.reasoning,
|
|
313
|
+
...(analysis.attackVector ? { attackVector: analysis.attackVector } : {}),
|
|
314
|
+
...(analysis.businessImpact ? { businessImpact: analysis.businessImpact } : {}),
|
|
315
|
+
...(analysis.fix ? { fix: analysis.fix } : {}),
|
|
316
|
+
};
|
|
317
|
+
|
|
318
|
+
if (analysis.exploitability === 'false_positive') {
|
|
319
|
+
finding.confidence = 'low';
|
|
320
|
+
} else if (analysis.exploitability === 'unlikely') {
|
|
321
|
+
if (finding.confidence === 'high') finding.confidence = 'medium';
|
|
322
|
+
} else if (analysis.exploitability === 'confirmed') {
|
|
323
|
+
finding.confidence = 'high';
|
|
324
|
+
}
|
|
325
|
+
}
|
|
326
|
+
}
|
|
327
|
+
|
|
328
|
+
return findings;
|
|
329
|
+
}
|
|
330
|
+
|
|
331
|
+
// ===========================================================================
|
|
332
|
+
// MULTI-TIER PIPELINE (Anthropic only)
|
|
333
|
+
// ===========================================================================
|
|
334
|
+
|
|
335
|
+
async _analyzeTiered(findings, context) {
|
|
336
|
+
const results = new Map();
|
|
337
|
+
|
|
338
|
+
// Model selection: Anthropic uses tier-specific models; others use provider's default
|
|
339
|
+
const tier1Model = this._isAnthropic ? TIER1_MODEL : null;
|
|
340
|
+
const tier2Model = this._isAnthropic ? TIER2_MODEL : null;
|
|
341
|
+
const tier3Model = this._isAnthropic ? TIER3_MODEL : null;
|
|
342
|
+
const providerLabel = this._isAnthropic ? 'Haiku' : this.provider.name;
|
|
343
|
+
|
|
344
|
+
// ── Tier 1: Haiku triage ────────────────────────────────────────────────
|
|
345
|
+
if (this.verbose) console.log(` [Tier 1] Triaging ${findings.length} findings with ${providerLabel}...`);
|
|
346
|
+
|
|
347
|
+
const triageMap = await this._runTriage(findings, context, tier1Model);
|
|
348
|
+
|
|
349
|
+
const toReview = findings.filter(f => triageMap.get(this._findingId(f)) === 'review');
|
|
350
|
+
const toEscalate = findings.filter(f => triageMap.get(this._findingId(f)) === 'escalate');
|
|
351
|
+
const skipped = findings.length - toReview.length - toEscalate.length;
|
|
352
|
+
|
|
353
|
+
this._skippedCount += skipped;
|
|
354
|
+
|
|
355
|
+
if (this.verbose) {
|
|
356
|
+
console.log(` [Tier 1] Results: ${toEscalate.length} escalate, ${toReview.length} review, ${skipped} skip`);
|
|
357
|
+
}
|
|
358
|
+
|
|
359
|
+
// ── Tier 2: Sonnet deep analysis ────────────────────────────────────────
|
|
360
|
+
if (toReview.length > 0 && this.spentCents < this.budgetCents) {
|
|
361
|
+
const tier2Label = this._isAnthropic ? 'Sonnet' : this.provider.name;
|
|
362
|
+
if (this.verbose) console.log(` [Tier 2] Deep-analyzing ${toReview.length} findings with ${tier2Label}...`);
|
|
363
|
+
const tier2Results = await this._runDeepAnalysis(toReview, context, tier2Model);
|
|
364
|
+
for (const [id, analysis] of tier2Results) results.set(id, analysis);
|
|
365
|
+
this._tier2Count += toReview.length;
|
|
366
|
+
}
|
|
367
|
+
|
|
368
|
+
// ── Tier 3: Opus exploit chain ──────────────────────────────────────────
|
|
369
|
+
if (toEscalate.length > 0 && this.spentCents < this.budgetCents) {
|
|
370
|
+
const tier3Label = this._isAnthropic ? 'Opus' : this.provider.name;
|
|
371
|
+
if (this.verbose) console.log(` [Tier 3] Running exploit-chain analysis on ${toEscalate.length} findings with ${tier3Label}...`);
|
|
372
|
+
const tier3Results = await this._runExploitChain(toEscalate, context, tier3Model);
|
|
373
|
+
for (const [id, analysis] of tier3Results) results.set(id, analysis);
|
|
374
|
+
this._tier3Count += toEscalate.length;
|
|
375
|
+
}
|
|
376
|
+
|
|
377
|
+
this.analyzedCount += findings.length - skipped;
|
|
378
|
+
return results;
|
|
379
|
+
}
|
|
380
|
+
|
|
381
|
+
/** Tier 1: quick triage — returns Map<findingId, 'skip'|'review'|'escalate'> */
|
|
382
|
+
async _runTriage(findings, context, model = null) {
|
|
383
|
+
const triageMap = new Map();
|
|
384
|
+
// Default everything to 'review' so nothing is silently dropped on error
|
|
385
|
+
for (const f of findings) triageMap.set(this._findingId(f), 'review');
|
|
386
|
+
|
|
387
|
+
const batchSize = 10; // Haiku can handle larger batches
|
|
388
|
+
for (let i = 0; i < findings.length; i += batchSize) {
|
|
389
|
+
if (this.spentCents >= this.budgetCents) break;
|
|
390
|
+
const batch = findings.slice(i, i + batchSize);
|
|
391
|
+
// Tier 1 is about cheap, fast signal — no file context, metadata only.
|
|
392
|
+
// File context is fetched only in Tier 2+ where it's worth the token cost.
|
|
393
|
+
const items = batch.map(f => ({
|
|
394
|
+
findingId: this._findingId(f),
|
|
395
|
+
rule: f.rule,
|
|
396
|
+
severity: f.severity,
|
|
397
|
+
title: f.title,
|
|
398
|
+
file: f.file ? path.basename(f.file) : 'unknown',
|
|
399
|
+
line: f.line,
|
|
400
|
+
matched: (f.matched || '').slice(0, 200),
|
|
401
|
+
description: (f.description || '').slice(0, 120),
|
|
402
|
+
}));
|
|
403
|
+
|
|
404
|
+
const prompt = `Triage these ${items.length} security findings. For each, decide: "skip" (obvious false-positive), "review" (needs deeper analysis), or "escalate" (confirmed critical, clear user-input-to-dangerous-sink path).\n\nFindings:\n${JSON.stringify(items, null, 2)}`;
|
|
405
|
+
|
|
406
|
+
try {
|
|
407
|
+
const result = await this.provider.completeWithTools(
|
|
408
|
+
TRIAGE_SYSTEM,
|
|
409
|
+
prompt,
|
|
410
|
+
'triage_findings',
|
|
411
|
+
TRIAGE_SCHEMA,
|
|
412
|
+
{ maxTokens: 1024, ...(model ? { model } : {}) }
|
|
413
|
+
);
|
|
414
|
+
|
|
415
|
+
this._trackCost(prompt.length, JSON.stringify(result || '').length);
|
|
416
|
+
|
|
417
|
+
for (const item of (result?.results ?? [])) {
|
|
418
|
+
if (triageMap.has(item.findingId)) {
|
|
419
|
+
triageMap.set(item.findingId, item.tier);
|
|
420
|
+
}
|
|
421
|
+
}
|
|
422
|
+
} catch (err) {
|
|
423
|
+
if (this.verbose) console.log(` [Tier 1] Batch failed: ${err.message}`);
|
|
424
|
+
}
|
|
425
|
+
}
|
|
426
|
+
|
|
427
|
+
return triageMap;
|
|
428
|
+
}
|
|
429
|
+
|
|
430
|
+
/** Tier 2: deep taint analysis — returns Map<findingId, analysis> */
|
|
431
|
+
async _runDeepAnalysis(findings, context, model = null) {
|
|
432
|
+
const results = new Map();
|
|
433
|
+
|
|
434
|
+
for (let i = 0; i < findings.length; i += this.batchSize) {
|
|
435
|
+
if (this.spentCents >= this.budgetCents) break;
|
|
436
|
+
const batch = findings.slice(i, i + this.batchSize);
|
|
437
|
+
const items = batch.map(f => ({
|
|
438
|
+
findingId: this._findingId(f),
|
|
439
|
+
rule: f.rule,
|
|
440
|
+
severity: f.severity,
|
|
441
|
+
title: f.title,
|
|
442
|
+
description: f.description,
|
|
443
|
+
file: f.file ? path.basename(f.file) : 'unknown',
|
|
444
|
+
line: f.line,
|
|
445
|
+
matched: (f.matched || '').slice(0, 200),
|
|
446
|
+
codeContext: this._getFileContext(f),
|
|
447
|
+
}));
|
|
448
|
+
|
|
449
|
+
let projectContext = this._buildProjectContext(context);
|
|
450
|
+
const prompt = `Analyze these ${items.length} security findings for taint reachability and exploitability.${projectContext}\n\nFindings:\n${JSON.stringify(items, null, 2)}`;
|
|
451
|
+
|
|
452
|
+
try {
|
|
453
|
+
const result = await this.provider.completeWithTools(
|
|
454
|
+
DEEP_SYSTEM,
|
|
455
|
+
prompt,
|
|
456
|
+
'report_analysis',
|
|
457
|
+
DEEP_ANALYSIS_SCHEMA,
|
|
458
|
+
{ maxTokens: 1500, ...(model ? { model } : {}) }
|
|
459
|
+
);
|
|
460
|
+
|
|
461
|
+
this._trackCost(prompt.length, JSON.stringify(result || '').length);
|
|
462
|
+
|
|
463
|
+
for (const item of (result?.results ?? [])) {
|
|
464
|
+
results.set(item.findingId, item);
|
|
465
|
+
}
|
|
466
|
+
} catch (err) {
|
|
467
|
+
if (this.verbose) console.log(` [Tier 2] Batch failed: ${err.message}`);
|
|
468
|
+
// Fallback: try plain text completion + parse
|
|
469
|
+
try {
|
|
470
|
+
const fallbackResult = await this._runSingleTierBatch(batch, context, model);
|
|
471
|
+
for (const [id, analysis] of fallbackResult) results.set(id, analysis);
|
|
472
|
+
} catch { /* ignore */ }
|
|
473
|
+
}
|
|
474
|
+
}
|
|
475
|
+
|
|
476
|
+
return results;
|
|
477
|
+
}
|
|
478
|
+
|
|
479
|
+
/** Tier 3: exploit-chain analysis — returns Map<findingId, analysis> */
|
|
480
|
+
async _runExploitChain(findings, context, model = null) {
|
|
481
|
+
const results = new Map();
|
|
482
|
+
|
|
483
|
+
// Single findings per call for maximum depth
|
|
484
|
+
for (const finding of findings) {
|
|
485
|
+
if (this.spentCents >= this.budgetCents) break;
|
|
486
|
+
|
|
487
|
+
const item = {
|
|
488
|
+
findingId: this._findingId(finding),
|
|
489
|
+
rule: finding.rule,
|
|
490
|
+
severity: finding.severity,
|
|
491
|
+
title: finding.title,
|
|
492
|
+
description: finding.description,
|
|
493
|
+
file: finding.file ? path.basename(finding.file) : 'unknown',
|
|
494
|
+
line: finding.line,
|
|
495
|
+
matched: (finding.matched || '').slice(0, 400),
|
|
496
|
+
codeContext: this._getFileContext(finding), // Full context window
|
|
497
|
+
};
|
|
498
|
+
|
|
499
|
+
const prompt = `Perform full exploit-chain analysis on this security finding.\n\nFinding:\n${JSON.stringify(item, null, 2)}`;
|
|
500
|
+
|
|
501
|
+
try {
|
|
502
|
+
const result = await this.provider.completeWithTools(
|
|
503
|
+
EXPLOIT_SYSTEM,
|
|
504
|
+
prompt,
|
|
505
|
+
'report_exploit_chain',
|
|
506
|
+
EXPLOIT_SCHEMA,
|
|
507
|
+
{ maxTokens: 2048, ...(model ? { model } : {}) }
|
|
508
|
+
);
|
|
509
|
+
|
|
510
|
+
this._trackCost(prompt.length, JSON.stringify(result || '').length);
|
|
511
|
+
|
|
512
|
+
for (const analysis of (result?.results ?? [])) {
|
|
513
|
+
results.set(analysis.findingId, analysis);
|
|
514
|
+
}
|
|
515
|
+
} catch (err) {
|
|
516
|
+
if (this.verbose) console.log(` [Tier 3] Failed for ${item.findingId}: ${err.message}`);
|
|
517
|
+
// Fallback to Tier 2 analysis on error
|
|
518
|
+
try {
|
|
519
|
+
const fallback = await this._runDeepAnalysis([finding], context, this._isAnthropic ? TIER2_MODEL : null);
|
|
520
|
+
for (const [id, analysis] of fallback) results.set(id, analysis);
|
|
521
|
+
} catch { /* ignore */ }
|
|
522
|
+
}
|
|
523
|
+
}
|
|
524
|
+
|
|
525
|
+
return results;
|
|
526
|
+
}
|
|
527
|
+
|
|
528
|
+
// ===========================================================================
|
|
529
|
+
// SINGLE-TIER PIPELINE (non-Anthropic providers)
|
|
530
|
+
// ===========================================================================
|
|
531
|
+
|
|
532
|
+
async _analyzeSingleTier(findings, context) {
|
|
533
|
+
const results = new Map();
|
|
534
|
+
|
|
535
|
+
for (let i = 0; i < findings.length; i += this.batchSize) {
|
|
536
|
+
if (this.spentCents >= this.budgetCents) {
|
|
537
|
+
if (this.verbose) console.log(` Deep analysis: budget exhausted (${this.spentCents}c / ${this.budgetCents}c)`);
|
|
538
|
+
break;
|
|
539
|
+
}
|
|
540
|
+
|
|
541
|
+
const batch = findings.slice(i, i + this.batchSize);
|
|
542
|
+
try {
|
|
543
|
+
const batchResults = await this._runSingleTierBatch(batch, context);
|
|
544
|
+
for (const [id, analysis] of batchResults) results.set(id, analysis);
|
|
545
|
+
this.analyzedCount += batch.length;
|
|
546
|
+
} catch (err) {
|
|
547
|
+
if (this.verbose) console.log(` Deep analysis batch failed: ${err.message}`);
|
|
548
|
+
// Continue with remaining batches
|
|
549
|
+
}
|
|
550
|
+
}
|
|
551
|
+
|
|
552
|
+
return results;
|
|
553
|
+
}
|
|
554
|
+
|
|
555
|
+
async _runSingleTierBatch(batch, context, model = null) {
|
|
556
|
+
const results = new Map();
|
|
557
|
+
const prompt = this._buildSingleTierPrompt(batch, context);
|
|
558
|
+
|
|
559
|
+
const response = await this.provider.complete(
|
|
560
|
+
SINGLE_TIER_SYSTEM,
|
|
561
|
+
prompt,
|
|
562
|
+
{ maxTokens: 4000, ...(model ? { model } : {}) }
|
|
563
|
+
);
|
|
564
|
+
|
|
565
|
+
this._trackCost(prompt.length, response.length);
|
|
566
|
+
|
|
567
|
+
const analyses = this._parseTextResponse(response);
|
|
568
|
+
for (const analysis of analyses) {
|
|
569
|
+
results.set(analysis.findingId, analysis);
|
|
570
|
+
}
|
|
571
|
+
return results;
|
|
572
|
+
}
|
|
573
|
+
|
|
574
|
+
_buildSingleTierPrompt(findings, context) {
|
|
575
|
+
const items = findings.map(f => ({
|
|
576
|
+
findingId: this._findingId(f),
|
|
577
|
+
rule: f.rule,
|
|
578
|
+
severity: f.severity,
|
|
579
|
+
title: f.title,
|
|
580
|
+
description: f.description,
|
|
581
|
+
file: f.file ? path.basename(f.file) : 'unknown',
|
|
582
|
+
line: f.line,
|
|
583
|
+
matched: (f.matched || '').slice(0, 200),
|
|
584
|
+
codeContext: this._getFileContext(f),
|
|
585
|
+
}));
|
|
586
|
+
|
|
587
|
+
const projectContext = this._buildProjectContext(context);
|
|
588
|
+
return `Analyze these ${items.length} security findings for taint reachability and exploitability.${projectContext}\n\nFindings:\n${JSON.stringify(items, null, 2)}`;
|
|
589
|
+
}
|
|
590
|
+
|
|
591
|
+
// ===========================================================================
|
|
592
|
+
// HELPERS
|
|
593
|
+
// ===========================================================================
|
|
594
|
+
|
|
595
|
+
_buildProjectContext(context) {
|
|
596
|
+
const parts = [];
|
|
597
|
+
|
|
598
|
+
// Playbook context (accumulated across scans — richer than single-run recon)
|
|
599
|
+
if (context.rootPath) {
|
|
600
|
+
try {
|
|
601
|
+
// Use cached playbook context if available (set by analyze())
|
|
602
|
+
if (this._playbookContext) {
|
|
603
|
+
parts.push(`Repo playbook:\n${this._playbookContext}`);
|
|
604
|
+
}
|
|
605
|
+
} catch { /* ignore */ }
|
|
606
|
+
}
|
|
607
|
+
|
|
608
|
+
// Single-run recon context
|
|
609
|
+
if (context.recon) {
|
|
610
|
+
const r = context.recon;
|
|
611
|
+
const reconParts = [];
|
|
612
|
+
if (r.frameworks?.length) reconParts.push(`Frameworks: ${r.frameworks.join(', ')}`);
|
|
613
|
+
if (r.databases?.length) reconParts.push(`Databases: ${r.databases.join(', ')}`);
|
|
614
|
+
if (r.authPatterns?.length) reconParts.push(`Auth: ${r.authPatterns.join(', ')}`);
|
|
615
|
+
if (reconParts.length) parts.push(reconParts.join('\n'));
|
|
616
|
+
}
|
|
617
|
+
|
|
618
|
+
return parts.length ? `\n\nProject context:\n${parts.join('\n\n')}` : '';
|
|
619
|
+
}
|
|
620
|
+
|
|
621
|
+
/**
|
|
622
|
+
* Get file content around the finding for LLM context, enriched with AST scope context.
|
|
623
|
+
* @param {object} finding
|
|
624
|
+
* @param {number} windowLines — Lines before/after (default: 20 = 40 line window)
|
|
625
|
+
*/
|
|
626
|
+
_getFileContext(finding, windowLines = 20) {
|
|
627
|
+
if (!finding.file) return '';
|
|
628
|
+
|
|
629
|
+
try {
|
|
630
|
+
const content = fs.readFileSync(finding.file, 'utf-8');
|
|
631
|
+
const lines = content.split('\n');
|
|
632
|
+
const lineNum = finding.line || 1;
|
|
633
|
+
|
|
634
|
+
// Extract AST & Scope structural context
|
|
635
|
+
let astHeader = '';
|
|
636
|
+
try {
|
|
637
|
+
const parsed = ASTParser.parse(content, finding.file);
|
|
638
|
+
const scopeTree = ScopeTree.build(parsed.ast, content);
|
|
639
|
+
const encFn = scopeTree.getEnclosingFunction(lineNum);
|
|
640
|
+
const taint = TaintTracker.evaluateFinding({
|
|
641
|
+
file: finding.file,
|
|
642
|
+
line: lineNum,
|
|
643
|
+
matched: finding.matched || '',
|
|
644
|
+
code: content,
|
|
645
|
+
scopeTree,
|
|
646
|
+
});
|
|
647
|
+
const guard = GuardrailDetector.checkProtection(finding, content);
|
|
648
|
+
|
|
649
|
+
const details = [];
|
|
650
|
+
if (encFn) details.push(`Enclosing Function: ${encFn.name}(${encFn.params ? encFn.params.join(', ') : ''}) [lines ${encFn.range.startLine}-${encFn.range.endLine}]`);
|
|
651
|
+
if (taint.sanitizer) details.push(`Detected Sanitizer: ${taint.sanitizer}`);
|
|
652
|
+
if (taint.isStatic) details.push(`Static Evaluation: Value is a hardcoded constant`);
|
|
653
|
+
if (guard.isProtected) details.push(`AI Defense Guardrail: ${guard.guardrail}`);
|
|
654
|
+
|
|
655
|
+
if (details.length > 0) {
|
|
656
|
+
astHeader = `[AST Structural Context]\n${details.map(d => `* ${d}`).join('\n')}\n\n`;
|
|
657
|
+
}
|
|
658
|
+
} catch { /* AST fallback */ }
|
|
659
|
+
|
|
660
|
+
let context;
|
|
661
|
+
if (this.largeContext) {
|
|
662
|
+
context = lines.map((l, i) => `${i + 1}: ${l}`).join('\n');
|
|
663
|
+
} else {
|
|
664
|
+
const start = Math.max(0, lineNum - windowLines - 1);
|
|
665
|
+
const end = Math.min(lines.length, lineNum + windowLines);
|
|
666
|
+
context = lines.slice(start, end)
|
|
667
|
+
.map((l, i) => `${start + i + 1}: ${l}`)
|
|
668
|
+
.join('\n');
|
|
669
|
+
}
|
|
670
|
+
|
|
671
|
+
const fullContext = astHeader + context;
|
|
672
|
+
if (fullContext.length > this.maxFileChars) {
|
|
673
|
+
return fullContext.slice(0, this.maxFileChars) + '\n... (truncated)';
|
|
674
|
+
}
|
|
675
|
+
|
|
676
|
+
return fullContext;
|
|
677
|
+
} catch {
|
|
678
|
+
return '';
|
|
679
|
+
}
|
|
680
|
+
}
|
|
681
|
+
|
|
682
|
+
_findingId(finding) {
|
|
683
|
+
const file = finding.file ? path.basename(finding.file) : 'unknown';
|
|
684
|
+
return `${file}:${finding.line}:${finding.rule}`;
|
|
685
|
+
}
|
|
686
|
+
|
|
687
|
+
_trackCost(promptChars, responseChars) {
|
|
688
|
+
const inputTokens = Math.ceil(promptChars / 4);
|
|
689
|
+
const outputTokens = Math.ceil(responseChars / 4);
|
|
690
|
+
this.spentCents += (inputTokens / 1000) * COST_PER_1K_INPUT
|
|
691
|
+
+ (outputTokens / 1000) * COST_PER_1K_OUTPUT;
|
|
692
|
+
}
|
|
693
|
+
|
|
694
|
+
_parseTextResponse(text) {
|
|
695
|
+
const cleaned = text
|
|
696
|
+
.replace(/^```(?:json)?\s*/i, '')
|
|
697
|
+
.replace(/\s*```\s*$/i, '')
|
|
698
|
+
.trim();
|
|
699
|
+
|
|
700
|
+
let parsed = null;
|
|
701
|
+
try {
|
|
702
|
+
parsed = JSON.parse(cleaned);
|
|
703
|
+
} catch {
|
|
704
|
+
parsed = null;
|
|
705
|
+
}
|
|
706
|
+
|
|
707
|
+
// Reasoning models often emit prose before/after the JSON despite
|
|
708
|
+
// instructions — fall back to extracting the first valid JSON array.
|
|
709
|
+
if (!Array.isArray(parsed)) {
|
|
710
|
+
parsed = this._extractJsonArray(text);
|
|
711
|
+
}
|
|
712
|
+
if (!Array.isArray(parsed)) return [];
|
|
713
|
+
|
|
714
|
+
return parsed.filter(item =>
|
|
715
|
+
item.findingId &&
|
|
716
|
+
typeof item.tainted === 'boolean' &&
|
|
717
|
+
typeof item.sanitized === 'boolean' &&
|
|
718
|
+
['confirmed', 'likely', 'unlikely', 'false_positive'].includes(item.exploitability)
|
|
719
|
+
);
|
|
720
|
+
}
|
|
721
|
+
|
|
722
|
+
/**
|
|
723
|
+
* Scan free text for the first balanced, parseable JSON array.
|
|
724
|
+
* Handles models that wrap JSON in prose or markdown.
|
|
725
|
+
*/
|
|
726
|
+
_extractJsonArray(text) {
|
|
727
|
+
for (let start = 0; start < text.length; start++) {
|
|
728
|
+
if (text[start] !== '[') continue;
|
|
729
|
+
let depth = 0, inStr = false, esc = false;
|
|
730
|
+
for (let i = start; i < text.length; i++) {
|
|
731
|
+
const ch = text[i];
|
|
732
|
+
if (inStr) {
|
|
733
|
+
if (esc) esc = false;
|
|
734
|
+
else if (ch === '\\') esc = true;
|
|
735
|
+
else if (ch === '"') inStr = false;
|
|
736
|
+
continue;
|
|
737
|
+
}
|
|
738
|
+
if (ch === '"') inStr = true;
|
|
739
|
+
else if (ch === '[') depth++;
|
|
740
|
+
else if (ch === ']') {
|
|
741
|
+
depth--;
|
|
742
|
+
if (depth === 0) {
|
|
743
|
+
try {
|
|
744
|
+
const candidate = JSON.parse(text.slice(start, i + 1));
|
|
745
|
+
if (Array.isArray(candidate) && candidate.length > 0) return candidate;
|
|
746
|
+
} catch { /* not valid JSON — keep scanning */ }
|
|
747
|
+
break;
|
|
748
|
+
}
|
|
749
|
+
}
|
|
750
|
+
}
|
|
751
|
+
}
|
|
752
|
+
return null;
|
|
753
|
+
}
|
|
754
|
+
|
|
755
|
+
_estimateCost(count) {
|
|
756
|
+
const inputCost = (count * EST_INPUT_TOKENS_PER_FINDING / 1000) * COST_PER_1K_INPUT;
|
|
757
|
+
const outputCost = (count * EST_OUTPUT_TOKENS_PER_FINDING / 1000) * COST_PER_1K_OUTPUT;
|
|
758
|
+
return inputCost + outputCost;
|
|
759
|
+
}
|
|
760
|
+
|
|
761
|
+
getStats() {
|
|
762
|
+
return {
|
|
763
|
+
analyzedCount: this.analyzedCount,
|
|
764
|
+
skippedCount: this._skippedCount,
|
|
765
|
+
tier2Count: this._tier2Count,
|
|
766
|
+
tier3Count: this._tier3Count,
|
|
767
|
+
spentCents: Math.round(this.spentCents * 100) / 100,
|
|
768
|
+
budgetCents: this.budgetCents,
|
|
769
|
+
provider: this.provider?.name || 'none',
|
|
770
|
+
multiTier: this._supportsTools,
|
|
771
|
+
isAnthropic: this._isAnthropic,
|
|
772
|
+
};
|
|
773
|
+
}
|
|
774
|
+
}
|
|
775
|
+
|
|
776
|
+
export default DeepAnalyzer;
|