praxis-sec 1.2.0 → 1.2.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/cli/agents/abom-generator.js +1 -1
- package/cli/agents/agent-attestation-agent.js +10 -1
- package/cli/agents/agent-config-scanner.js +1 -1
- package/cli/agents/ai-infra-inventory-agent.js +482 -482
- package/cli/agents/base-agent.js +1 -1
- package/cli/agents/endpoint-agent-abuse-agent.js +1 -1
- package/cli/agents/html-reporter.js +3 -3
- package/cli/agents/index.js +2 -2
- package/cli/agents/injection-tester.js +8 -1
- package/cli/agents/memory-poisoning-agent.js +1 -1
- package/cli/agents/model-file-scanner.js +1 -1
- package/cli/agents/orchestrator.js +11 -6
- package/cli/agents/prompt-injection-prober.js +228 -228
- package/cli/bin/praxis.js +7 -3
- package/cli/commands/agent-fix.js +1091 -1245
- package/cli/commands/audit.js +1228 -1216
- package/cli/commands/baseline.js +1 -1
- package/cli/commands/benchmark.js +1 -1
- package/cli/commands/ci.js +45 -21
- package/cli/commands/deps.js +11 -5
- package/cli/commands/env-audit.js +1 -1
- package/cli/commands/fix.js +1 -1
- package/cli/commands/mcp.js +1 -1
- package/cli/commands/red-team.js +350 -350
- package/cli/commands/remediate.js +1 -1
- package/cli/commands/rotate.js +1 -1
- package/cli/commands/rules.js +1 -1
- package/cli/commands/scan.js +554 -554
- package/cli/commands/score.js +1 -1
- package/cli/commands/undo.js +22 -77
- package/cli/commands/vibe-check.js +1 -1
- package/cli/core/fix-plan.js +274 -0
- package/cli/core/fs.js +27 -0
- package/cli/core/git-clone.js +8 -6
- package/cli/core/glob.js +56 -0
- package/cli/core/output/html-theme.js +158 -158
- package/cli/core/output/sarif.js +2 -2
- package/cli/core/web/jobs.js +2 -2
- package/cli/core/web/server.js +19 -8
- package/cli/data/threatpacks/latest.json +41 -41
- package/cli/integrations/github-action.js +136 -0
- package/cli/utils/plugin-loader.js +15 -95
- package/cli/utils/rule-import.js +227 -227
- package/cli/utils/rule-registry.js +425 -425
- package/cli/utils/scan-fingerprint.js +1 -1
- package/cli/utils/score-history.js +118 -118
- package/docs/USAGE.md +16 -9
- package/docs/design/WEB-UI.md +4 -5
- package/package.json +13 -4
|
@@ -1,482 +1,482 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* AI Infrastructure Inventory Agent
|
|
3
|
-
* =================================
|
|
4
|
-
*
|
|
5
|
-
* Completes Praxis's view of the AI system beyond application code — the
|
|
6
|
-
* three production layers where AI actually runs:
|
|
7
|
-
*
|
|
8
|
-
* Lane 1 Model gateways — LiteLLM / Portkey / Helicone / Cloudflare
|
|
9
|
-
* AI Gateway / OpenRouter configs. The choke
|
|
10
|
-
* point through which all LLM traffic flows.
|
|
11
|
-
* Lane 2 AI infrastructure — self-hosted LLM runtimes (Ollama, vLLM, TGI,
|
|
12
|
-
* SGLang, Triton, llama.cpp, NIM, ...) in
|
|
13
|
-
* Docker/K8s/Helm/compose/Terraform, and
|
|
14
|
-
* managed AI compute (Bedrock/SageMaker/Vertex).
|
|
15
|
-
* Lane 3 AI API endpoints — OpenAPI specs + framework routes (FastAPI,
|
|
16
|
-
* Flask, Express, Spring, Django) with auth-style
|
|
17
|
-
* capture and {id}-path BOLA candidates.
|
|
18
|
-
* Lane 4 AI data pipelines — dataset loaders that fetch/execute remote
|
|
19
|
-
* content, template injection in dataset configs,
|
|
20
|
-
* and eval-harness / agent-sandbox misconfigs
|
|
21
|
-
* (the July 2026 OpenAI–Hugging Face incident
|
|
22
|
-
* vectors: remote-code dataset loaders, template
|
|
23
|
-
* injection, disabled safety gates, broad egress).
|
|
24
|
-
*
|
|
25
|
-
* Inventory findings are MEDIUM/LOW — they map the attack surface rather
|
|
26
|
-
* than assert a vulnerability. Risk indicators (exposed keys, no-auth
|
|
27
|
-
* runtimes) escalate to HIGH.
|
|
28
|
-
*/
|
|
29
|
-
|
|
30
|
-
import fs from 'fs';
|
|
31
|
-
import path from 'path';
|
|
32
|
-
import { BaseAgent, createFinding, ruleTableLineMask } from './base-agent.js';
|
|
33
|
-
|
|
34
|
-
// =============================================================================
|
|
35
|
-
// LANE 1 — MODEL GATEWAYS
|
|
36
|
-
// =============================================================================
|
|
37
|
-
|
|
38
|
-
const GATEWAY_CHECKS = [
|
|
39
|
-
{
|
|
40
|
-
rule: 'AI_GATEWAY_LITELLM',
|
|
41
|
-
title: 'LiteLLM Proxy Gateway Detected',
|
|
42
|
-
glob: ['**/litellm*.yaml', '**/litellm*.yml', '**/config.yaml', '**/config.yml'],
|
|
43
|
-
regex: /model_list\s*:/i,
|
|
44
|
-
severity: 'medium',
|
|
45
|
-
description: 'A LiteLLM proxy config with model_list routes LLM traffic. The proxy is the choke point: a misconfiguration (open admin, missing master key, unverified model routes) exposes every prompt and response passing through it.',
|
|
46
|
-
fix: 'Review model_list entries and auth (master key, per-key budgets). Pin model versions and set egress controls.',
|
|
47
|
-
},
|
|
48
|
-
{
|
|
49
|
-
rule: 'AI_GATEWAY_PORTKEY',
|
|
50
|
-
title: 'Portkey Gateway Configuration Detected',
|
|
51
|
-
glob: ['**/portkey-config.json', '**/.portkey*'],
|
|
52
|
-
regex: /(?:provider|virtual_keys?|api_key)\s*:/i,
|
|
53
|
-
severity: 'medium',
|
|
54
|
-
description: 'Portkey gateway config found. Portkey configs commonly contain virtual API keys — treat as credential material.',
|
|
55
|
-
fix: 'Never commit virtual keys. Move to secrets manager and rotate any committed value.',
|
|
56
|
-
},
|
|
57
|
-
{
|
|
58
|
-
rule: 'AI_GATEWAY_HELICONE',
|
|
59
|
-
title: 'Helicone Proxy Endpoint Detected',
|
|
60
|
-
regex: /oai\.helicone\.ai|helicone\.ai\/v1/i,
|
|
61
|
-
severity: 'medium',
|
|
62
|
-
description: 'LLM calls route through the Helicone observability proxy — all prompt/response traffic transits a third party.',
|
|
63
|
-
fix: 'Verify the Helicone account and data-retention settings; exclude sensitive prompts from logging.',
|
|
64
|
-
},
|
|
65
|
-
{
|
|
66
|
-
rule: 'AI_GATEWAY_CLOUDFLARE_AI',
|
|
67
|
-
title: 'Cloudflare AI Gateway Detected',
|
|
68
|
-
regex: /gateway\.ai\.cloudflare\.com/i,
|
|
69
|
-
severity: 'low',
|
|
70
|
-
description: 'LLM traffic routes through Cloudflare AI Gateway.',
|
|
71
|
-
fix: 'Review gateway logs retention and per-app egress policy.',
|
|
72
|
-
},
|
|
73
|
-
{
|
|
74
|
-
rule: 'AI_GATEWAY_OPENROUTER',
|
|
75
|
-
title: 'OpenRouter Usage Detected',
|
|
76
|
-
regex: /openrouter\.ai\/api|OPENROUTER_API_KEY/i,
|
|
77
|
-
severity: 'medium',
|
|
78
|
-
description: 'OpenRouter (multi-model gateway) is configured. All prompts route through a third-party aggregator.',
|
|
79
|
-
fix: 'Verify model routing policy and keep API keys out of committed files.',
|
|
80
|
-
},
|
|
81
|
-
];
|
|
82
|
-
|
|
83
|
-
// =============================================================================
|
|
84
|
-
// LANE 2 — AI INFRASTRUCTURE / IaC
|
|
85
|
-
// =============================================================================
|
|
86
|
-
|
|
87
|
-
const RUNTIME_IMAGES = [
|
|
88
|
-
'ollama', 'ghcr.io/ollama', 'vllm', 'vllm/vllm-openai',
|
|
89
|
-
'text-generation-inference', 'ghcr.io/huggingface/text-generation-inference',
|
|
90
|
-
'sglang', 'lmsysorg/sglang', 'triton', 'nvcr.io/nvidia/tritonserver',
|
|
91
|
-
'nvidia-nim', 'nvcr.io/nvidia/nim', 'localai', 'llama.cpp',
|
|
92
|
-
'ghcr.io/ggml-org/llama.cpp', 'lorax', 'aphrodite', 'openllm',
|
|
93
|
-
'xinference', 'ray-llm', 'infinity', 'fastchat',
|
|
94
|
-
];
|
|
95
|
-
const RUNTIME_IMAGE_RE = new RegExp(`(?:FROM|image:)\\s*[\"']?(${RUNTIME_IMAGES.map(i => i.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')).join('|')})`, 'i');
|
|
96
|
-
|
|
97
|
-
const MANAGED_COMPUTE = [
|
|
98
|
-
{ rule: 'AI_MANAGED_BEDROCK', regex: /aws_bedrock_provisioned_model_throughput|aws_bedrock_custom_model/i, title: 'AWS Bedrock Provisioned Model', description: 'Terraform provisions AWS Bedrock model capacity — committed throughput is a billing + model-governance surface.' },
|
|
99
|
-
{ rule: 'AI_MANAGED_SAGEMAKER', regex: /aws_sagemaker_endpoint\s*["']?[a-zA-Z0-9_-]*(?:llm|model|inference|endpoint)/i, title: 'AWS SageMaker LLM Endpoint', description: 'Terraform provisions a SageMaker endpoint for an LLM — verify IAM scope and endpoint auth.' },
|
|
100
|
-
{ rule: 'AI_MANAGED_VERTEX', regex: /google_vertex_ai_endpoint/i, title: 'Google Vertex AI Endpoint', description: 'Terraform provisions a Vertex AI endpoint — verify network policy and model provenance.' },
|
|
101
|
-
];
|
|
102
|
-
|
|
103
|
-
// =============================================================================
|
|
104
|
-
// LANE 3 — AI API ENDPOINTS
|
|
105
|
-
// =============================================================================
|
|
106
|
-
|
|
107
|
-
const FRAMEWORK_ROUTE_RE = [
|
|
108
|
-
{ name: 'FastAPI/Starlette', regex: /@(?:app|router)\.(?:get|post|put|patch|delete)\s*\(\s*["']([^"']+)["']/g },
|
|
109
|
-
{ name: 'Flask', regex: /@(?:app|bp)\.route\s*\(\s*["']([^"']+)["']/g },
|
|
110
|
-
{ name: 'Express', regex: /(?:app|router)\.(?:get|post|put|patch|delete)\s*\(\s*["']([^"']+)["']/g },
|
|
111
|
-
{ name: 'Spring', regex: /@(?:Get|Post|Put|Patch|Delete|Request)Mapping\s*\(\s*(?:value\s*=\s*)?["']([^"']+)["']/g },
|
|
112
|
-
{ name: 'Django', regex: /(?:path|re_path)\s*\(\s*["']([^"']+)["']/g },
|
|
113
|
-
];
|
|
114
|
-
|
|
115
|
-
const BOLA_SEGMENT = /[\/{][^{}\/]*(?:\{|:)[a-zA-Z_][a-zA-Z0-9_]*[\}][^{}\/]*/;
|
|
116
|
-
|
|
117
|
-
// =============================================================================
|
|
118
|
-
// LANE 4 — AI DATA PIPELINES & EVAL HARNESSES
|
|
119
|
-
// =============================================================================
|
|
120
|
-
// Vectors from the July 2026 OpenAI–Hugging Face autonomous-agent incident:
|
|
121
|
-
// remote-code dataset loaders and dataset template injection (initial access),
|
|
122
|
-
// and eval-harness / agent-sandbox misconfigurations (OpenAI side).
|
|
123
|
-
|
|
124
|
-
// Remote-code dataset loader: fetch remote content then exec/eval/pickle-load it
|
|
125
|
-
const DATASET_REMOTE_LOADER = /(?:load_dataset|dataset_from_script|requests\.get|urllib\.request|httpx\.get|fetch\s*\(\s*["'`]https?:\/\/)[\s\S]{0,300}(?:\bexec\s*\(|\beval\s*\(|pickle\.loads|__import__\(|compile\s*\()/i;
|
|
126
|
-
|
|
127
|
-
// Unsafe dataset/model loading flags
|
|
128
|
-
const UNSAFE_DATASET_FLAG = /trust_remote_code\s*=\s*True|trust_remote_code\s*:\s*true/i;
|
|
129
|
-
|
|
130
|
-
// Template injection in dataset configs: template expressions invoking OS/module
|
|
131
|
-
const DATASET_TEMPLATE_INJECTION = /\{\{\s*(?:__import__|os\.|subprocess|exec|eval|__builtins__)[\s\S]{0,120}?\}\}|\$\{\s*(?:__import__|os\.|subprocess|exec|eval|process\.env)[\s\S]{0,120}?\}/i;
|
|
132
|
-
|
|
133
|
-
/**
|
|
134
|
-
* First 1-based line where `re` matches something that is not a rule table
|
|
135
|
-
* describing itself, or 0 when every match is rule-table prose.
|
|
136
|
-
*
|
|
137
|
-
* The lanes below run patterns against whole file contents, so on one of our own
|
|
138
|
-
* rule tables the match is the table's `description:`/`title:` prose naming the
|
|
139
|
-
* very risk being checked. Skipping only those lines keeps real configuration in
|
|
140
|
-
* the same file reportable, and keeps the reported line honest.
|
|
141
|
-
*/
|
|
142
|
-
function realMatchLine(content, re) {
|
|
143
|
-
const mask = ruleTableLineMask(content.split('\n'));
|
|
144
|
-
if (!mask) {
|
|
145
|
-
re.lastIndex = 0;
|
|
146
|
-
const first = re.exec(content);
|
|
147
|
-
return first ? content.slice(0, first.index).split('\n').length : 0;
|
|
148
|
-
}
|
|
149
|
-
re.lastIndex = 0;
|
|
150
|
-
let m;
|
|
151
|
-
while ((m = re.exec(content)) !== null) {
|
|
152
|
-
const lineIdx = content.slice(0, m.index).split('\n').length - 1;
|
|
153
|
-
if (!mask.has(lineIdx)) return lineIdx + 1;
|
|
154
|
-
if (!re.global) break;
|
|
155
|
-
if (m.index === re.lastIndex) re.lastIndex++;
|
|
156
|
-
}
|
|
157
|
-
return 0;
|
|
158
|
-
}
|
|
159
|
-
|
|
160
|
-
// Eval-harness / agent-sandbox: disabled safety gates, broad tool scope, egress
|
|
161
|
-
const EVAL_HARNESS_RISK = [
|
|
162
|
-
{
|
|
163
|
-
rule: 'AI_EVAL_HARNESS_DISABLED_GUARDRAILS',
|
|
164
|
-
title: 'Eval Harness Disables Safety Guardrails',
|
|
165
|
-
regex: /(?:cyber_refusals|safety_refusals|refusals|guardrails|safety_gates|harm_belen|filters)\s*[:=]\s*(?:disabled|false|off|none|0)/i,
|
|
166
|
-
severity: 'high',
|
|
167
|
-
description: 'An evaluation harness or agent-sandbox config disables model safety guardrails. This is the configuration class behind the OpenAI–Hugging Face July 2026 incident (models evaluated with cyber refusals disabled), and enables unchecked autonomous behavior during evaluation.',
|
|
168
|
-
fix: 'Keep safety guardrails enabled during evaluation, or gate their removal behind explicit containment controls (no internet, isolated egress).',
|
|
169
|
-
},
|
|
170
|
-
{
|
|
171
|
-
rule: 'AI_EVAL_HARNESS_BROAD_TOOL_SCOPE',
|
|
172
|
-
title: 'Eval Harness Grants Broad Tool/Shell Scope',
|
|
173
|
-
regex: /(?:tools?|tool_scope|permissions|allowed_commands)\s*[:=]\s*[\[{][\s\S]{0,200}(?:shell|code_exec|exec_command|bash|subprocess|registry_proxy|all_tools)/i, // praxis-ignore AGENT_TOOL_SHELL_ACCESS — detector definition
|
|
174
|
-
severity: 'medium',
|
|
175
|
-
description: 'An eval harness or agent sandbox grants shell/code-execution tool scope. Combined with network egress this reproduces the sandbox-escape pattern that allowed the July 2026 autonomous agent to pivot from evaluation into production infrastructure.',
|
|
176
|
-
fix: 'Scope evaluation tools to the minimum surface. Shell/code-execution tools should never coexist with internet access in the same sandbox.',
|
|
177
|
-
},
|
|
178
|
-
{
|
|
179
|
-
rule: 'AI_EVAL_HARNESS_BROAD_EGRESS',
|
|
180
|
-
title: 'Eval Harness Allows Unrestricted Network Egress',
|
|
181
|
-
regex: /(?:internet_egress|egress|network_access|allow_remote_downloads|bypass_egress|no_network)\s*[:=]\s*(?:constrained|unrestricted|true|enabled|any|all|false)/i,
|
|
182
|
-
severity: 'high',
|
|
183
|
-
description: 'An eval harness or sandbox allows unrestricted network egress (or explicitly bypasses egress controls). The OpenAI models escaped their sealed evaluation only after finding an egress path — unrestricted egress removes that barrier entirely.',
|
|
184
|
-
fix: 'Default evaluation environments to no-network. If egress is required, allowlist specific endpoints and proxy through a logging gateway.',
|
|
185
|
-
},
|
|
186
|
-
];
|
|
187
|
-
|
|
188
|
-
// =============================================================================
|
|
189
|
-
// AGENT
|
|
190
|
-
// =============================================================================
|
|
191
|
-
|
|
192
|
-
export class AiInfraInventoryAgent extends BaseAgent {
|
|
193
|
-
constructor() {
|
|
194
|
-
super(
|
|
195
|
-
'AiInfraInventoryAgent',
|
|
196
|
-
'Maps the production AI system: model gateways, self-hosted/managed AI infrastructure, and AI API endpoints',
|
|
197
|
-
'llm'
|
|
198
|
-
);
|
|
199
|
-
}
|
|
200
|
-
|
|
201
|
-
shouldRun() {
|
|
202
|
-
return true; // these layers exist in most modern AI-bearing repos
|
|
203
|
-
}
|
|
204
|
-
|
|
205
|
-
async analyze(context) {
|
|
206
|
-
const { files = [], rootPath } = context;
|
|
207
|
-
const findings = [];
|
|
208
|
-
|
|
209
|
-
const read = (f) => {
|
|
210
|
-
try { return fs.readFileSync(f, 'utf8'); } catch { return ''; }
|
|
211
|
-
};
|
|
212
|
-
|
|
213
|
-
// ── Lane 1: model gateways ─────────────────────────────────────────────
|
|
214
|
-
for (const check of GATEWAY_CHECKS) {
|
|
215
|
-
for (const file of files) {
|
|
216
|
-
const rel = path.relative(rootPath, file).replace(/\\/g, '/');
|
|
217
|
-
if (check.glob) {
|
|
218
|
-
const matches = check.glob.some(g => {
|
|
219
|
-
const parts = g.replace(/\*\*\//g, '').split('/');
|
|
220
|
-
const last = parts[parts.length - 1];
|
|
221
|
-
return (last.includes('*') ? new RegExp(`^${last.replace(/\*/g, '.*')}$`).test(path.basename(file)) : last === path.basename(file));
|
|
222
|
-
});
|
|
223
|
-
if (!matches) continue;
|
|
224
|
-
}
|
|
225
|
-
const content = read(file);
|
|
226
|
-
if (!check.regex.test(content)) continue;
|
|
227
|
-
const gwLine = realMatchLine(content, check.regex);
|
|
228
|
-
if (gwLine === 0) continue;
|
|
229
|
-
findings.push(createFinding({
|
|
230
|
-
file,
|
|
231
|
-
line: gwLine,
|
|
232
|
-
severity: check.severity,
|
|
233
|
-
category: this.category,
|
|
234
|
-
rule: check.rule,
|
|
235
|
-
title: check.title,
|
|
236
|
-
description: check.description,
|
|
237
|
-
matched: check.rule,
|
|
238
|
-
confidence: 'high',
|
|
239
|
-
cwe: 'CWE-1357',
|
|
240
|
-
owasp: 'ASI10',
|
|
241
|
-
fix: check.fix,
|
|
242
|
-
}));
|
|
243
|
-
}
|
|
244
|
-
}
|
|
245
|
-
|
|
246
|
-
// ── Lane 2: AI infrastructure ──────────────────────────────────────────
|
|
247
|
-
for (const file of files) {
|
|
248
|
-
const rel = path.relative(rootPath, file).replace(/\\/g, '/');
|
|
249
|
-
const isIaC = /(?:Dockerfile|docker-compose[^/]*\.ya?ml|\.ya?ml|\.tf|k8s|helm|deployment)/i.test(rel);
|
|
250
|
-
if (!isIaC) continue;
|
|
251
|
-
const content = read(file);
|
|
252
|
-
if (!content) continue;
|
|
253
|
-
|
|
254
|
-
const runtimeMatch = RUNTIME_IMAGE_RE.exec(content);
|
|
255
|
-
if (runtimeMatch) {
|
|
256
|
-
findings.push(createFinding({
|
|
257
|
-
file,
|
|
258
|
-
line: 1,
|
|
259
|
-
severity: 'medium',
|
|
260
|
-
category: this.category,
|
|
261
|
-
rule: 'AI_INFRA_RUNTIME_DEPLOYMENT',
|
|
262
|
-
title: `Self-Hosted LLM Runtime (${runtimeMatch[1]})`,
|
|
263
|
-
description: `The repo deploys ${runtimeMatch[1]}, a self-hosted LLM runtime. Operational responsibility: the deployment must enforce endpoint auth, egress limits, model provenance, and patching — a publicly reachable unauthenticated inference endpoint is a direct data/model theft surface.`,
|
|
264
|
-
matched: runtimeMatch[1],
|
|
265
|
-
confidence: 'high',
|
|
266
|
-
cwe: 'CWE-306',
|
|
267
|
-
owasp: 'ASI06',
|
|
268
|
-
fix: 'Authenticate the inference endpoint, restrict network egress, pin runtime/model versions, and monitor inference logs.',
|
|
269
|
-
}));
|
|
270
|
-
}
|
|
271
|
-
|
|
272
|
-
for (const mc of MANAGED_COMPUTE) {
|
|
273
|
-
const mcLine = mc.regex.test(content) ? realMatchLine(content, mc.regex) : 0;
|
|
274
|
-
if (mcLine) {
|
|
275
|
-
findings.push(createFinding({
|
|
276
|
-
file,
|
|
277
|
-
line: mcLine,
|
|
278
|
-
severity: 'medium',
|
|
279
|
-
category: this.category,
|
|
280
|
-
rule: mc.rule,
|
|
281
|
-
title: mc.title,
|
|
282
|
-
description: mc.description,
|
|
283
|
-
matched: mc.rule,
|
|
284
|
-
confidence: 'high',
|
|
285
|
-
cwe: 'CWE-1357',
|
|
286
|
-
owasp: 'ASI10',
|
|
287
|
-
fix: 'Review IAM/network policies around the managed AI resource.',
|
|
288
|
-
}));
|
|
289
|
-
}
|
|
290
|
-
}
|
|
291
|
-
}
|
|
292
|
-
|
|
293
|
-
// ── Lane 3: API endpoints ─────────────────────────────────────────────
|
|
294
|
-
for (const file of files) {
|
|
295
|
-
const rel = path.relative(rootPath, file).replace(/\\/g, '/');
|
|
296
|
-
const isSpec = /openapi|swagger/i.test(rel) && /\.(json|ya?ml)$/.test(rel);
|
|
297
|
-
if (isSpec) {
|
|
298
|
-
const content = read(file);
|
|
299
|
-
let spec = null;
|
|
300
|
-
try { spec = JSON.parse(content); } catch { /* yaml specs not parsed — see FRAMEWORK_ROUTE_RE */ }
|
|
301
|
-
if (spec) {
|
|
302
|
-
const paths = spec.paths || {};
|
|
303
|
-
const securitySchemes = spec.components?.securitySchemes || spec.securityDefinitions || {};
|
|
304
|
-
const authStyles = Object.values(securitySchemes).map(s => s.type || s.scheme || 'unknown');
|
|
305
|
-
const routeEntries = [];
|
|
306
|
-
for (const [p, methods] of Object.entries(paths)) {
|
|
307
|
-
for (const m of Object.keys(methods || {})) {
|
|
308
|
-
if (['get', 'post', 'put', 'patch', 'delete'].includes(m.toLowerCase())) routeEntries.push({ path: p, method: m });
|
|
309
|
-
}
|
|
310
|
-
}
|
|
311
|
-
if (routeEntries.length > 0) {
|
|
312
|
-
const bola = routeEntries.filter(r => BOLA_SEGMENT.test(r.path));
|
|
313
|
-
findings.push(createFinding({
|
|
314
|
-
file,
|
|
315
|
-
line: 1,
|
|
316
|
-
severity: 'low',
|
|
317
|
-
category: this.category,
|
|
318
|
-
rule: 'AI_API_SPEC_INVENTORY',
|
|
319
|
-
title: `API Spec: ${routeEntries.length} Routes (auth: ${authStyles.length ? authStyles.join(',') : 'none declared'})`,
|
|
320
|
-
description: `OpenAPI/Swagger spec exposes ${routeEntries.length} routes. Auth styles declared: ${authStyles.length ? authStyles.join(', ') : 'NONE — every route may be unauthenticated'}. This is the attack-surface map for the AI service.`,
|
|
321
|
-
matched: `${routeEntries.length} routes`,
|
|
322
|
-
confidence: 'high',
|
|
323
|
-
cwe: 'CWE-1059',
|
|
324
|
-
owasp: 'ASI02',
|
|
325
|
-
fix: 'Ensure every route has an explicit auth scheme; review the spec for admin/debug routes.',
|
|
326
|
-
}));
|
|
327
|
-
if (bola.length > 0) {
|
|
328
|
-
findings.push(createFinding({
|
|
329
|
-
file,
|
|
330
|
-
line: 1,
|
|
331
|
-
severity: 'medium',
|
|
332
|
-
category: this.category,
|
|
333
|
-
rule: 'AI_API_BOLA_CANDIDATE',
|
|
334
|
-
title: `BOLA Candidates: ${bola.length} Object-ID Path Route(s)`,
|
|
335
|
-
description: `Route(s) with object-id path segments (e.g. ${bola[0].path}) are classic Broken Object Level Authorization candidates — every object access must be ownership-checked.`,
|
|
336
|
-
matched: bola.slice(0, 3).map(r => r.path).join(', '),
|
|
337
|
-
confidence: 'medium',
|
|
338
|
-
cwe: 'CWE-639',
|
|
339
|
-
owasp: 'ASI02',
|
|
340
|
-
fix: 'Verify per-object authorization checks (ownership/tenant scoping) on every flagged route.',
|
|
341
|
-
}));
|
|
342
|
-
}
|
|
343
|
-
}
|
|
344
|
-
}
|
|
345
|
-
continue;
|
|
346
|
-
}
|
|
347
|
-
|
|
348
|
-
const ext = path.extname(file).toLowerCase();
|
|
349
|
-
if (!['.py', '.js', '.ts', '.java'].includes(ext)) continue;
|
|
350
|
-
const content = read(file);
|
|
351
|
-
if (!content) continue;
|
|
352
|
-
|
|
353
|
-
for (const fr of FRAMEWORK_ROUTE_RE) {
|
|
354
|
-
fr.regex.lastIndex = 0;
|
|
355
|
-
let m;
|
|
356
|
-
let count = 0;
|
|
357
|
-
const sample = [];
|
|
358
|
-
while ((m = fr.regex.exec(content)) !== null) {
|
|
359
|
-
count++;
|
|
360
|
-
if (sample.length < 3) sample.push(m[1]);
|
|
361
|
-
}
|
|
362
|
-
if (count === 0) continue;
|
|
363
|
-
const bolaCount = sample.filter(r => BOLA_SEGMENT.test(r)).length;
|
|
364
|
-
findings.push(createFinding({
|
|
365
|
-
file,
|
|
366
|
-
line: 1,
|
|
367
|
-
severity: 'low',
|
|
368
|
-
category: this.category,
|
|
369
|
-
rule: 'AI_API_FRAMEWORK_ROUTES',
|
|
370
|
-
title: `${fr.name}: ${count} Route(s) Discovered${bolaCount ? `, ${bolaCount}+ with object-id segments` : ''}`,
|
|
371
|
-
description: `Framework route discovery found ${count} routes (${sample.join(', ')}...). ${bolaCount ? 'Object-id segments present — check BOLA.' : ''}`,
|
|
372
|
-
matched: sample.join(', '),
|
|
373
|
-
confidence: 'high',
|
|
374
|
-
cwe: 'CWE-1059',
|
|
375
|
-
owasp: 'ASI02',
|
|
376
|
-
fix: 'Map every route to an auth requirement. Flag admin/internal routes for review.',
|
|
377
|
-
}));
|
|
378
|
-
}
|
|
379
|
-
}
|
|
380
|
-
|
|
381
|
-
// ── Lane 4: AI data pipelines & eval harnesses ─────────────────────────
|
|
382
|
-
const lane4Files = files.filter(f => {
|
|
383
|
-
const rel = path.relative(rootPath, f).replace(/\\/g, '/');
|
|
384
|
-
const isDataPipeline = /(?:dataset|loader|load_data|pipeline|hf_|hugging|train|preprocess|ingest)/i.test(rel)
|
|
385
|
-
|| /(?:dataset_infos\.json|config\.json|dataset\.py|\.py$|\.js$|\.json$|\.yaml$|\.yml$)/i.test(path.basename(f));
|
|
386
|
-
const isEvalHarness = /(?:eval|harness|benchmark|sandbox|exploitgym|ctf|scaffold)/i.test(rel)
|
|
387
|
-
|| /eval.*\.(?:yaml|yml|json|py|toml|cfg)$/i.test(path.basename(f));
|
|
388
|
-
return (isDataPipeline || isEvalHarness) && !/(?:node_modules|__tests__|test_|\.test\.)/i.test(rel);
|
|
389
|
-
});
|
|
390
|
-
|
|
391
|
-
for (const file of lane4Files) {
|
|
392
|
-
const content = read(file);
|
|
393
|
-
if (!content) continue;
|
|
394
|
-
const rel = path.relative(rootPath, file).replace(/\\/g, '/');
|
|
395
|
-
|
|
396
|
-
// Remote-code dataset loader
|
|
397
|
-
const dataset_remote_loader_line = DATASET_REMOTE_LOADER.test(content) ? realMatchLine(content, DATASET_REMOTE_LOADER) : 0;
|
|
398
|
-
if (dataset_remote_loader_line) {
|
|
399
|
-
findings.push(createFinding({
|
|
400
|
-
file,
|
|
401
|
-
line: dataset_remote_loader_line,
|
|
402
|
-
severity: 'critical',
|
|
403
|
-
category: this.category,
|
|
404
|
-
rule: 'AI_DATASET_REMOTE_LOADER',
|
|
405
|
-
title: 'Dataset Loader Fetches and Executes Remote Content',
|
|
406
|
-
description: `Dataset pipeline (${rel}) fetches remote content and executes it (exec/eval/pickle.loads). This is the exact initial-access vector used against Hugging Face in July 2026 — a malicious dataset abused the remote-code dataset loader path to run code on a processing worker. Loading an untrusted dataset is equivalent to running its code.`,
|
|
407
|
-
matched: 'remote fetch → code execution',
|
|
408
|
-
confidence: 'high',
|
|
409
|
-
cwe: 'CWE-502',
|
|
410
|
-
owasp: 'ASI05',
|
|
411
|
-
fix: 'Never load datasets with trust_remote_code from unverified sources. Verify dataset provenance, pin revisions/commits, and load inside an isolated worker with no credentials.',
|
|
412
|
-
}));
|
|
413
|
-
}
|
|
414
|
-
|
|
415
|
-
// Unsafe trust_remote_code flag (
|
|
416
|
-
const unsafe_dataset_flag_line = UNSAFE_DATASET_FLAG.test(content) ? realMatchLine(content, UNSAFE_DATASET_FLAG) : 0;
|
|
417
|
-
if (unsafe_dataset_flag_line) {
|
|
418
|
-
findings.push(createFinding({
|
|
419
|
-
file,
|
|
420
|
-
line: unsafe_dataset_flag_line,
|
|
421
|
-
severity: 'high',
|
|
422
|
-
category: this.category,
|
|
423
|
-
rule: 'AI_DATASET_TRUST_REMOTE_CODE',
|
|
424
|
-
title: 'Dataset/Model Loading with trust_remote_code',
|
|
425
|
-
description: `Data pipeline enables trust_remote_code — executing arbitrary code shipped with a dataset/model repo. Multiple 2025-26 incidents (Hugging Face pickle malware, FaceHugger Diffusers bypass, the July 2026 HF breach) abuse this trust boundary for RCE.`,
|
|
426
|
-
matched: 'trust_remote_code=True',
|
|
427
|
-
confidence: 'high',
|
|
428
|
-
cwe: 'CWE-502',
|
|
429
|
-
owasp: 'ASI05',
|
|
430
|
-
fix: 'Disable trust_remote_code by default. If required, pin the exact repo revision, verify provenance, and load in a sandboxed worker.',
|
|
431
|
-
}));
|
|
432
|
-
}
|
|
433
|
-
|
|
434
|
-
// Template injection in dataset config
|
|
435
|
-
const dataset_template_injection_line = DATASET_TEMPLATE_INJECTION.test(content) ? realMatchLine(content, DATASET_TEMPLATE_INJECTION) : 0;
|
|
436
|
-
if (dataset_template_injection_line) {
|
|
437
|
-
findings.push(createFinding({
|
|
438
|
-
file,
|
|
439
|
-
line: dataset_template_injection_line,
|
|
440
|
-
severity: 'critical',
|
|
441
|
-
category: this.category,
|
|
442
|
-
rule: 'AI_DATASET_TEMPLATE_INJECTION',
|
|
443
|
-
title: 'Template Injection in Dataset Configuration',
|
|
444
|
-
description: `Dataset config (${rel}) contains a template expression invoking OS/module functions. Template injection in dataset configuration was the second initial-access vector in the July 2026 Hugging Face breach — a malicious dataset config executed code on a processing worker.`,
|
|
445
|
-
matched: '{{ ... }} / ${ ... } with OS call',
|
|
446
|
-
confidence: 'high',
|
|
447
|
-
cwe: 'CWE-1336',
|
|
448
|
-
owasp: 'ASI05',
|
|
449
|
-
fix: 'Never render dataset config templates with expression evaluation. Treat config fields as plain data and reject template delimiters in untrusted inputs.',
|
|
450
|
-
}));
|
|
451
|
-
}
|
|
452
|
-
|
|
453
|
-
// Eval-harness / sandbox misconfigurations
|
|
454
|
-
for (const check of EVAL_HARNESS_RISK) {
|
|
455
|
-
const harness_line = check.regex.test(content) ? realMatchLine(content, check.regex) : 0;
|
|
456
|
-
if (harness_line) {
|
|
457
|
-
const lines = content.split('\n');
|
|
458
|
-
const lineText = lines[harness_line - 1] || '';
|
|
459
|
-
if (lineText.includes('// praxis-ignore') || lineText.includes('# praxis-ignore')) continue;
|
|
460
|
-
findings.push(createFinding({
|
|
461
|
-
file,
|
|
462
|
-
line: harness_line,
|
|
463
|
-
severity: check.severity,
|
|
464
|
-
category: this.category,
|
|
465
|
-
rule: check.rule,
|
|
466
|
-
title: check.title,
|
|
467
|
-
description: check.description,
|
|
468
|
-
matched: check.rule,
|
|
469
|
-
confidence: 'high',
|
|
470
|
-
cwe: 'CWE-284',
|
|
471
|
-
owasp: 'ASI08',
|
|
472
|
-
fix: check.fix,
|
|
473
|
-
}));
|
|
474
|
-
}
|
|
475
|
-
}
|
|
476
|
-
}
|
|
477
|
-
|
|
478
|
-
return findings;
|
|
479
|
-
}
|
|
480
|
-
}
|
|
481
|
-
|
|
482
|
-
export default AiInfraInventoryAgent;
|
|
1
|
+
/**
|
|
2
|
+
* AI Infrastructure Inventory Agent
|
|
3
|
+
* =================================
|
|
4
|
+
*
|
|
5
|
+
* Completes Praxis's view of the AI system beyond application code — the
|
|
6
|
+
* three production layers where AI actually runs:
|
|
7
|
+
*
|
|
8
|
+
* Lane 1 Model gateways — LiteLLM / Portkey / Helicone / Cloudflare
|
|
9
|
+
* AI Gateway / OpenRouter configs. The choke
|
|
10
|
+
* point through which all LLM traffic flows.
|
|
11
|
+
* Lane 2 AI infrastructure — self-hosted LLM runtimes (Ollama, vLLM, TGI,
|
|
12
|
+
* SGLang, Triton, llama.cpp, NIM, ...) in
|
|
13
|
+
* Docker/K8s/Helm/compose/Terraform, and
|
|
14
|
+
* managed AI compute (Bedrock/SageMaker/Vertex).
|
|
15
|
+
* Lane 3 AI API endpoints — OpenAPI specs + framework routes (FastAPI,
|
|
16
|
+
* Flask, Express, Spring, Django) with auth-style
|
|
17
|
+
* capture and {id}-path BOLA candidates.
|
|
18
|
+
* Lane 4 AI data pipelines — dataset loaders that fetch/execute remote
|
|
19
|
+
* content, template injection in dataset configs,
|
|
20
|
+
* and eval-harness / agent-sandbox misconfigs
|
|
21
|
+
* (the July 2026 OpenAI–Hugging Face incident
|
|
22
|
+
* vectors: remote-code dataset loaders, template
|
|
23
|
+
* injection, disabled safety gates, broad egress).
|
|
24
|
+
*
|
|
25
|
+
* Inventory findings are MEDIUM/LOW — they map the attack surface rather
|
|
26
|
+
* than assert a vulnerability. Risk indicators (exposed keys, no-auth
|
|
27
|
+
* runtimes) escalate to HIGH.
|
|
28
|
+
*/
|
|
29
|
+
|
|
30
|
+
import fs from 'fs';
|
|
31
|
+
import path from 'path';
|
|
32
|
+
import { BaseAgent, createFinding, ruleTableLineMask } from './base-agent.js';
|
|
33
|
+
|
|
34
|
+
// =============================================================================
|
|
35
|
+
// LANE 1 — MODEL GATEWAYS
|
|
36
|
+
// =============================================================================
|
|
37
|
+
|
|
38
|
+
const GATEWAY_CHECKS = [
|
|
39
|
+
{
|
|
40
|
+
rule: 'AI_GATEWAY_LITELLM',
|
|
41
|
+
title: 'LiteLLM Proxy Gateway Detected',
|
|
42
|
+
glob: ['**/litellm*.yaml', '**/litellm*.yml', '**/config.yaml', '**/config.yml'],
|
|
43
|
+
regex: /model_list\s*:/i,
|
|
44
|
+
severity: 'medium',
|
|
45
|
+
description: 'A LiteLLM proxy config with model_list routes LLM traffic. The proxy is the choke point: a misconfiguration (open admin, missing master key, unverified model routes) exposes every prompt and response passing through it.',
|
|
46
|
+
fix: 'Review model_list entries and auth (master key, per-key budgets). Pin model versions and set egress controls.',
|
|
47
|
+
},
|
|
48
|
+
{
|
|
49
|
+
rule: 'AI_GATEWAY_PORTKEY',
|
|
50
|
+
title: 'Portkey Gateway Configuration Detected',
|
|
51
|
+
glob: ['**/portkey-config.json', '**/.portkey*'],
|
|
52
|
+
regex: /(?:provider|virtual_keys?|api_key)\s*:/i,
|
|
53
|
+
severity: 'medium',
|
|
54
|
+
description: 'Portkey gateway config found. Portkey configs commonly contain virtual API keys — treat as credential material.',
|
|
55
|
+
fix: 'Never commit virtual keys. Move to secrets manager and rotate any committed value.',
|
|
56
|
+
},
|
|
57
|
+
{
|
|
58
|
+
rule: 'AI_GATEWAY_HELICONE',
|
|
59
|
+
title: 'Helicone Proxy Endpoint Detected',
|
|
60
|
+
regex: /oai\.helicone\.ai|helicone\.ai\/v1/i,
|
|
61
|
+
severity: 'medium',
|
|
62
|
+
description: 'LLM calls route through the Helicone observability proxy — all prompt/response traffic transits a third party.',
|
|
63
|
+
fix: 'Verify the Helicone account and data-retention settings; exclude sensitive prompts from logging.',
|
|
64
|
+
},
|
|
65
|
+
{
|
|
66
|
+
rule: 'AI_GATEWAY_CLOUDFLARE_AI',
|
|
67
|
+
title: 'Cloudflare AI Gateway Detected',
|
|
68
|
+
regex: /gateway\.ai\.cloudflare\.com/i,
|
|
69
|
+
severity: 'low',
|
|
70
|
+
description: 'LLM traffic routes through Cloudflare AI Gateway.',
|
|
71
|
+
fix: 'Review gateway logs retention and per-app egress policy.',
|
|
72
|
+
},
|
|
73
|
+
{
|
|
74
|
+
rule: 'AI_GATEWAY_OPENROUTER',
|
|
75
|
+
title: 'OpenRouter Usage Detected',
|
|
76
|
+
regex: /openrouter\.ai\/api|OPENROUTER_API_KEY/i,
|
|
77
|
+
severity: 'medium',
|
|
78
|
+
description: 'OpenRouter (multi-model gateway) is configured. All prompts route through a third-party aggregator.',
|
|
79
|
+
fix: 'Verify model routing policy and keep API keys out of committed files.',
|
|
80
|
+
},
|
|
81
|
+
];
|
|
82
|
+
|
|
83
|
+
// =============================================================================
|
|
84
|
+
// LANE 2 — AI INFRASTRUCTURE / IaC
|
|
85
|
+
// =============================================================================
|
|
86
|
+
|
|
87
|
+
const RUNTIME_IMAGES = [
|
|
88
|
+
'ollama', 'ghcr.io/ollama', 'vllm', 'vllm/vllm-openai',
|
|
89
|
+
'text-generation-inference', 'ghcr.io/huggingface/text-generation-inference',
|
|
90
|
+
'sglang', 'lmsysorg/sglang', 'triton', 'nvcr.io/nvidia/tritonserver',
|
|
91
|
+
'nvidia-nim', 'nvcr.io/nvidia/nim', 'localai', 'llama.cpp',
|
|
92
|
+
'ghcr.io/ggml-org/llama.cpp', 'lorax', 'aphrodite', 'openllm',
|
|
93
|
+
'xinference', 'ray-llm', 'infinity', 'fastchat',
|
|
94
|
+
];
|
|
95
|
+
const RUNTIME_IMAGE_RE = new RegExp(`(?:FROM|image:)\\s*[\"']?(${RUNTIME_IMAGES.map(i => i.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')).join('|')})`, 'i');
|
|
96
|
+
|
|
97
|
+
const MANAGED_COMPUTE = [
|
|
98
|
+
{ rule: 'AI_MANAGED_BEDROCK', regex: /aws_bedrock_provisioned_model_throughput|aws_bedrock_custom_model/i, title: 'AWS Bedrock Provisioned Model', description: 'Terraform provisions AWS Bedrock model capacity — committed throughput is a billing + model-governance surface.' },
|
|
99
|
+
{ rule: 'AI_MANAGED_SAGEMAKER', regex: /aws_sagemaker_endpoint\s*["']?[a-zA-Z0-9_-]*(?:llm|model|inference|endpoint)/i, title: 'AWS SageMaker LLM Endpoint', description: 'Terraform provisions a SageMaker endpoint for an LLM — verify IAM scope and endpoint auth.' },
|
|
100
|
+
{ rule: 'AI_MANAGED_VERTEX', regex: /google_vertex_ai_endpoint/i, title: 'Google Vertex AI Endpoint', description: 'Terraform provisions a Vertex AI endpoint — verify network policy and model provenance.' },
|
|
101
|
+
];
|
|
102
|
+
|
|
103
|
+
// =============================================================================
|
|
104
|
+
// LANE 3 — AI API ENDPOINTS
|
|
105
|
+
// =============================================================================
|
|
106
|
+
|
|
107
|
+
const FRAMEWORK_ROUTE_RE = [
|
|
108
|
+
{ name: 'FastAPI/Starlette', regex: /@(?:app|router)\.(?:get|post|put|patch|delete)\s*\(\s*["']([^"']+)["']/g },
|
|
109
|
+
{ name: 'Flask', regex: /@(?:app|bp)\.route\s*\(\s*["']([^"']+)["']/g },
|
|
110
|
+
{ name: 'Express', regex: /(?:app|router)\.(?:get|post|put|patch|delete)\s*\(\s*["']([^"']+)["']/g },
|
|
111
|
+
{ name: 'Spring', regex: /@(?:Get|Post|Put|Patch|Delete|Request)Mapping\s*\(\s*(?:value\s*=\s*)?["']([^"']+)["']/g },
|
|
112
|
+
{ name: 'Django', regex: /(?:path|re_path)\s*\(\s*["']([^"']+)["']/g },
|
|
113
|
+
];
|
|
114
|
+
|
|
115
|
+
const BOLA_SEGMENT = /[\/{][^{}\/]*(?:\{|:)[a-zA-Z_][a-zA-Z0-9_]*[\}][^{}\/]*/;
|
|
116
|
+
|
|
117
|
+
// =============================================================================
|
|
118
|
+
// LANE 4 — AI DATA PIPELINES & EVAL HARNESSES
|
|
119
|
+
// =============================================================================
|
|
120
|
+
// Vectors from the July 2026 OpenAI–Hugging Face autonomous-agent incident:
|
|
121
|
+
// remote-code dataset loaders and dataset template injection (initial access),
|
|
122
|
+
// and eval-harness / agent-sandbox misconfigurations (OpenAI side).
|
|
123
|
+
|
|
124
|
+
// Remote-code dataset loader: fetch remote content then exec/eval/pickle-load it
|
|
125
|
+
const DATASET_REMOTE_LOADER = /(?:load_dataset|dataset_from_script|requests\.get|urllib\.request|httpx\.get|fetch\s*\(\s*["'`]https?:\/\/)[\s\S]{0,300}(?:\bexec\s*\(|\beval\s*\(|pickle\.loads|__import__\(|compile\s*\()/i;
|
|
126
|
+
|
|
127
|
+
// Unsafe dataset/model loading flags
|
|
128
|
+
const UNSAFE_DATASET_FLAG = /trust_remote_code\s*=\s*True|trust_remote_code\s*:\s*true/i;
|
|
129
|
+
|
|
130
|
+
// Template injection in dataset configs: template expressions invoking OS/module
|
|
131
|
+
const DATASET_TEMPLATE_INJECTION = /\{\{\s*(?:__import__|os\.|subprocess|exec|eval|__builtins__)[\s\S]{0,120}?\}\}|\$\{\s*(?:__import__|os\.|subprocess|exec|eval|process\.env)[\s\S]{0,120}?\}/i;
|
|
132
|
+
|
|
133
|
+
/**
|
|
134
|
+
* First 1-based line where `re` matches something that is not a rule table
|
|
135
|
+
* describing itself, or 0 when every match is rule-table prose.
|
|
136
|
+
*
|
|
137
|
+
* The lanes below run patterns against whole file contents, so on one of our own
|
|
138
|
+
* rule tables the match is the table's `description:`/`title:` prose naming the
|
|
139
|
+
* very risk being checked. Skipping only those lines keeps real configuration in
|
|
140
|
+
* the same file reportable, and keeps the reported line honest.
|
|
141
|
+
*/
|
|
142
|
+
function realMatchLine(content, re) {
|
|
143
|
+
const mask = ruleTableLineMask(content.split('\n'));
|
|
144
|
+
if (!mask) {
|
|
145
|
+
re.lastIndex = 0;
|
|
146
|
+
const first = re.exec(content);
|
|
147
|
+
return first ? content.slice(0, first.index).split('\n').length : 0;
|
|
148
|
+
}
|
|
149
|
+
re.lastIndex = 0;
|
|
150
|
+
let m;
|
|
151
|
+
while ((m = re.exec(content)) !== null) {
|
|
152
|
+
const lineIdx = content.slice(0, m.index).split('\n').length - 1;
|
|
153
|
+
if (!mask.has(lineIdx)) return lineIdx + 1;
|
|
154
|
+
if (!re.global) break;
|
|
155
|
+
if (m.index === re.lastIndex) re.lastIndex++;
|
|
156
|
+
}
|
|
157
|
+
return 0;
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
// Eval-harness / agent-sandbox: disabled safety gates, broad tool scope, egress
|
|
161
|
+
const EVAL_HARNESS_RISK = [
|
|
162
|
+
{
|
|
163
|
+
rule: 'AI_EVAL_HARNESS_DISABLED_GUARDRAILS',
|
|
164
|
+
title: 'Eval Harness Disables Safety Guardrails',
|
|
165
|
+
regex: /(?:cyber_refusals|safety_refusals|refusals|guardrails|safety_gates|harm_belen|filters)\s*[:=]\s*(?:disabled|false|off|none|0)/i,
|
|
166
|
+
severity: 'high',
|
|
167
|
+
description: 'An evaluation harness or agent-sandbox config disables model safety guardrails. This is the configuration class behind the OpenAI–Hugging Face July 2026 incident (models evaluated with cyber refusals disabled), and enables unchecked autonomous behavior during evaluation.',
|
|
168
|
+
fix: 'Keep safety guardrails enabled during evaluation, or gate their removal behind explicit containment controls (no internet, isolated egress).',
|
|
169
|
+
},
|
|
170
|
+
{
|
|
171
|
+
rule: 'AI_EVAL_HARNESS_BROAD_TOOL_SCOPE',
|
|
172
|
+
title: 'Eval Harness Grants Broad Tool/Shell Scope',
|
|
173
|
+
regex: /(?:tools?|tool_scope|permissions|allowed_commands)\s*[:=]\s*[\[{][\s\S]{0,200}(?:shell|code_exec|exec_command|bash|subprocess|registry_proxy|all_tools)/i, // praxis-ignore AGENT_TOOL_SHELL_ACCESS — detector definition
|
|
174
|
+
severity: 'medium',
|
|
175
|
+
description: 'An eval harness or agent sandbox grants shell/code-execution tool scope. Combined with network egress this reproduces the sandbox-escape pattern that allowed the July 2026 autonomous agent to pivot from evaluation into production infrastructure.',
|
|
176
|
+
fix: 'Scope evaluation tools to the minimum surface. Shell/code-execution tools should never coexist with internet access in the same sandbox.',
|
|
177
|
+
},
|
|
178
|
+
{
|
|
179
|
+
rule: 'AI_EVAL_HARNESS_BROAD_EGRESS',
|
|
180
|
+
title: 'Eval Harness Allows Unrestricted Network Egress',
|
|
181
|
+
regex: /(?:internet_egress|egress|network_access|allow_remote_downloads|bypass_egress|no_network)\s*[:=]\s*(?:constrained|unrestricted|true|enabled|any|all|false)/i,
|
|
182
|
+
severity: 'high',
|
|
183
|
+
description: 'An eval harness or sandbox allows unrestricted network egress (or explicitly bypasses egress controls). The OpenAI models escaped their sealed evaluation only after finding an egress path — unrestricted egress removes that barrier entirely.',
|
|
184
|
+
fix: 'Default evaluation environments to no-network. If egress is required, allowlist specific endpoints and proxy through a logging gateway.',
|
|
185
|
+
},
|
|
186
|
+
];
|
|
187
|
+
|
|
188
|
+
// =============================================================================
|
|
189
|
+
// AGENT
|
|
190
|
+
// =============================================================================
|
|
191
|
+
|
|
192
|
+
export class AiInfraInventoryAgent extends BaseAgent {
|
|
193
|
+
constructor() {
|
|
194
|
+
super(
|
|
195
|
+
'AiInfraInventoryAgent',
|
|
196
|
+
'Maps the production AI system: model gateways, self-hosted/managed AI infrastructure, and AI API endpoints',
|
|
197
|
+
'llm'
|
|
198
|
+
);
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
shouldRun() {
|
|
202
|
+
return true; // these layers exist in most modern AI-bearing repos
|
|
203
|
+
}
|
|
204
|
+
|
|
205
|
+
async analyze(context) {
|
|
206
|
+
const { files = [], rootPath } = context;
|
|
207
|
+
const findings = [];
|
|
208
|
+
|
|
209
|
+
const read = (f) => {
|
|
210
|
+
try { return fs.readFileSync(f, 'utf8'); } catch { return ''; }
|
|
211
|
+
};
|
|
212
|
+
|
|
213
|
+
// ── Lane 1: model gateways ─────────────────────────────────────────────
|
|
214
|
+
for (const check of GATEWAY_CHECKS) {
|
|
215
|
+
for (const file of files) {
|
|
216
|
+
const rel = path.relative(rootPath, file).replace(/\\/g, '/');
|
|
217
|
+
if (check.glob) {
|
|
218
|
+
const matches = check.glob.some(g => {
|
|
219
|
+
const parts = g.replace(/\*\*\//g, '').split('/');
|
|
220
|
+
const last = parts[parts.length - 1];
|
|
221
|
+
return (last.includes('*') ? new RegExp(`^${last.replace(/\*/g, '.*')}$`).test(path.basename(file)) : last === path.basename(file));
|
|
222
|
+
});
|
|
223
|
+
if (!matches) continue;
|
|
224
|
+
}
|
|
225
|
+
const content = read(file);
|
|
226
|
+
if (!check.regex.test(content)) continue;
|
|
227
|
+
const gwLine = realMatchLine(content, check.regex);
|
|
228
|
+
if (gwLine === 0) continue;
|
|
229
|
+
findings.push(createFinding({
|
|
230
|
+
file,
|
|
231
|
+
line: gwLine,
|
|
232
|
+
severity: check.severity,
|
|
233
|
+
category: this.category,
|
|
234
|
+
rule: check.rule,
|
|
235
|
+
title: check.title,
|
|
236
|
+
description: check.description,
|
|
237
|
+
matched: check.rule,
|
|
238
|
+
confidence: 'high',
|
|
239
|
+
cwe: 'CWE-1357',
|
|
240
|
+
owasp: 'ASI10',
|
|
241
|
+
fix: check.fix,
|
|
242
|
+
}));
|
|
243
|
+
}
|
|
244
|
+
}
|
|
245
|
+
|
|
246
|
+
// ── Lane 2: AI infrastructure ──────────────────────────────────────────
|
|
247
|
+
for (const file of files) {
|
|
248
|
+
const rel = path.relative(rootPath, file).replace(/\\/g, '/');
|
|
249
|
+
const isIaC = /(?:Dockerfile|docker-compose[^/]*\.ya?ml|\.ya?ml|\.tf|k8s|helm|deployment)/i.test(rel);
|
|
250
|
+
if (!isIaC) continue;
|
|
251
|
+
const content = read(file);
|
|
252
|
+
if (!content) continue;
|
|
253
|
+
|
|
254
|
+
const runtimeMatch = RUNTIME_IMAGE_RE.exec(content);
|
|
255
|
+
if (runtimeMatch) {
|
|
256
|
+
findings.push(createFinding({
|
|
257
|
+
file,
|
|
258
|
+
line: 1,
|
|
259
|
+
severity: 'medium',
|
|
260
|
+
category: this.category,
|
|
261
|
+
rule: 'AI_INFRA_RUNTIME_DEPLOYMENT',
|
|
262
|
+
title: `Self-Hosted LLM Runtime (${runtimeMatch[1]})`,
|
|
263
|
+
description: `The repo deploys ${runtimeMatch[1]}, a self-hosted LLM runtime. Operational responsibility: the deployment must enforce endpoint auth, egress limits, model provenance, and patching — a publicly reachable unauthenticated inference endpoint is a direct data/model theft surface.`,
|
|
264
|
+
matched: runtimeMatch[1],
|
|
265
|
+
confidence: 'high',
|
|
266
|
+
cwe: 'CWE-306',
|
|
267
|
+
owasp: 'ASI06',
|
|
268
|
+
fix: 'Authenticate the inference endpoint, restrict network egress, pin runtime/model versions, and monitor inference logs.',
|
|
269
|
+
}));
|
|
270
|
+
}
|
|
271
|
+
|
|
272
|
+
for (const mc of MANAGED_COMPUTE) {
|
|
273
|
+
const mcLine = mc.regex.test(content) ? realMatchLine(content, mc.regex) : 0;
|
|
274
|
+
if (mcLine) {
|
|
275
|
+
findings.push(createFinding({
|
|
276
|
+
file,
|
|
277
|
+
line: mcLine,
|
|
278
|
+
severity: 'medium',
|
|
279
|
+
category: this.category,
|
|
280
|
+
rule: mc.rule,
|
|
281
|
+
title: mc.title,
|
|
282
|
+
description: mc.description,
|
|
283
|
+
matched: mc.rule,
|
|
284
|
+
confidence: 'high',
|
|
285
|
+
cwe: 'CWE-1357',
|
|
286
|
+
owasp: 'ASI10',
|
|
287
|
+
fix: 'Review IAM/network policies around the managed AI resource.',
|
|
288
|
+
}));
|
|
289
|
+
}
|
|
290
|
+
}
|
|
291
|
+
}
|
|
292
|
+
|
|
293
|
+
// ── Lane 3: API endpoints ─────────────────────────────────────────────
|
|
294
|
+
for (const file of files) {
|
|
295
|
+
const rel = path.relative(rootPath, file).replace(/\\/g, '/');
|
|
296
|
+
const isSpec = /openapi|swagger/i.test(rel) && /\.(json|ya?ml)$/.test(rel);
|
|
297
|
+
if (isSpec) {
|
|
298
|
+
const content = read(file);
|
|
299
|
+
let spec = null;
|
|
300
|
+
try { spec = JSON.parse(content); } catch { /* yaml specs not parsed — see FRAMEWORK_ROUTE_RE */ }
|
|
301
|
+
if (spec) {
|
|
302
|
+
const paths = spec.paths || {};
|
|
303
|
+
const securitySchemes = spec.components?.securitySchemes || spec.securityDefinitions || {};
|
|
304
|
+
const authStyles = Object.values(securitySchemes).map(s => s.type || s.scheme || 'unknown');
|
|
305
|
+
const routeEntries = [];
|
|
306
|
+
for (const [p, methods] of Object.entries(paths)) {
|
|
307
|
+
for (const m of Object.keys(methods || {})) {
|
|
308
|
+
if (['get', 'post', 'put', 'patch', 'delete'].includes(m.toLowerCase())) routeEntries.push({ path: p, method: m });
|
|
309
|
+
}
|
|
310
|
+
}
|
|
311
|
+
if (routeEntries.length > 0) {
|
|
312
|
+
const bola = routeEntries.filter(r => BOLA_SEGMENT.test(r.path));
|
|
313
|
+
findings.push(createFinding({
|
|
314
|
+
file,
|
|
315
|
+
line: 1,
|
|
316
|
+
severity: 'low',
|
|
317
|
+
category: this.category,
|
|
318
|
+
rule: 'AI_API_SPEC_INVENTORY',
|
|
319
|
+
title: `API Spec: ${routeEntries.length} Routes (auth: ${authStyles.length ? authStyles.join(',') : 'none declared'})`,
|
|
320
|
+
description: `OpenAPI/Swagger spec exposes ${routeEntries.length} routes. Auth styles declared: ${authStyles.length ? authStyles.join(', ') : 'NONE — every route may be unauthenticated'}. This is the attack-surface map for the AI service.`,
|
|
321
|
+
matched: `${routeEntries.length} routes`,
|
|
322
|
+
confidence: 'high',
|
|
323
|
+
cwe: 'CWE-1059',
|
|
324
|
+
owasp: 'ASI02',
|
|
325
|
+
fix: 'Ensure every route has an explicit auth scheme; review the spec for admin/debug routes.',
|
|
326
|
+
}));
|
|
327
|
+
if (bola.length > 0) {
|
|
328
|
+
findings.push(createFinding({
|
|
329
|
+
file,
|
|
330
|
+
line: 1,
|
|
331
|
+
severity: 'medium',
|
|
332
|
+
category: this.category,
|
|
333
|
+
rule: 'AI_API_BOLA_CANDIDATE',
|
|
334
|
+
title: `BOLA Candidates: ${bola.length} Object-ID Path Route(s)`,
|
|
335
|
+
description: `Route(s) with object-id path segments (e.g. ${bola[0].path}) are classic Broken Object Level Authorization candidates — every object access must be ownership-checked.`,
|
|
336
|
+
matched: bola.slice(0, 3).map(r => r.path).join(', '),
|
|
337
|
+
confidence: 'medium',
|
|
338
|
+
cwe: 'CWE-639',
|
|
339
|
+
owasp: 'ASI02',
|
|
340
|
+
fix: 'Verify per-object authorization checks (ownership/tenant scoping) on every flagged route.',
|
|
341
|
+
}));
|
|
342
|
+
}
|
|
343
|
+
}
|
|
344
|
+
}
|
|
345
|
+
continue;
|
|
346
|
+
}
|
|
347
|
+
|
|
348
|
+
const ext = path.extname(file).toLowerCase();
|
|
349
|
+
if (!['.py', '.js', '.ts', '.java'].includes(ext)) continue;
|
|
350
|
+
const content = read(file);
|
|
351
|
+
if (!content) continue;
|
|
352
|
+
|
|
353
|
+
for (const fr of FRAMEWORK_ROUTE_RE) {
|
|
354
|
+
fr.regex.lastIndex = 0;
|
|
355
|
+
let m;
|
|
356
|
+
let count = 0;
|
|
357
|
+
const sample = [];
|
|
358
|
+
while ((m = fr.regex.exec(content)) !== null) {
|
|
359
|
+
count++;
|
|
360
|
+
if (sample.length < 3) sample.push(m[1]);
|
|
361
|
+
}
|
|
362
|
+
if (count === 0) continue;
|
|
363
|
+
const bolaCount = sample.filter(r => BOLA_SEGMENT.test(r)).length;
|
|
364
|
+
findings.push(createFinding({
|
|
365
|
+
file,
|
|
366
|
+
line: 1,
|
|
367
|
+
severity: 'low',
|
|
368
|
+
category: this.category,
|
|
369
|
+
rule: 'AI_API_FRAMEWORK_ROUTES',
|
|
370
|
+
title: `${fr.name}: ${count} Route(s) Discovered${bolaCount ? `, ${bolaCount}+ with object-id segments` : ''}`,
|
|
371
|
+
description: `Framework route discovery found ${count} routes (${sample.join(', ')}...). ${bolaCount ? 'Object-id segments present — check BOLA.' : ''}`,
|
|
372
|
+
matched: sample.join(', '),
|
|
373
|
+
confidence: 'high',
|
|
374
|
+
cwe: 'CWE-1059',
|
|
375
|
+
owasp: 'ASI02',
|
|
376
|
+
fix: 'Map every route to an auth requirement. Flag admin/internal routes for review.',
|
|
377
|
+
}));
|
|
378
|
+
}
|
|
379
|
+
}
|
|
380
|
+
|
|
381
|
+
// ── Lane 4: AI data pipelines & eval harnesses ─────────────────────────
|
|
382
|
+
const lane4Files = files.filter(f => {
|
|
383
|
+
const rel = path.relative(rootPath, f).replace(/\\/g, '/');
|
|
384
|
+
const isDataPipeline = /(?:dataset|loader|load_data|pipeline|hf_|hugging|train|preprocess|ingest)/i.test(rel)
|
|
385
|
+
|| /(?:dataset_infos\.json|config\.json|dataset\.py|\.py$|\.js$|\.json$|\.yaml$|\.yml$)/i.test(path.basename(f));
|
|
386
|
+
const isEvalHarness = /(?:eval|harness|benchmark|sandbox|exploitgym|ctf|scaffold)/i.test(rel)
|
|
387
|
+
|| /eval.*\.(?:yaml|yml|json|py|toml|cfg)$/i.test(path.basename(f));
|
|
388
|
+
return (isDataPipeline || isEvalHarness) && !/(?:node_modules|__tests__|test_|\.test\.)/i.test(rel);
|
|
389
|
+
});
|
|
390
|
+
|
|
391
|
+
for (const file of lane4Files) {
|
|
392
|
+
const content = read(file);
|
|
393
|
+
if (!content) continue;
|
|
394
|
+
const rel = path.relative(rootPath, file).replace(/\\/g, '/');
|
|
395
|
+
|
|
396
|
+
// Remote-code dataset loader
|
|
397
|
+
const dataset_remote_loader_line = DATASET_REMOTE_LOADER.test(content) ? realMatchLine(content, DATASET_REMOTE_LOADER) : 0;
|
|
398
|
+
if (dataset_remote_loader_line) {
|
|
399
|
+
findings.push(createFinding({
|
|
400
|
+
file,
|
|
401
|
+
line: dataset_remote_loader_line,
|
|
402
|
+
severity: 'critical',
|
|
403
|
+
category: this.category,
|
|
404
|
+
rule: 'AI_DATASET_REMOTE_LOADER',
|
|
405
|
+
title: 'Dataset Loader Fetches and Executes Remote Content',
|
|
406
|
+
description: `Dataset pipeline (${rel}) fetches remote content and executes it (exec/eval/pickle.loads). This is the exact initial-access vector used against Hugging Face in July 2026 — a malicious dataset abused the remote-code dataset loader path to run code on a processing worker. Loading an untrusted dataset is equivalent to running its code.`,
|
|
407
|
+
matched: 'remote fetch → code execution',
|
|
408
|
+
confidence: 'high',
|
|
409
|
+
cwe: 'CWE-502',
|
|
410
|
+
owasp: 'ASI05',
|
|
411
|
+
fix: 'Never load datasets with trust_remote_code from unverified sources. Verify dataset provenance, pin revisions/commits, and load inside an isolated worker with no credentials.',
|
|
412
|
+
}));
|
|
413
|
+
}
|
|
414
|
+
|
|
415
|
+
// Unsafe trust_remote_code flag (companion)
|
|
416
|
+
const unsafe_dataset_flag_line = UNSAFE_DATASET_FLAG.test(content) ? realMatchLine(content, UNSAFE_DATASET_FLAG) : 0;
|
|
417
|
+
if (unsafe_dataset_flag_line) {
|
|
418
|
+
findings.push(createFinding({
|
|
419
|
+
file,
|
|
420
|
+
line: unsafe_dataset_flag_line,
|
|
421
|
+
severity: 'high',
|
|
422
|
+
category: this.category,
|
|
423
|
+
rule: 'AI_DATASET_TRUST_REMOTE_CODE',
|
|
424
|
+
title: 'Dataset/Model Loading with trust_remote_code',
|
|
425
|
+
description: `Data pipeline enables trust_remote_code — executing arbitrary code shipped with a dataset/model repo. Multiple 2025-26 incidents (Hugging Face pickle malware, FaceHugger Diffusers bypass, the July 2026 HF breach) abuse this trust boundary for RCE.`,
|
|
426
|
+
matched: 'trust_remote_code=True',
|
|
427
|
+
confidence: 'high',
|
|
428
|
+
cwe: 'CWE-502',
|
|
429
|
+
owasp: 'ASI05',
|
|
430
|
+
fix: 'Disable trust_remote_code by default. If required, pin the exact repo revision, verify provenance, and load in a sandboxed worker.',
|
|
431
|
+
}));
|
|
432
|
+
}
|
|
433
|
+
|
|
434
|
+
// Template injection in dataset config
|
|
435
|
+
const dataset_template_injection_line = DATASET_TEMPLATE_INJECTION.test(content) ? realMatchLine(content, DATASET_TEMPLATE_INJECTION) : 0;
|
|
436
|
+
if (dataset_template_injection_line) {
|
|
437
|
+
findings.push(createFinding({
|
|
438
|
+
file,
|
|
439
|
+
line: dataset_template_injection_line,
|
|
440
|
+
severity: 'critical',
|
|
441
|
+
category: this.category,
|
|
442
|
+
rule: 'AI_DATASET_TEMPLATE_INJECTION',
|
|
443
|
+
title: 'Template Injection in Dataset Configuration',
|
|
444
|
+
description: `Dataset config (${rel}) contains a template expression invoking OS/module functions. Template injection in dataset configuration was the second initial-access vector in the July 2026 Hugging Face breach — a malicious dataset config executed code on a processing worker.`,
|
|
445
|
+
matched: '{{ ... }} / ${ ... } with OS call',
|
|
446
|
+
confidence: 'high',
|
|
447
|
+
cwe: 'CWE-1336',
|
|
448
|
+
owasp: 'ASI05',
|
|
449
|
+
fix: 'Never render dataset config templates with expression evaluation. Treat config fields as plain data and reject template delimiters in untrusted inputs.',
|
|
450
|
+
}));
|
|
451
|
+
}
|
|
452
|
+
|
|
453
|
+
// Eval-harness / sandbox misconfigurations
|
|
454
|
+
for (const check of EVAL_HARNESS_RISK) {
|
|
455
|
+
const harness_line = check.regex.test(content) ? realMatchLine(content, check.regex) : 0;
|
|
456
|
+
if (harness_line) {
|
|
457
|
+
const lines = content.split('\n');
|
|
458
|
+
const lineText = lines[harness_line - 1] || '';
|
|
459
|
+
if (lineText.includes('// praxis-ignore') || lineText.includes('# praxis-ignore')) continue;
|
|
460
|
+
findings.push(createFinding({
|
|
461
|
+
file,
|
|
462
|
+
line: harness_line,
|
|
463
|
+
severity: check.severity,
|
|
464
|
+
category: this.category,
|
|
465
|
+
rule: check.rule,
|
|
466
|
+
title: check.title,
|
|
467
|
+
description: check.description,
|
|
468
|
+
matched: check.rule,
|
|
469
|
+
confidence: 'high',
|
|
470
|
+
cwe: 'CWE-284',
|
|
471
|
+
owasp: 'ASI08',
|
|
472
|
+
fix: check.fix,
|
|
473
|
+
}));
|
|
474
|
+
}
|
|
475
|
+
}
|
|
476
|
+
}
|
|
477
|
+
|
|
478
|
+
return findings;
|
|
479
|
+
}
|
|
480
|
+
}
|
|
481
|
+
|
|
482
|
+
export default AiInfraInventoryAgent;
|