orithos-cli 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- orithos_cli/__init__.py +6 -0
- orithos_cli/agent.py +78 -0
- orithos_cli/agent_wizard.py +255 -0
- orithos_cli/auth.py +21 -0
- orithos_cli/cli.py +82 -0
- orithos_cli/compliance.py +133 -0
- orithos_cli/config.py +114 -0
- orithos_cli/configure.py +103 -0
- orithos_cli/connection.py +69 -0
- orithos_cli/connection_wizard.py +129 -0
- orithos_cli/discovery.py +78 -0
- orithos_cli/graph.py +102 -0
- orithos_cli/guardrail.py +131 -0
- orithos_cli/mcp.py +170 -0
- orithos_cli/output.py +178 -0
- orithos_cli/probes.py +50 -0
- orithos_cli/remediation.py +111 -0
- orithos_cli/runtime.py +100 -0
- orithos_cli/scan.py +758 -0
- orithos_cli/skill.py +64 -0
- orithos_cli/skillscan/__init__.py +37 -0
- orithos_cli/skillscan/checks/__init__.py +28 -0
- orithos_cli/skillscan/checks/credentials.py +112 -0
- orithos_cli/skillscan/checks/iocs.py +95 -0
- orithos_cli/skillscan/checks/manifest.py +156 -0
- orithos_cli/skillscan/checks/network.py +117 -0
- orithos_cli/skillscan/checks/obfuscation.py +130 -0
- orithos_cli/skillscan/checks/permissions.py +108 -0
- orithos_cli/skillscan/checks/shell.py +140 -0
- orithos_cli/skillscan/collect.py +205 -0
- orithos_cli/skillscan/model.py +99 -0
- orithos_cli/skillscan/report.py +130 -0
- orithos_cli/template.py +60 -0
- orithos_cli/verify.py +85 -0
- orithos_cli/wizard.py +314 -0
- orithos_cli-0.1.0.dist-info/METADATA +99 -0
- orithos_cli-0.1.0.dist-info/RECORD +40 -0
- orithos_cli-0.1.0.dist-info/WHEEL +5 -0
- orithos_cli-0.1.0.dist-info/entry_points.txt +2 -0
- orithos_cli-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
"""Check 6 — obfuscation and encoded payloads.
|
|
2
|
+
|
|
3
|
+
Flags base64/hex-encoded blobs that decode to executable content, `eval(atob(`
|
|
4
|
+
style chains, long char-code chains, and dense hex-escape runs.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import base64
|
|
10
|
+
import binascii
|
|
11
|
+
import re
|
|
12
|
+
|
|
13
|
+
from orithos_cli.skillscan.model import ArtifactFile, Finding, line_of, snippet
|
|
14
|
+
|
|
15
|
+
_B64_BLOB_RE = re.compile(r"[A-Za-z0-9+/]{120,}={0,2}")
|
|
16
|
+
_EVAL_ATOB_RE = re.compile(r"eval\s*\(\s*(?:atob|Buffer\.from)\s*\(")
|
|
17
|
+
_EVAL_B64_RE = re.compile(
|
|
18
|
+
r"\b(?:eval|exec)\s*\(\s*(?:base64\.b64decode|b64decode)\s*\("
|
|
19
|
+
)
|
|
20
|
+
_CHARCODE_RE = re.compile(r"String\.fromCharCode\((?:\s*\d+\s*,){6,}")
|
|
21
|
+
_HEX_RUN_RE = re.compile(r"(?:\\x[0-9a-fA-F]{2}){8,}")
|
|
22
|
+
|
|
23
|
+
# Decoded payload content that indicates executable/offensive intent.
|
|
24
|
+
_SUSPICIOUS_DECODED = (
|
|
25
|
+
"http://",
|
|
26
|
+
"https://",
|
|
27
|
+
"/bin/sh",
|
|
28
|
+
"/bin/bash",
|
|
29
|
+
"subprocess",
|
|
30
|
+
"child_process",
|
|
31
|
+
"eval(",
|
|
32
|
+
"exec(",
|
|
33
|
+
"curl ",
|
|
34
|
+
"wget ",
|
|
35
|
+
"import os",
|
|
36
|
+
"require(",
|
|
37
|
+
"powershell",
|
|
38
|
+
"cmd.exe",
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _try_b64(token: str) -> bytes | None:
|
|
43
|
+
padded = token + "=" * (-len(token) % 4)
|
|
44
|
+
try:
|
|
45
|
+
return base64.b64decode(padded, validate=False)
|
|
46
|
+
except (binascii.Error, ValueError):
|
|
47
|
+
return None
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def run(files: list[ArtifactFile]) -> list[Finding]:
|
|
51
|
+
out: list[Finding] = []
|
|
52
|
+
for f in files:
|
|
53
|
+
if f.kind != "code":
|
|
54
|
+
continue
|
|
55
|
+
|
|
56
|
+
m = _EVAL_ATOB_RE.search(f.text)
|
|
57
|
+
if m is None:
|
|
58
|
+
m = _EVAL_B64_RE.search(f.text)
|
|
59
|
+
if m:
|
|
60
|
+
out.append(
|
|
61
|
+
Finding(
|
|
62
|
+
check="obfuscation",
|
|
63
|
+
severity="critical",
|
|
64
|
+
title="eval() over decoded base64",
|
|
65
|
+
detail=(
|
|
66
|
+
f"`{f.path}` decodes base64 and evaluates it in one step — the "
|
|
67
|
+
"executed payload is unreadable to review."
|
|
68
|
+
),
|
|
69
|
+
file=f.path,
|
|
70
|
+
line=line_of(f.text, m.group(0)),
|
|
71
|
+
evidence=snippet(f.text, m.start()),
|
|
72
|
+
)
|
|
73
|
+
)
|
|
74
|
+
|
|
75
|
+
m = _CHARCODE_RE.search(f.text)
|
|
76
|
+
if m:
|
|
77
|
+
out.append(
|
|
78
|
+
Finding(
|
|
79
|
+
check="obfuscation",
|
|
80
|
+
severity="high",
|
|
81
|
+
title="Long String.fromCharCode chain",
|
|
82
|
+
detail=f"`{f.path}` builds a string character-by-character, typical of hidden payloads.",
|
|
83
|
+
file=f.path,
|
|
84
|
+
line=line_of(f.text, m.group(0)),
|
|
85
|
+
evidence=snippet(f.text, m.start()),
|
|
86
|
+
)
|
|
87
|
+
)
|
|
88
|
+
|
|
89
|
+
m = _HEX_RUN_RE.search(f.text)
|
|
90
|
+
if m:
|
|
91
|
+
out.append(
|
|
92
|
+
Finding(
|
|
93
|
+
check="obfuscation",
|
|
94
|
+
severity="high",
|
|
95
|
+
title="Dense hex-escape sequence",
|
|
96
|
+
detail=f"`{f.path}` contains a run of hex-escaped bytes — likely an embedded payload.",
|
|
97
|
+
file=f.path,
|
|
98
|
+
line=line_of(f.text, m.group(0)),
|
|
99
|
+
evidence=snippet(f.text, m.start(), 80),
|
|
100
|
+
)
|
|
101
|
+
)
|
|
102
|
+
|
|
103
|
+
for m in _B64_BLOB_RE.finditer(f.text):
|
|
104
|
+
token = m.group(0)
|
|
105
|
+
decoded = _try_b64(token)
|
|
106
|
+
if decoded is None:
|
|
107
|
+
continue
|
|
108
|
+
try:
|
|
109
|
+
text = decoded.decode("utf-8", errors="strict")
|
|
110
|
+
except UnicodeDecodeError:
|
|
111
|
+
continue
|
|
112
|
+
lowered = text.lower()
|
|
113
|
+
hits = [s for s in _SUSPICIOUS_DECODED if s in lowered]
|
|
114
|
+
if hits:
|
|
115
|
+
out.append(
|
|
116
|
+
Finding(
|
|
117
|
+
check="obfuscation",
|
|
118
|
+
severity="high",
|
|
119
|
+
title="Base64 blob decodes to executable content",
|
|
120
|
+
detail=(
|
|
121
|
+
f"`{f.path}` embeds a base64 blob whose plaintext contains: "
|
|
122
|
+
f"{', '.join(hits[:3])}."
|
|
123
|
+
),
|
|
124
|
+
file=f.path,
|
|
125
|
+
line=line_of(f.text, token[:40]),
|
|
126
|
+
evidence=snippet(f.text, m.start(), 80),
|
|
127
|
+
)
|
|
128
|
+
)
|
|
129
|
+
break # one blob finding per file is enough signal
|
|
130
|
+
return out
|
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
"""Check 2 — permission scope (declared vs implied).
|
|
2
|
+
|
|
3
|
+
Judges mismatches between what a manifest *declares* and what the artifact
|
|
4
|
+
would *need* to run: code that shells out or reaches the network while the
|
|
5
|
+
skill manifest declares no such capability.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import re
|
|
11
|
+
from collections.abc import Iterable
|
|
12
|
+
|
|
13
|
+
from orithos_cli.skillscan.model import ArtifactFile, Finding, line_of, snippet
|
|
14
|
+
|
|
15
|
+
_NETWORK_RE = re.compile(
|
|
16
|
+
r"\b(?:https?://|fetch\(|axios\.|requests\.get|requests\.post|urllib\.request)\b"
|
|
17
|
+
)
|
|
18
|
+
_SHELL_RE = re.compile(
|
|
19
|
+
r"\b(?:subprocess\.|child_process|os\.system|os\.popen|execSync|spawnSync)\b"
|
|
20
|
+
)
|
|
21
|
+
_FILE_WRITE_RE = re.compile(
|
|
22
|
+
r"\b(?:writeFile|write_text|open\([^)]*['\"]w|fs\.write|shutil\.)\b"
|
|
23
|
+
)
|
|
24
|
+
|
|
25
|
+
_CAPABILITY_PATTERNS: list[tuple[str, re.Pattern[str], str]] = [
|
|
26
|
+
("network", _NETWORK_RE, "makes network calls"),
|
|
27
|
+
("shell", _SHELL_RE, "executes shell commands"),
|
|
28
|
+
("filesystem-write", _FILE_WRITE_RE, "writes files"),
|
|
29
|
+
]
|
|
30
|
+
|
|
31
|
+
_TEXT_TOOLS = ("read", "glob", "grep", "ls")
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _declared_tools(fm: dict[str, object]) -> list[str]:
|
|
35
|
+
raw = fm.get("allowed-tools") or fm.get("allowed_tools") or fm.get("tools")
|
|
36
|
+
if isinstance(raw, str):
|
|
37
|
+
return [t.strip().lower() for t in raw.split(",") if t.strip()]
|
|
38
|
+
if isinstance(raw, list):
|
|
39
|
+
return [str(t).lower() for t in raw]
|
|
40
|
+
return []
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _code_files(files: Iterable[ArtifactFile]) -> list[ArtifactFile]:
|
|
44
|
+
return [f for f in files if f.kind == "code"]
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def run(files: list[ArtifactFile]) -> list[Finding]:
|
|
48
|
+
out: list[Finding] = []
|
|
49
|
+
code_files = _code_files(files)
|
|
50
|
+
|
|
51
|
+
for f in files:
|
|
52
|
+
if f.kind != "skill-md":
|
|
53
|
+
continue
|
|
54
|
+
fm = f.frontmatter or {}
|
|
55
|
+
tools = _declared_tools(fm)
|
|
56
|
+
if not tools:
|
|
57
|
+
out.append(
|
|
58
|
+
Finding(
|
|
59
|
+
check="permissions",
|
|
60
|
+
severity="medium",
|
|
61
|
+
title="Skill declares no tool permissions",
|
|
62
|
+
detail=(
|
|
63
|
+
f"`{f.path}` has no `allowed-tools` declaration — the skill's "
|
|
64
|
+
"capability scope is undefined. Declare the minimal tool set."
|
|
65
|
+
),
|
|
66
|
+
file=f.path,
|
|
67
|
+
line=1,
|
|
68
|
+
)
|
|
69
|
+
)
|
|
70
|
+
continue
|
|
71
|
+
tool_blob = " ".join(tools)
|
|
72
|
+
allows_shell = any(
|
|
73
|
+
t in tool_blob for t in ("bash", "shell", "exec", "terminal", "*", "all")
|
|
74
|
+
)
|
|
75
|
+
allows_net = any(
|
|
76
|
+
t in tool_blob
|
|
77
|
+
for t in ("webfetch", "web", "browser", "curl", "http", "*", "all")
|
|
78
|
+
)
|
|
79
|
+
declared_write = any(
|
|
80
|
+
t in tool_blob for t in ("write", "edit", "create", "*", "all")
|
|
81
|
+
) or any(t in _TEXT_TOOLS for t in tools)
|
|
82
|
+
|
|
83
|
+
for cap, pattern, verb in _CAPABILITY_PATTERNS:
|
|
84
|
+
if cap == "network" and (allows_net or "*" in tool_blob):
|
|
85
|
+
continue
|
|
86
|
+
if cap == "shell" and (allows_shell or "*" in tool_blob):
|
|
87
|
+
continue
|
|
88
|
+
if cap == "filesystem-write" and declared_write:
|
|
89
|
+
continue
|
|
90
|
+
for cf in code_files:
|
|
91
|
+
m = pattern.search(cf.text)
|
|
92
|
+
if m and "*" not in tool_blob:
|
|
93
|
+
out.append(
|
|
94
|
+
Finding(
|
|
95
|
+
check="permissions",
|
|
96
|
+
severity="high" if cap == "shell" else "medium",
|
|
97
|
+
title=f"Undeclared capability: code {verb}",
|
|
98
|
+
detail=(
|
|
99
|
+
f"`{cf.path}` {verb}, but `{f.path}` declares only: "
|
|
100
|
+
f"{', '.join(tools)}. Declared scope is narrower than behavior."
|
|
101
|
+
),
|
|
102
|
+
file=cf.path,
|
|
103
|
+
line=line_of(cf.text, m.group(0)),
|
|
104
|
+
evidence=snippet(cf.text, m.start()),
|
|
105
|
+
)
|
|
106
|
+
)
|
|
107
|
+
break
|
|
108
|
+
return out
|
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
"""Check 4 — shell execution and install hooks.
|
|
2
|
+
|
|
3
|
+
Flags command execution primitives, `curl | sh` download-execute chains, and
|
|
4
|
+
npm lifecycle scripts that run at install time.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import json
|
|
10
|
+
import re
|
|
11
|
+
|
|
12
|
+
from orithos_cli.skillscan.model import ArtifactFile, Finding, line_of, snippet
|
|
13
|
+
|
|
14
|
+
_PIPE_TO_SHELL_RE = re.compile(
|
|
15
|
+
r"\b(?:curl|wget)\b[^\n|;&]*\|[^\n]*\b(?:sh|bash|zsh|python3?|node)\b",
|
|
16
|
+
re.IGNORECASE,
|
|
17
|
+
)
|
|
18
|
+
_B64_PIPE_SHELL_RE = re.compile(
|
|
19
|
+
r"base64\s+(?:-d|--decode)[^\n]*\|[^\n]*\b(?:sh|bash|zsh)\b"
|
|
20
|
+
)
|
|
21
|
+
_PY_EXEC_RE = re.compile(
|
|
22
|
+
r"\b(?:subprocess\.(?:run|Popen|call|check_output)|os\.system|os\.popen|pty\.spawn)\b"
|
|
23
|
+
)
|
|
24
|
+
_PY_SHELL_TRUE_RE = re.compile(r"shell\s*=\s*True")
|
|
25
|
+
_JS_EXEC_RE = re.compile(r"\b(?:child_process|execSync|spawnSync|exec\(|spawn\()")
|
|
26
|
+
_NPM_RUN_HOOKS = ("preinstall", "install", "postinstall", "prepare")
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def run(files: list[ArtifactFile]) -> list[Finding]:
|
|
30
|
+
out: list[Finding] = []
|
|
31
|
+
for f in files:
|
|
32
|
+
if f.kind in ("code", "npm-manifest", "mcp-manifest", "python-project"):
|
|
33
|
+
m = _PIPE_TO_SHELL_RE.search(f.text)
|
|
34
|
+
if m:
|
|
35
|
+
out.append(
|
|
36
|
+
Finding(
|
|
37
|
+
check="shell",
|
|
38
|
+
severity="critical",
|
|
39
|
+
title="Download-and-execute chain",
|
|
40
|
+
detail=(
|
|
41
|
+
f"`{f.path}` pipes a download straight into a shell — the fetched "
|
|
42
|
+
"payload is executed without review."
|
|
43
|
+
),
|
|
44
|
+
file=f.path,
|
|
45
|
+
line=line_of(f.text, m.group(0)),
|
|
46
|
+
evidence=snippet(f.text, m.start()),
|
|
47
|
+
)
|
|
48
|
+
)
|
|
49
|
+
m = _B64_PIPE_SHELL_RE.search(f.text)
|
|
50
|
+
if m:
|
|
51
|
+
out.append(
|
|
52
|
+
Finding(
|
|
53
|
+
check="shell",
|
|
54
|
+
severity="critical",
|
|
55
|
+
title="Base64 payload piped into a shell",
|
|
56
|
+
detail=f"`{f.path}` decodes base64 and executes it — payload content is hidden from review.",
|
|
57
|
+
file=f.path,
|
|
58
|
+
line=line_of(f.text, m.group(0)),
|
|
59
|
+
evidence=snippet(f.text, m.start()),
|
|
60
|
+
)
|
|
61
|
+
)
|
|
62
|
+
|
|
63
|
+
if f.kind == "code":
|
|
64
|
+
m = _PY_SHELL_TRUE_RE.search(f.text)
|
|
65
|
+
if m and _PY_EXEC_RE.search(f.text):
|
|
66
|
+
out.append(
|
|
67
|
+
Finding(
|
|
68
|
+
check="shell",
|
|
69
|
+
severity="high",
|
|
70
|
+
title="subprocess with shell=True",
|
|
71
|
+
detail=f"`{f.path}` runs subprocesses through a shell, enabling injection via interpolated input.",
|
|
72
|
+
file=f.path,
|
|
73
|
+
line=line_of(f.text, m.group(0)),
|
|
74
|
+
evidence=snippet(f.text, m.start()),
|
|
75
|
+
)
|
|
76
|
+
)
|
|
77
|
+
for pattern, label in (
|
|
78
|
+
(_PY_EXEC_RE, "python"),
|
|
79
|
+
(_JS_EXEC_RE, "javascript"),
|
|
80
|
+
):
|
|
81
|
+
m = pattern.search(f.text)
|
|
82
|
+
if m:
|
|
83
|
+
out.append(
|
|
84
|
+
Finding(
|
|
85
|
+
check="shell",
|
|
86
|
+
severity="medium",
|
|
87
|
+
title="Shell command execution primitive",
|
|
88
|
+
detail=(
|
|
89
|
+
f"`{f.path}` ({label}) invokes shell/process execution. "
|
|
90
|
+
"Legitimate for some skills — verify what it runs and with which inputs."
|
|
91
|
+
),
|
|
92
|
+
file=f.path,
|
|
93
|
+
line=line_of(f.text, m.group(0)),
|
|
94
|
+
evidence=snippet(f.text, m.start()),
|
|
95
|
+
)
|
|
96
|
+
)
|
|
97
|
+
elif f.kind == "npm-manifest":
|
|
98
|
+
try:
|
|
99
|
+
data = json.loads(f.text)
|
|
100
|
+
except json.JSONDecodeError:
|
|
101
|
+
continue
|
|
102
|
+
scripts = data.get("scripts") if isinstance(data, dict) else None
|
|
103
|
+
if not isinstance(scripts, dict):
|
|
104
|
+
continue
|
|
105
|
+
for hook in _NPM_RUN_HOOKS:
|
|
106
|
+
cmd = scripts.get(hook)
|
|
107
|
+
if not cmd:
|
|
108
|
+
continue
|
|
109
|
+
out.append(
|
|
110
|
+
Finding(
|
|
111
|
+
check="shell",
|
|
112
|
+
severity="high",
|
|
113
|
+
title=f"npm install-time hook: {hook}",
|
|
114
|
+
detail=(
|
|
115
|
+
f"`{f.path}` runs `{hook}` on install. Lifecycle hooks execute "
|
|
116
|
+
"arbitrary commands before the package is ever imported."
|
|
117
|
+
),
|
|
118
|
+
file=f.path,
|
|
119
|
+
line=line_of(f.text, hook),
|
|
120
|
+
evidence=f"{hook}: {str(cmd)[:100]}",
|
|
121
|
+
)
|
|
122
|
+
)
|
|
123
|
+
elif f.kind == "python-project":
|
|
124
|
+
if "cmdclass" in f.text or "build_ext" in f.text:
|
|
125
|
+
out.append(
|
|
126
|
+
Finding(
|
|
127
|
+
check="shell",
|
|
128
|
+
severity="medium",
|
|
129
|
+
title="Python build-time hook",
|
|
130
|
+
detail=(
|
|
131
|
+
f"`{f.path}` defines a custom build hook (cmdclass/build_ext), "
|
|
132
|
+
"which can execute code at install time."
|
|
133
|
+
),
|
|
134
|
+
file=f.path,
|
|
135
|
+
line=line_of(f.text, "cmdclass")
|
|
136
|
+
if "cmdclass" in f.text
|
|
137
|
+
else line_of(f.text, "build_ext"),
|
|
138
|
+
)
|
|
139
|
+
)
|
|
140
|
+
return out
|
|
@@ -0,0 +1,205 @@
|
|
|
1
|
+
"""Artifact collection: discovery, loading, and safe unpacking.
|
|
2
|
+
|
|
3
|
+
The scanner accepts a local path or a URL. URLs are downloaded to a temp
|
|
4
|
+
directory; archives (zip / tar.gz) are extracted with traversal protection
|
|
5
|
+
(zip-slip and symlink members are rejected before extraction).
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import io
|
|
11
|
+
import tarfile
|
|
12
|
+
import tempfile
|
|
13
|
+
import zipfile
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
from typing import Any
|
|
16
|
+
|
|
17
|
+
import httpx
|
|
18
|
+
import yaml
|
|
19
|
+
|
|
20
|
+
from orithos_cli.skillscan.model import ArtifactFile
|
|
21
|
+
|
|
22
|
+
SKIP_DIRS = {
|
|
23
|
+
".git",
|
|
24
|
+
"node_modules",
|
|
25
|
+
".venv",
|
|
26
|
+
"venv",
|
|
27
|
+
"__pycache__",
|
|
28
|
+
"dist",
|
|
29
|
+
"build",
|
|
30
|
+
".next",
|
|
31
|
+
".mypy_cache",
|
|
32
|
+
".ruff_cache",
|
|
33
|
+
}
|
|
34
|
+
CODE_EXTENSIONS = {
|
|
35
|
+
".py": "code",
|
|
36
|
+
".js": "code",
|
|
37
|
+
".mjs": "code",
|
|
38
|
+
".cjs": "code",
|
|
39
|
+
".ts": "code",
|
|
40
|
+
".sh": "code",
|
|
41
|
+
".bash": "code",
|
|
42
|
+
".zsh": "code",
|
|
43
|
+
}
|
|
44
|
+
MAX_FILE_BYTES = 512 * 1024
|
|
45
|
+
MAX_FILES = 2000
|
|
46
|
+
SKILL_MANIFEST_NAMES = {"skill.md", "agent.md"}
|
|
47
|
+
MCP_MANIFEST_HINTS = ("mcpservers", "mcp_servers", "mcp-servers")
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _decode(raw: bytes) -> str:
|
|
51
|
+
return raw.decode("utf-8", errors="replace")
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def parse_frontmatter(text: str) -> dict[str, Any] | None:
|
|
55
|
+
"""Parse YAML frontmatter delimited by leading `---` lines."""
|
|
56
|
+
if not text.startswith("---"):
|
|
57
|
+
return None
|
|
58
|
+
end = text.find("\n---", 3)
|
|
59
|
+
if end < 0:
|
|
60
|
+
return None
|
|
61
|
+
body = text[3:end]
|
|
62
|
+
try:
|
|
63
|
+
loaded = yaml.safe_load(body)
|
|
64
|
+
except yaml.YAMLError:
|
|
65
|
+
return None
|
|
66
|
+
return loaded if isinstance(loaded, dict) else None
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def classify(path: Path, text: str) -> tuple[str, dict[str, Any] | None]:
|
|
70
|
+
"""Return (kind, frontmatter) for a loaded file."""
|
|
71
|
+
name = path.name.lower()
|
|
72
|
+
suffix = path.suffix.lower()
|
|
73
|
+
if suffix == ".md":
|
|
74
|
+
if name in SKILL_MANIFEST_NAMES:
|
|
75
|
+
return "skill-md", parse_frontmatter(text)
|
|
76
|
+
return "text", None
|
|
77
|
+
if name == "package.json":
|
|
78
|
+
return "npm-manifest", None
|
|
79
|
+
if name in ("pyproject.toml", "setup.py", "setup.cfg"):
|
|
80
|
+
return "python-project", None
|
|
81
|
+
if suffix == ".json" and any(h in text.lower() for h in MCP_MANIFEST_HINTS):
|
|
82
|
+
return "mcp-manifest", None
|
|
83
|
+
if suffix in CODE_EXTENSIONS:
|
|
84
|
+
return CODE_EXTENSIONS[suffix], None
|
|
85
|
+
if name.endswith(".md"):
|
|
86
|
+
return "text", None
|
|
87
|
+
return "text", None
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def load_local(root: Path) -> tuple[list[ArtifactFile], list[str], list[str]]:
|
|
91
|
+
"""Walk a local artifact directory. Returns (files, skipped, errors)."""
|
|
92
|
+
files: list[ArtifactFile] = []
|
|
93
|
+
skipped: list[str] = []
|
|
94
|
+
errors: list[str] = []
|
|
95
|
+
|
|
96
|
+
if root.is_file():
|
|
97
|
+
candidates = [root]
|
|
98
|
+
base = root.parent
|
|
99
|
+
else:
|
|
100
|
+
candidates = sorted(p for p in root.rglob("*") if p.is_file())
|
|
101
|
+
base = root
|
|
102
|
+
|
|
103
|
+
count = 0
|
|
104
|
+
for p in candidates:
|
|
105
|
+
if any(part in SKIP_DIRS for part in p.parts):
|
|
106
|
+
continue
|
|
107
|
+
if count >= MAX_FILES:
|
|
108
|
+
skipped.append(f"{p.relative_to(base)} (file cap {MAX_FILES} reached)")
|
|
109
|
+
continue
|
|
110
|
+
rel = str(p.relative_to(base))
|
|
111
|
+
try:
|
|
112
|
+
size = p.stat().st_size
|
|
113
|
+
except OSError as exc:
|
|
114
|
+
errors.append(f"{rel}: {exc}")
|
|
115
|
+
continue
|
|
116
|
+
if size > MAX_FILE_BYTES:
|
|
117
|
+
skipped.append(f"{rel} ({size} bytes > {MAX_FILE_BYTES} cap)")
|
|
118
|
+
continue
|
|
119
|
+
try:
|
|
120
|
+
text = _decode(p.read_bytes())
|
|
121
|
+
except OSError as exc:
|
|
122
|
+
errors.append(f"{rel}: {exc}")
|
|
123
|
+
continue
|
|
124
|
+
if "\x00" in text:
|
|
125
|
+
skipped.append(f"{rel} (binary)")
|
|
126
|
+
continue
|
|
127
|
+
kind, fm = classify(p, text)
|
|
128
|
+
files.append(ArtifactFile(path=rel, kind=kind, text=text, frontmatter=fm))
|
|
129
|
+
count += 1
|
|
130
|
+
|
|
131
|
+
return files, skipped, errors
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def _safe_zip_members(zf: zipfile.ZipFile, dest: Path) -> list[zipfile.ZipInfo]:
|
|
135
|
+
members: list[zipfile.ZipInfo] = []
|
|
136
|
+
dest_resolved = dest.resolve()
|
|
137
|
+
for info in zf.infolist():
|
|
138
|
+
name = info.filename
|
|
139
|
+
if name.startswith("/") or ".." in Path(name).parts:
|
|
140
|
+
raise ValueError(f"unsafe archive member: {name}")
|
|
141
|
+
target = (dest / name).resolve()
|
|
142
|
+
if not str(target).startswith(str(dest_resolved)):
|
|
143
|
+
raise ValueError(f"archive member escapes extraction dir: {name}")
|
|
144
|
+
members.append(info)
|
|
145
|
+
return members
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def _safe_extract_tar(tf: tarfile.TarFile, dest: Path) -> None:
|
|
149
|
+
dest_resolved = dest.resolve()
|
|
150
|
+
for member in tf.getmembers():
|
|
151
|
+
if member.issym() or member.islnk():
|
|
152
|
+
raise ValueError(f"unsafe archive member (link): {member.name}")
|
|
153
|
+
if member.name.startswith("/") or ".." in Path(member.name).parts:
|
|
154
|
+
raise ValueError(f"unsafe archive member: {member.name}")
|
|
155
|
+
target = (dest / member.name).resolve()
|
|
156
|
+
if not str(target).startswith(str(dest_resolved)):
|
|
157
|
+
raise ValueError(f"archive member escapes extraction dir: {member.name}")
|
|
158
|
+
# Python 3.12: data filter re-validates and strips dangerous metadata.
|
|
159
|
+
tf.extractall(dest, filter="data")
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def fetch_url(url: str) -> Path:
|
|
163
|
+
"""Download a URL into a temp dir; unpack archives safely.
|
|
164
|
+
|
|
165
|
+
Returns a path suitable for `load_local`.
|
|
166
|
+
"""
|
|
167
|
+
with httpx.Client(follow_redirects=True, timeout=30.0) as client:
|
|
168
|
+
resp = client.get(url)
|
|
169
|
+
resp.raise_for_status()
|
|
170
|
+
payload = resp.content
|
|
171
|
+
name = url.rsplit("/", 1)[-1].split("?")[0] or "artifact"
|
|
172
|
+
|
|
173
|
+
tmp_root = Path(tempfile.mkdtemp(prefix="orithos-skillscan-"))
|
|
174
|
+
lower = name.lower()
|
|
175
|
+
if lower.endswith(".zip"):
|
|
176
|
+
dest = tmp_root / "artifact"
|
|
177
|
+
dest.mkdir()
|
|
178
|
+
with zipfile.ZipFile(io.BytesIO(payload)) as zf:
|
|
179
|
+
zf.extractall(dest, members=_safe_zip_members(zf, dest))
|
|
180
|
+
return dest
|
|
181
|
+
if lower.endswith((".tar.gz", ".tgz")):
|
|
182
|
+
dest = tmp_root / "artifact"
|
|
183
|
+
dest.mkdir()
|
|
184
|
+
with tarfile.open(fileobj=io.BytesIO(payload), mode="r:gz") as tf:
|
|
185
|
+
_safe_extract_tar(tf, dest)
|
|
186
|
+
return dest
|
|
187
|
+
|
|
188
|
+
target = tmp_root / (name or "artifact.txt")
|
|
189
|
+
target.write_bytes(payload)
|
|
190
|
+
return target
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def resolve_source(source: str) -> Path:
|
|
194
|
+
"""Turn a CLI argument (path or URL) into a local path to scan."""
|
|
195
|
+
if source.startswith(("http://", "https://")):
|
|
196
|
+
return fetch_url(source)
|
|
197
|
+
path = Path(source).expanduser()
|
|
198
|
+
if not path.exists():
|
|
199
|
+
raise FileNotFoundError(f"path does not exist: {source}")
|
|
200
|
+
return path
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def load_artifact(root: Path) -> tuple[list[ArtifactFile], list[str], list[str]]:
|
|
204
|
+
"""Load every scannable file under `root`."""
|
|
205
|
+
return load_local(root)
|
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
"""Core data model for the skill scanner."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass, field
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
SEVERITY_ORDER: dict[str, int] = {
|
|
9
|
+
"critical": 0,
|
|
10
|
+
"high": 1,
|
|
11
|
+
"medium": 2,
|
|
12
|
+
"low": 3,
|
|
13
|
+
"info": 4,
|
|
14
|
+
}
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def severity_at_least(severity: str, floor: str) -> bool:
|
|
18
|
+
"""True when `severity` is at or above the `floor` severity."""
|
|
19
|
+
return SEVERITY_ORDER.get(severity, 99) <= SEVERITY_ORDER.get(floor, 99)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@dataclass
|
|
23
|
+
class Finding:
|
|
24
|
+
"""One scanner finding. `file` is relative to the scanned artifact root."""
|
|
25
|
+
|
|
26
|
+
check: str
|
|
27
|
+
severity: str
|
|
28
|
+
title: str
|
|
29
|
+
detail: str
|
|
30
|
+
file: str
|
|
31
|
+
line: int | None = None
|
|
32
|
+
evidence: str = ""
|
|
33
|
+
|
|
34
|
+
def to_dict(self) -> dict[str, Any]:
|
|
35
|
+
out: dict[str, Any] = {
|
|
36
|
+
"check": self.check,
|
|
37
|
+
"severity": self.severity,
|
|
38
|
+
"title": self.title,
|
|
39
|
+
"detail": self.detail,
|
|
40
|
+
"file": self.file,
|
|
41
|
+
}
|
|
42
|
+
if self.line is not None:
|
|
43
|
+
out["line"] = self.line
|
|
44
|
+
if self.evidence:
|
|
45
|
+
out["evidence"] = self.evidence
|
|
46
|
+
return out
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
@dataclass
|
|
50
|
+
class ArtifactFile:
|
|
51
|
+
"""A text file loaded from the scanned artifact."""
|
|
52
|
+
|
|
53
|
+
path: str
|
|
54
|
+
kind: str # skill-md | npm-manifest | python-project | mcp-manifest | code | text
|
|
55
|
+
text: str
|
|
56
|
+
frontmatter: dict[str, Any] | None = None
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
@dataclass
|
|
60
|
+
class ScanResult:
|
|
61
|
+
"""Aggregated scanner result for one artifact."""
|
|
62
|
+
|
|
63
|
+
source: str
|
|
64
|
+
files_scanned: int
|
|
65
|
+
findings: list[Finding] = field(default_factory=list)
|
|
66
|
+
skipped: list[str] = field(default_factory=list)
|
|
67
|
+
errors: list[str] = field(default_factory=list)
|
|
68
|
+
|
|
69
|
+
def counts(self) -> dict[str, int]:
|
|
70
|
+
counts: dict[str, int] = {}
|
|
71
|
+
for f in self.findings:
|
|
72
|
+
counts[f.severity] = counts.get(f.severity, 0) + 1
|
|
73
|
+
return counts
|
|
74
|
+
|
|
75
|
+
def sorted_findings(self) -> list[Finding]:
|
|
76
|
+
return sorted(
|
|
77
|
+
self.findings,
|
|
78
|
+
key=lambda f: (SEVERITY_ORDER.get(f.severity, 99), f.file, f.line or 0),
|
|
79
|
+
)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def line_of(text: str, needle: str) -> int | None:
|
|
83
|
+
"""1-based line number of the first occurrence of `needle`, or None."""
|
|
84
|
+
idx = text.find(needle)
|
|
85
|
+
if idx < 0:
|
|
86
|
+
return None
|
|
87
|
+
return text[:idx].count("\n") + 1
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def snippet(text: str, idx: int, length: int = 120) -> str:
|
|
91
|
+
"""Short single-line evidence snippet around `idx`."""
|
|
92
|
+
start = text.rfind("\n", 0, idx) + 1
|
|
93
|
+
end = text.find("\n", idx)
|
|
94
|
+
if end < 0:
|
|
95
|
+
end = len(text)
|
|
96
|
+
raw = text[start:end].strip()
|
|
97
|
+
if len(raw) > length:
|
|
98
|
+
raw = raw[: length - 1] + "…"
|
|
99
|
+
return raw
|