auditagent-toolkit 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- auditagent_toolkit-0.2.0.dist-info/METADATA +453 -0
- auditagent_toolkit-0.2.0.dist-info/RECORD +27 -0
- auditagent_toolkit-0.2.0.dist-info/WHEEL +4 -0
- auditagent_toolkit-0.2.0.dist-info/entry_points.txt +2 -0
- auditagent_toolkit-0.2.0.dist-info/licenses/LICENSE +201 -0
- auditagent_toolkit-0.2.0.dist-info/licenses/NOTICE +8 -0
- auditagent_tools/__init__.py +15 -0
- auditagent_tools/assemble_report.py +54 -0
- auditagent_tools/atomic_output.py +73 -0
- auditagent_tools/binary_handling.py +71 -0
- auditagent_tools/cli.py +171 -0
- auditagent_tools/determinism.py +66 -0
- auditagent_tools/exclusion_policy.py +84 -0
- auditagent_tools/git_state.py +78 -0
- auditagent_tools/hidden_policy.py +85 -0
- auditagent_tools/inventory.py +76 -0
- auditagent_tools/json_contracts.py +70 -0
- auditagent_tools/output_boundary.py +72 -0
- auditagent_tools/permission_handling.py +73 -0
- auditagent_tools/readonly_immutability.py +82 -0
- auditagent_tools/result_coherence.py +67 -0
- auditagent_tools/score_module.py +46 -0
- auditagent_tools/sensitive_data.py +73 -0
- auditagent_tools/size_limits.py +76 -0
- auditagent_tools/symlink_policy.py +87 -0
- auditagent_tools/test_runner.py +70 -0
- auditagent_tools/write_surface.py +80 -0
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
import argparse
|
|
3
|
+
import json
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from datetime import datetime, timezone
|
|
6
|
+
|
|
7
|
+
def audit_result_coherence(scan_result: str | Path, index_result: str | Path) -> dict:
|
|
8
|
+
"""Compara rutas, tamaños, hashes y sumarios de dos resultados JSON."""
|
|
9
|
+
started = datetime.now(timezone.utc).isoformat()
|
|
10
|
+
scan_path = Path(scan_result).expanduser().resolve()
|
|
11
|
+
index_path = Path(index_result).expanduser().resolve()
|
|
12
|
+
scan = json.loads(scan_path.read_text(encoding="utf-8"))
|
|
13
|
+
index = json.loads(index_path.read_text(encoding="utf-8"))
|
|
14
|
+
scan_files = {
|
|
15
|
+
x["path"]: x for x in scan.get("entries", [])
|
|
16
|
+
if isinstance(x, dict) and x.get("kind") == "file" and isinstance(x.get("path"), str)
|
|
17
|
+
}
|
|
18
|
+
index_docs = {
|
|
19
|
+
x["path"]: x for x in index.get("documents", [])
|
|
20
|
+
if isinstance(x, dict) and isinstance(x.get("path"), str)
|
|
21
|
+
}
|
|
22
|
+
scan_paths, index_paths = set(scan_files), set(index_docs)
|
|
23
|
+
only_scan, only_index = sorted(scan_paths - index_paths), sorted(index_paths - scan_paths)
|
|
24
|
+
mismatches = []
|
|
25
|
+
for path in sorted(scan_paths & index_paths):
|
|
26
|
+
for field in ("size_bytes", "sha256"):
|
|
27
|
+
if scan_files[path].get(field) != index_docs[path].get(field):
|
|
28
|
+
mismatches.append({
|
|
29
|
+
"path": path, "field": field,
|
|
30
|
+
"scan": scan_files[path].get(field), "index": index_docs[path].get(field)
|
|
31
|
+
})
|
|
32
|
+
scan_bytes = sum(int(x.get("size_bytes", 0)) for x in scan_files.values())
|
|
33
|
+
index_bytes = sum(int(x.get("size_bytes", 0)) for x in index_docs.values())
|
|
34
|
+
token_sum = sum(int(x.get("token_count", 0)) for x in index_docs.values())
|
|
35
|
+
findings = []
|
|
36
|
+
if only_scan: findings.append({"severity": "MEDIUM", "title": "Solo en escaneo", "detail": ", ".join(only_scan)})
|
|
37
|
+
if only_index: findings.append({"severity": "HIGH", "title": "Solo en índice", "detail": ", ".join(only_index)})
|
|
38
|
+
if mismatches: findings.append({"severity": "HIGH", "title": "Hashes o tamaños divergentes", "detail": f"{len(mismatches)} diferencias."})
|
|
39
|
+
if scan.get("summary", {}).get("total_bytes") != scan_bytes:
|
|
40
|
+
findings.append({"severity": "MEDIUM", "title": "Resumen de bytes del escaneo inconsistente", "detail": f"declarado={scan.get('summary', {}).get('total_bytes')} recalculado={scan_bytes}"})
|
|
41
|
+
if index.get("summary", {}).get("tokens") != token_sum:
|
|
42
|
+
findings.append({"severity": "MEDIUM", "title": "Resumen de tokens inconsistente", "detail": f"declarado={index.get('summary', {}).get('tokens')} recalculado={token_sum}"})
|
|
43
|
+
return {
|
|
44
|
+
"schema_version": "1.0", "module": "result_coherence",
|
|
45
|
+
"status": "FAIL" if any(x["severity"] == "HIGH" for x in findings) else ("WARN" if findings else "PASS"),
|
|
46
|
+
"confidence": "high", "root": str(scan.get("root") or index.get("root") or ""),
|
|
47
|
+
"summary": {
|
|
48
|
+
"scan_files": len(scan_files), "index_documents": len(index_docs),
|
|
49
|
+
"only_scan": len(only_scan), "only_index": len(only_index),
|
|
50
|
+
"mismatches": len(mismatches), "scan_bytes_recalculated": scan_bytes,
|
|
51
|
+
"index_bytes_recalculated": index_bytes, "tokens_recalculated": token_sum
|
|
52
|
+
},
|
|
53
|
+
"findings": findings,
|
|
54
|
+
"evidence": [{"only_scan": only_scan, "only_index": only_index, "mismatches": mismatches}],
|
|
55
|
+
"limitations": ["No valida semántica de tokenización ni conformidad completa con JSON Schema."],
|
|
56
|
+
"started_at": started, "finished_at": datetime.now(timezone.utc).isoformat(),
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
if __name__ == "__main__":
|
|
60
|
+
p = argparse.ArgumentParser(description="Coherencia independiente entre escaneo e índice.")
|
|
61
|
+
p.add_argument("scan_result")
|
|
62
|
+
p.add_argument("index_result")
|
|
63
|
+
p.add_argument("--json-output")
|
|
64
|
+
a = p.parse_args()
|
|
65
|
+
r = audit_result_coherence(a.scan_result, a.index_result)
|
|
66
|
+
s = json.dumps(r, ensure_ascii=False, indent=2)
|
|
67
|
+
Path(a.json_output).write_text(s + "\n", encoding="utf-8") if a.json_output else print(s)
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
import argparse
|
|
3
|
+
import json
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from datetime import datetime, timezone
|
|
6
|
+
|
|
7
|
+
def score_module_result(result_file: str | Path) -> dict:
|
|
8
|
+
"""Puntúa un único resultado modular sin requerir el resto de la auditoría."""
|
|
9
|
+
started = datetime.now(timezone.utc).isoformat()
|
|
10
|
+
path = Path(result_file).expanduser().resolve()
|
|
11
|
+
result = json.loads(path.read_text(encoding="utf-8"))
|
|
12
|
+
base = {"PASS": 100, "WARN": 75, "FAIL": 30, "NOT_RUN": 0, "ERROR": 0}.get(result.get("status"), 0)
|
|
13
|
+
penalties = {"CRITICAL": 40, "HIGH": 25, "MEDIUM": 12, "LOW": 4, "OBSERVATION": 0}
|
|
14
|
+
score = base
|
|
15
|
+
breakdown = []
|
|
16
|
+
for finding in result.get("findings", []):
|
|
17
|
+
severity = finding.get("severity", "OBSERVATION")
|
|
18
|
+
penalty = penalties.get(severity, 0)
|
|
19
|
+
score -= penalty
|
|
20
|
+
breakdown.append({"severity": severity, "penalty": penalty, "title": finding.get("title")})
|
|
21
|
+
confidence_factor = {"high": 1.0, "medium": 0.9, "low": 0.75}.get(result.get("confidence"), 0.75)
|
|
22
|
+
score = max(0, min(100, round(score * confidence_factor)))
|
|
23
|
+
return {
|
|
24
|
+
"schema_version": "1.0", "module": "score_module",
|
|
25
|
+
"status": "PASS", "confidence": "high", "root": str(path),
|
|
26
|
+
"summary": {
|
|
27
|
+
"target_module": result.get("module"),
|
|
28
|
+
"raw_status": result.get("status"),
|
|
29
|
+
"target_confidence": result.get("confidence"),
|
|
30
|
+
"score": score,
|
|
31
|
+
"scale": 100
|
|
32
|
+
},
|
|
33
|
+
"findings": [],
|
|
34
|
+
"evidence": [{"penalty_breakdown": breakdown, "confidence_factor": confidence_factor}],
|
|
35
|
+
"limitations": ["La puntuación mide el resultado de un módulo, no la seguridad total del proyecto."],
|
|
36
|
+
"started_at": started, "finished_at": datetime.now(timezone.utc).isoformat(),
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
if __name__ == "__main__":
|
|
40
|
+
p = argparse.ArgumentParser(description="Puntuación independiente de un módulo.")
|
|
41
|
+
p.add_argument("result_file")
|
|
42
|
+
p.add_argument("--json-output")
|
|
43
|
+
a = p.parse_args()
|
|
44
|
+
r = score_module_result(a.result_file)
|
|
45
|
+
s = json.dumps(r, ensure_ascii=False, indent=2)
|
|
46
|
+
Path(a.json_output).write_text(s + "\n", encoding="utf-8") if a.json_output else print(s)
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
import argparse
|
|
3
|
+
import json
|
|
4
|
+
import re
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from datetime import datetime, timezone
|
|
7
|
+
|
|
8
|
+
def audit_sensitive_data(root: str | Path = ".", max_file_bytes: int = 2_000_000) -> dict:
|
|
9
|
+
"""Busca secretos, credenciales, correos y rutas locales; siempre redacta el valor."""
|
|
10
|
+
started = datetime.now(timezone.utc).isoformat()
|
|
11
|
+
start = Path(root).expanduser().resolve()
|
|
12
|
+
start = start if start.is_dir() else start.parent
|
|
13
|
+
markers = (".git", ".colmena", "pyproject.toml")
|
|
14
|
+
probe = start
|
|
15
|
+
while probe.parent != probe and not any((probe / m).exists() for m in markers):
|
|
16
|
+
probe = probe.parent
|
|
17
|
+
repo = probe if any((probe / m).exists() for m in markers) else start
|
|
18
|
+
skip = {".git", ".venv", "__pycache__", ".pytest_cache", ".mypy_cache", ".ruff_cache"}
|
|
19
|
+
patterns = [
|
|
20
|
+
("PRIVATE_KEY", "HIGH", re.compile(r"-----BEGIN (?:[A-Z ]+ )?PRIVATE KEY-----")),
|
|
21
|
+
("TOKEN", "HIGH", re.compile(r"\b(?:sk-[A-Za-z0-9_-]{16,}|ghp_[A-Za-z0-9]{20,}|github_pat_[A-Za-z0-9_]{20,}|AKIA[0-9A-Z]{16}|AIza[0-9A-Za-z_-]{20,}|xox[baprs]-[A-Za-z0-9-]{10,})\b")),
|
|
22
|
+
("SENSITIVE_ASSIGNMENT", "MEDIUM", re.compile(r"\b(?:api[_-]?key|client[_-]?secret|access[_-]?token|password|passwd|secret)\b\s*[:=]\s*[\"']?[^\"'\s,}]{4,}", re.I)),
|
|
23
|
+
("CREDENTIAL_URL", "HIGH", re.compile(r"\b[a-z][a-z0-9+.-]*://[^/\s:@]+:[^/\s@]+@", re.I)),
|
|
24
|
+
("EMAIL", "LOW", re.compile(r"\b[A-Z0-9._%+-]+@[A-Z0-9.-]+\.[A-Z]{2,}\b", re.I)),
|
|
25
|
+
("LOCAL_PATH", "LOW", re.compile(r"(?:/home/|/Users/)[^/\s\"']+|[A-Za-z]:\\Users\\[^\\\s\"']+", re.I)),
|
|
26
|
+
]
|
|
27
|
+
findings, scanned, skipped = [], 0, 0
|
|
28
|
+
for path in sorted(repo.rglob("*")):
|
|
29
|
+
if not path.is_file() or path.is_symlink():
|
|
30
|
+
continue
|
|
31
|
+
rel = path.relative_to(repo)
|
|
32
|
+
if any(part in skip or part.endswith(".egg-info") for part in rel.parts) or path.suffix == ".pyc":
|
|
33
|
+
continue
|
|
34
|
+
try:
|
|
35
|
+
raw = path.read_bytes()
|
|
36
|
+
except OSError:
|
|
37
|
+
skipped += 1
|
|
38
|
+
continue
|
|
39
|
+
if len(raw) > max_file_bytes or b"\x00" in raw:
|
|
40
|
+
skipped += 1
|
|
41
|
+
continue
|
|
42
|
+
scanned += 1
|
|
43
|
+
text = raw.decode("utf-8", errors="replace")
|
|
44
|
+
for lineno, line in enumerate(text.splitlines(), 1):
|
|
45
|
+
for label, severity, pattern in patterns:
|
|
46
|
+
if pattern.search(line):
|
|
47
|
+
findings.append({
|
|
48
|
+
"severity": severity, "title": label,
|
|
49
|
+
"detail": "Coincidencia redactada; revisar contexto y descartar falsos positivos.",
|
|
50
|
+
"path": rel.as_posix(), "line": lineno, "redacted": True
|
|
51
|
+
})
|
|
52
|
+
high = any(x["severity"] in {"CRITICAL", "HIGH"} for x in findings)
|
|
53
|
+
return {
|
|
54
|
+
"schema_version": "1.0", "module": "sensitive_data",
|
|
55
|
+
"status": "FAIL" if high else ("WARN" if findings else "PASS"),
|
|
56
|
+
"confidence": "medium", "root": str(repo),
|
|
57
|
+
"summary": {"files_scanned": scanned, "files_skipped": skipped, "matches": len(findings)},
|
|
58
|
+
"findings": findings,
|
|
59
|
+
"evidence": [{"redaction": "Los valores coincidentes nunca se incluyen en la salida."}],
|
|
60
|
+
"limitations": ["Las expresiones regulares producen falsos positivos y no sustituyen un escáner de secretos con entropía e historial Git."],
|
|
61
|
+
"started_at": started,
|
|
62
|
+
"finished_at": datetime.now(timezone.utc).isoformat(),
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
if __name__ == "__main__":
|
|
66
|
+
p = argparse.ArgumentParser(description="Escaneo redactado de información sensible.")
|
|
67
|
+
p.add_argument("root", nargs="?", default=".")
|
|
68
|
+
p.add_argument("--max-file-bytes", type=int, default=2_000_000)
|
|
69
|
+
p.add_argument("--json-output")
|
|
70
|
+
a = p.parse_args()
|
|
71
|
+
r = audit_sensitive_data(a.root, a.max_file_bytes)
|
|
72
|
+
s = json.dumps(r, ensure_ascii=False, indent=2)
|
|
73
|
+
Path(a.json_output).write_text(s + "\n", encoding="utf-8") if a.json_output else print(s)
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
import argparse
|
|
3
|
+
import json
|
|
4
|
+
import os
|
|
5
|
+
import shlex
|
|
6
|
+
import subprocess
|
|
7
|
+
import tempfile
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from datetime import datetime, timezone
|
|
10
|
+
|
|
11
|
+
def audit_size_limits(
|
|
12
|
+
root: str | Path,
|
|
13
|
+
request_template: str | Path,
|
|
14
|
+
command_template: str,
|
|
15
|
+
limit: int = 1024,
|
|
16
|
+
option_key: str = "options.max_file_bytes",
|
|
17
|
+
timeout_seconds: int = 120
|
|
18
|
+
) -> dict:
|
|
19
|
+
"""Prueba límite-1, límite y límite+1 sin depender del nombre del proyecto."""
|
|
20
|
+
started = datetime.now(timezone.utc).isoformat()
|
|
21
|
+
repo = Path(root).expanduser().resolve()
|
|
22
|
+
template = Path(request_template).expanduser().resolve()
|
|
23
|
+
def set_dotted(obj: dict, dotted: str, value) -> None:
|
|
24
|
+
cur = obj
|
|
25
|
+
parts = dotted.split(".")
|
|
26
|
+
for part in parts[:-1]: cur = cur.setdefault(part, {})
|
|
27
|
+
cur[parts[-1]] = value
|
|
28
|
+
def collect_paths(obj) -> set[str]:
|
|
29
|
+
found = set()
|
|
30
|
+
if isinstance(obj, dict):
|
|
31
|
+
for k, v in obj.items():
|
|
32
|
+
if k == "path" and isinstance(v, str): found.add(v)
|
|
33
|
+
found |= collect_paths(v)
|
|
34
|
+
elif isinstance(obj, list):
|
|
35
|
+
for v in obj: found |= collect_paths(v)
|
|
36
|
+
return found
|
|
37
|
+
with tempfile.TemporaryDirectory(prefix="auditagent-size-") as temp:
|
|
38
|
+
t = Path(temp); workspace = t / "workspace"; workspace.mkdir()
|
|
39
|
+
sizes = {"below.bin": max(0, limit - 1), "equal.bin": limit, "above.bin": limit + 1}
|
|
40
|
+
for name, size in sizes.items(): (workspace / name).write_bytes(b"a" * size)
|
|
41
|
+
req = json.loads(template.read_text(encoding="utf-8"))
|
|
42
|
+
req["root"] = str(workspace); set_dotted(req, option_key, limit)
|
|
43
|
+
request, output = t / "request.json", t / "output.json"
|
|
44
|
+
request.write_text(json.dumps(req, ensure_ascii=False, indent=2), encoding="utf-8")
|
|
45
|
+
cmd = [x.format(request=str(request), output=str(output), root=str(workspace)) for x in shlex.split(command_template)]
|
|
46
|
+
env = os.environ.copy(); env.update({"PYTHONDONTWRITEBYTECODE": "1", "TMPDIR": temp})
|
|
47
|
+
cp = subprocess.run(cmd, cwd=repo, env=env, text=True, capture_output=True, timeout=timeout_seconds, check=False)
|
|
48
|
+
data = json.loads(output.read_text(encoding="utf-8")) if output.exists() else {}
|
|
49
|
+
paths = collect_paths(data)
|
|
50
|
+
findings = []
|
|
51
|
+
if "below.bin" not in paths: findings.append({"severity": "MEDIUM", "title": "Archivo bajo límite ausente", "detail": "below.bin"})
|
|
52
|
+
if "equal.bin" not in paths: findings.append({"severity": "LOW", "title": "Semántica del límite exacto", "detail": "equal.bin no apareció; confirmar si el límite es inclusivo."})
|
|
53
|
+
return {
|
|
54
|
+
"schema_version": "1.0", "module": "size_limits",
|
|
55
|
+
"status": "FAIL" if any(x["severity"] == "MEDIUM" for x in findings) else ("WARN" if findings else "PASS"),
|
|
56
|
+
"confidence": "medium", "root": str(repo),
|
|
57
|
+
"summary": {"limit": limit, "paths_observed": sorted(paths), "returncode": cp.returncode},
|
|
58
|
+
"findings": findings,
|
|
59
|
+
"evidence": [{"sizes": sizes, "command": cmd, "stderr": cp.stderr[-2000:]}],
|
|
60
|
+
"limitations": ["La presencia de la ruta no siempre indica que el contenido haya sido indexado; revisar campos de estado del formato específico."],
|
|
61
|
+
"started_at": started, "finished_at": datetime.now(timezone.utc).isoformat(),
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
if __name__ == "__main__":
|
|
65
|
+
p = argparse.ArgumentParser(description="Prueba de límites de tamaño.")
|
|
66
|
+
p.add_argument("root")
|
|
67
|
+
p.add_argument("--request", required=True)
|
|
68
|
+
p.add_argument("--command", required=True)
|
|
69
|
+
p.add_argument("--limit", type=int, default=1024)
|
|
70
|
+
p.add_argument("--option-key", default="options.max_file_bytes")
|
|
71
|
+
p.add_argument("--timeout", type=int, default=120)
|
|
72
|
+
p.add_argument("--json-output")
|
|
73
|
+
a = p.parse_args()
|
|
74
|
+
r = audit_size_limits(a.root, a.request, a.command, a.limit, a.option_key, a.timeout)
|
|
75
|
+
s = json.dumps(r, ensure_ascii=False, indent=2)
|
|
76
|
+
Path(a.json_output).write_text(s + "\n", encoding="utf-8") if a.json_output else print(s)
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
import argparse
|
|
3
|
+
import json
|
|
4
|
+
import os
|
|
5
|
+
import shlex
|
|
6
|
+
import subprocess
|
|
7
|
+
import tempfile
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from datetime import datetime, timezone
|
|
10
|
+
|
|
11
|
+
def audit_symlink_policy(
|
|
12
|
+
root: str | Path,
|
|
13
|
+
request_template: str | Path,
|
|
14
|
+
command_template: str,
|
|
15
|
+
option_key: str = "options.follow_symlinks",
|
|
16
|
+
timeout_seconds: int = 120
|
|
17
|
+
) -> dict:
|
|
18
|
+
"""Prueba enlaces internos, externos, rotos y de directorio con follow_symlinks=false."""
|
|
19
|
+
started = datetime.now(timezone.utc).isoformat()
|
|
20
|
+
repo = Path(root).expanduser().resolve()
|
|
21
|
+
template = Path(request_template).expanduser().resolve()
|
|
22
|
+
def set_dotted(obj: dict, dotted: str, value) -> None:
|
|
23
|
+
cur = obj
|
|
24
|
+
parts = dotted.split(".")
|
|
25
|
+
for part in parts[:-1]:
|
|
26
|
+
cur = cur.setdefault(part, {})
|
|
27
|
+
cur[parts[-1]] = value
|
|
28
|
+
def collect_paths(obj) -> set[str]:
|
|
29
|
+
found = set()
|
|
30
|
+
if isinstance(obj, dict):
|
|
31
|
+
for k, v in obj.items():
|
|
32
|
+
if k == "path" and isinstance(v, str):
|
|
33
|
+
found.add(v)
|
|
34
|
+
found |= collect_paths(v)
|
|
35
|
+
elif isinstance(obj, list):
|
|
36
|
+
for v in obj:
|
|
37
|
+
found |= collect_paths(v)
|
|
38
|
+
return found
|
|
39
|
+
with tempfile.TemporaryDirectory(prefix="auditagent-symlink-") as temp:
|
|
40
|
+
t = Path(temp)
|
|
41
|
+
workspace, external = t / "workspace", t / "external"
|
|
42
|
+
workspace.mkdir(); external.mkdir()
|
|
43
|
+
(workspace / "inside.txt").write_text("inside\n", encoding="utf-8")
|
|
44
|
+
(external / "outside.txt").write_text("outside\n", encoding="utf-8")
|
|
45
|
+
(workspace / "link-internal").symlink_to(workspace / "inside.txt")
|
|
46
|
+
(workspace / "link-external").symlink_to(external / "outside.txt")
|
|
47
|
+
(workspace / "link-dir-external").symlink_to(external, target_is_directory=True)
|
|
48
|
+
(workspace / "link-broken").symlink_to(t / "missing.txt")
|
|
49
|
+
req = json.loads(template.read_text(encoding="utf-8"))
|
|
50
|
+
req["root"] = str(workspace)
|
|
51
|
+
set_dotted(req, option_key, False)
|
|
52
|
+
request, output = t / "request.json", t / "output.json"
|
|
53
|
+
request.write_text(json.dumps(req, ensure_ascii=False, indent=2), encoding="utf-8")
|
|
54
|
+
cmd = [x.format(request=str(request), output=str(output), root=str(workspace)) for x in shlex.split(command_template)]
|
|
55
|
+
env = os.environ.copy(); env.update({"PYTHONDONTWRITEBYTECODE": "1", "TMPDIR": temp})
|
|
56
|
+
cp = subprocess.run(cmd, cwd=repo, env=env, text=True, capture_output=True, timeout=timeout_seconds, check=False)
|
|
57
|
+
data = json.loads(output.read_text(encoding="utf-8")) if output.exists() else {}
|
|
58
|
+
paths = collect_paths(data)
|
|
59
|
+
escaped = [p for p in paths if "outside.txt" in p or p.startswith("../")]
|
|
60
|
+
findings = []
|
|
61
|
+
if escaped:
|
|
62
|
+
findings.append({"severity": "HIGH", "title": "Contenido externo alcanzado", "detail": ", ".join(sorted(escaped))})
|
|
63
|
+
if cp.returncode != 0:
|
|
64
|
+
findings.append({"severity": "MEDIUM", "title": "Ejecución fallida", "detail": cp.stderr[-2000:]})
|
|
65
|
+
return {
|
|
66
|
+
"schema_version": "1.0", "module": "symlink_policy",
|
|
67
|
+
"status": "FAIL" if escaped else ("WARN" if cp.returncode else "PASS"),
|
|
68
|
+
"confidence": "medium", "root": str(repo),
|
|
69
|
+
"summary": {"returncode": cp.returncode, "paths_observed": len(paths), "external_paths": len(escaped)},
|
|
70
|
+
"findings": findings,
|
|
71
|
+
"evidence": [{"command": cmd, "paths": sorted(paths), "stdout": cp.stdout[-3000:], "stderr": cp.stderr[-3000:]}],
|
|
72
|
+
"limitations": ["La interpretación depende del formato de salida y no demuestra resistencia a sustitución concurrente de rutas."],
|
|
73
|
+
"started_at": started, "finished_at": datetime.now(timezone.utc).isoformat(),
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
if __name__ == "__main__":
|
|
77
|
+
p = argparse.ArgumentParser(description="Prueba de política de symlinks.")
|
|
78
|
+
p.add_argument("root")
|
|
79
|
+
p.add_argument("--request", required=True)
|
|
80
|
+
p.add_argument("--command", required=True)
|
|
81
|
+
p.add_argument("--option-key", default="options.follow_symlinks")
|
|
82
|
+
p.add_argument("--timeout", type=int, default=120)
|
|
83
|
+
p.add_argument("--json-output")
|
|
84
|
+
a = p.parse_args()
|
|
85
|
+
r = audit_symlink_policy(a.root, a.request, a.command, a.option_key, a.timeout)
|
|
86
|
+
s = json.dumps(r, ensure_ascii=False, indent=2)
|
|
87
|
+
Path(a.json_output).write_text(s + "\n", encoding="utf-8") if a.json_output else print(s)
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
import argparse
|
|
3
|
+
import json
|
|
4
|
+
import os
|
|
5
|
+
import subprocess
|
|
6
|
+
import tempfile
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from datetime import datetime, timezone
|
|
9
|
+
|
|
10
|
+
def audit_test_runner(root: str | Path = ".", timeout_seconds: int = 180) -> dict:
|
|
11
|
+
"""Ejecuta compilación y pruebas con cachés y temporales fuera del repositorio."""
|
|
12
|
+
started = datetime.now(timezone.utc).isoformat()
|
|
13
|
+
start = Path(root).expanduser().resolve()
|
|
14
|
+
start = start if start.is_dir() else start.parent
|
|
15
|
+
markers = (".git", ".colmena", "pyproject.toml")
|
|
16
|
+
probe = start
|
|
17
|
+
while probe.parent != probe and not any((probe / m).exists() for m in markers):
|
|
18
|
+
probe = probe.parent
|
|
19
|
+
repo = probe if any((probe / m).exists() for m in markers) else start
|
|
20
|
+
commands = []
|
|
21
|
+
python_files = [
|
|
22
|
+
p for p in repo.rglob("*.py")
|
|
23
|
+
if not any(part in {".git", ".venv", "__pycache__"} or part.endswith(".egg-info") for part in p.relative_to(repo).parts)
|
|
24
|
+
]
|
|
25
|
+
if python_files:
|
|
26
|
+
commands.append([os.sys.executable, "-m", "py_compile", *[str(p) for p in python_files]])
|
|
27
|
+
if (repo / "tests").exists():
|
|
28
|
+
commands.append([os.sys.executable, "-m", "unittest", "discover", "-s", "tests", "-p", "test_*.py", "-v"])
|
|
29
|
+
results = []
|
|
30
|
+
with tempfile.TemporaryDirectory(prefix="auditagent-tests-") as temp:
|
|
31
|
+
env = os.environ.copy()
|
|
32
|
+
env.update({
|
|
33
|
+
"PYTHONDONTWRITEBYTECODE": "1",
|
|
34
|
+
"PYTHONPYCACHEPREFIX": str(Path(temp) / "pycache"),
|
|
35
|
+
"TMPDIR": temp,
|
|
36
|
+
})
|
|
37
|
+
for cmd in commands:
|
|
38
|
+
try:
|
|
39
|
+
cp = subprocess.run(cmd, cwd=repo, env=env, text=True, capture_output=True, timeout=timeout_seconds, check=False)
|
|
40
|
+
results.append({
|
|
41
|
+
"command": cmd, "returncode": cp.returncode,
|
|
42
|
+
"stdout": cp.stdout[-12000:], "stderr": cp.stderr[-12000:]
|
|
43
|
+
})
|
|
44
|
+
except subprocess.TimeoutExpired:
|
|
45
|
+
results.append({"command": cmd, "returncode": None, "timeout": True})
|
|
46
|
+
failed = [x for x in results if x.get("returncode") not in (0,)]
|
|
47
|
+
return {
|
|
48
|
+
"schema_version": "1.0", "module": "test_runner",
|
|
49
|
+
"status": "FAIL" if failed else ("PASS" if results else "NOT_RUN"),
|
|
50
|
+
"confidence": "high", "root": str(repo),
|
|
51
|
+
"summary": {"commands": len(results), "failed": len(failed), "python_files": len(python_files)},
|
|
52
|
+
"findings": [
|
|
53
|
+
{"severity": "HIGH", "title": "Comando de prueba fallido", "detail": " ".join(map(str, x["command"]))}
|
|
54
|
+
for x in failed
|
|
55
|
+
],
|
|
56
|
+
"evidence": results,
|
|
57
|
+
"limitations": ["No instala dependencias ni ejecuta suites que requieran servicios externos."],
|
|
58
|
+
"started_at": started,
|
|
59
|
+
"finished_at": datetime.now(timezone.utc).isoformat(),
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
if __name__ == "__main__":
|
|
63
|
+
p = argparse.ArgumentParser(description="Compilación y pruebas independientes.")
|
|
64
|
+
p.add_argument("root", nargs="?", default=".")
|
|
65
|
+
p.add_argument("--timeout", type=int, default=180)
|
|
66
|
+
p.add_argument("--json-output")
|
|
67
|
+
a = p.parse_args()
|
|
68
|
+
r = audit_test_runner(a.root, a.timeout)
|
|
69
|
+
s = json.dumps(r, ensure_ascii=False, indent=2)
|
|
70
|
+
Path(a.json_output).write_text(s + "\n", encoding="utf-8") if a.json_output else print(s)
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
import argparse
|
|
3
|
+
import ast
|
|
4
|
+
import json
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from datetime import datetime, timezone
|
|
7
|
+
|
|
8
|
+
def audit_write_surface(root: str | Path = ".") -> dict:
|
|
9
|
+
"""Localiza APIs de escritura en Python y las clasifica sin ejecutar el código."""
|
|
10
|
+
started = datetime.now(timezone.utc).isoformat()
|
|
11
|
+
start = Path(root).expanduser().resolve()
|
|
12
|
+
start = start if start.is_dir() else start.parent
|
|
13
|
+
markers = (".git", ".colmena", "pyproject.toml")
|
|
14
|
+
probe = start
|
|
15
|
+
while probe.parent != probe and not any((probe / m).exists() for m in markers):
|
|
16
|
+
probe = probe.parent
|
|
17
|
+
repo = probe if any((probe / m).exists() for m in markers) else start
|
|
18
|
+
write_names = {
|
|
19
|
+
"write_text", "write_bytes", "mkdir", "unlink", "remove", "rename",
|
|
20
|
+
"replace", "rmdir", "chmod", "chown", "symlink_to", "hardlink_to",
|
|
21
|
+
"touch", "mkstemp", "NamedTemporaryFile", "atomic_write_json"
|
|
22
|
+
}
|
|
23
|
+
findings, parsed, parse_errors = [], 0, []
|
|
24
|
+
for path in sorted(repo.rglob("*.py")):
|
|
25
|
+
rel = path.relative_to(repo)
|
|
26
|
+
if any(part in {".git", ".venv", "__pycache__"} or part.endswith(".egg-info") for part in rel.parts):
|
|
27
|
+
continue
|
|
28
|
+
try:
|
|
29
|
+
tree = ast.parse(path.read_text(encoding="utf-8"), filename=str(path))
|
|
30
|
+
parsed += 1
|
|
31
|
+
except (OSError, UnicodeError, SyntaxError) as exc:
|
|
32
|
+
parse_errors.append({"path": rel.as_posix(), "error": type(exc).__name__})
|
|
33
|
+
continue
|
|
34
|
+
for node in ast.walk(tree):
|
|
35
|
+
if isinstance(node, ast.Call):
|
|
36
|
+
name = None
|
|
37
|
+
if isinstance(node.func, ast.Name):
|
|
38
|
+
name = node.func.id
|
|
39
|
+
elif isinstance(node.func, ast.Attribute):
|
|
40
|
+
name = node.func.attr
|
|
41
|
+
if name in write_names:
|
|
42
|
+
in_tests = any(part.lower().startswith("test") or part == "tests" for part in rel.parts)
|
|
43
|
+
findings.append({
|
|
44
|
+
"severity": "OBSERVATION" if in_tests else "MEDIUM",
|
|
45
|
+
"title": f"API de escritura: {name}",
|
|
46
|
+
"detail": "Requiere revisión contextual; una escritura de salida externa puede ser legítima.",
|
|
47
|
+
"path": rel.as_posix(), "line": getattr(node, "lineno", None)
|
|
48
|
+
})
|
|
49
|
+
if name == "open" and len(node.args) >= 2 and isinstance(node.args[1], ast.Constant):
|
|
50
|
+
mode = str(node.args[1].value)
|
|
51
|
+
if any(x in mode for x in ("w", "a", "x", "+")):
|
|
52
|
+
findings.append({
|
|
53
|
+
"severity": "OBSERVATION" if "tests" in rel.parts else "MEDIUM",
|
|
54
|
+
"title": f"open() con modo {mode!r}",
|
|
55
|
+
"detail": "Revisar el destino y la guarda aplicada.",
|
|
56
|
+
"path": rel.as_posix(), "line": getattr(node, "lineno", None)
|
|
57
|
+
})
|
|
58
|
+
return {
|
|
59
|
+
"schema_version": "1.0", "module": "write_surface",
|
|
60
|
+
"status": "WARN" if findings or parse_errors else "PASS",
|
|
61
|
+
"confidence": "high", "root": str(repo),
|
|
62
|
+
"summary": {"python_files_parsed": parsed, "write_sites": len(findings), "parse_errors": len(parse_errors)},
|
|
63
|
+
"findings": findings + [
|
|
64
|
+
{"severity": "LOW", "title": "Archivo no analizado", "detail": x["error"], "path": x["path"]}
|
|
65
|
+
for x in parse_errors
|
|
66
|
+
],
|
|
67
|
+
"evidence": [{"method": "Python AST", "write_api_names": sorted(write_names)}],
|
|
68
|
+
"limitations": ["No detecta escrituras indirectas por librerías externas, llamadas dinámicas, shell o extensiones nativas."],
|
|
69
|
+
"started_at": started,
|
|
70
|
+
"finished_at": datetime.now(timezone.utc).isoformat(),
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
if __name__ == "__main__":
|
|
74
|
+
p = argparse.ArgumentParser(description="Superficie estática de escritura.")
|
|
75
|
+
p.add_argument("root", nargs="?", default=".")
|
|
76
|
+
p.add_argument("--json-output")
|
|
77
|
+
a = p.parse_args()
|
|
78
|
+
r = audit_write_surface(a.root)
|
|
79
|
+
s = json.dumps(r, ensure_ascii=False, indent=2)
|
|
80
|
+
Path(a.json_output).write_text(s + "\n", encoding="utf-8") if a.json_output else print(s)
|