auditagent-toolkit 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,67 @@
1
+ from __future__ import annotations
2
+ import argparse
3
+ import json
4
+ from pathlib import Path
5
+ from datetime import datetime, timezone
6
+
7
+ def audit_result_coherence(scan_result: str | Path, index_result: str | Path) -> dict:
8
+ """Compara rutas, tamaños, hashes y sumarios de dos resultados JSON."""
9
+ started = datetime.now(timezone.utc).isoformat()
10
+ scan_path = Path(scan_result).expanduser().resolve()
11
+ index_path = Path(index_result).expanduser().resolve()
12
+ scan = json.loads(scan_path.read_text(encoding="utf-8"))
13
+ index = json.loads(index_path.read_text(encoding="utf-8"))
14
+ scan_files = {
15
+ x["path"]: x for x in scan.get("entries", [])
16
+ if isinstance(x, dict) and x.get("kind") == "file" and isinstance(x.get("path"), str)
17
+ }
18
+ index_docs = {
19
+ x["path"]: x for x in index.get("documents", [])
20
+ if isinstance(x, dict) and isinstance(x.get("path"), str)
21
+ }
22
+ scan_paths, index_paths = set(scan_files), set(index_docs)
23
+ only_scan, only_index = sorted(scan_paths - index_paths), sorted(index_paths - scan_paths)
24
+ mismatches = []
25
+ for path in sorted(scan_paths & index_paths):
26
+ for field in ("size_bytes", "sha256"):
27
+ if scan_files[path].get(field) != index_docs[path].get(field):
28
+ mismatches.append({
29
+ "path": path, "field": field,
30
+ "scan": scan_files[path].get(field), "index": index_docs[path].get(field)
31
+ })
32
+ scan_bytes = sum(int(x.get("size_bytes", 0)) for x in scan_files.values())
33
+ index_bytes = sum(int(x.get("size_bytes", 0)) for x in index_docs.values())
34
+ token_sum = sum(int(x.get("token_count", 0)) for x in index_docs.values())
35
+ findings = []
36
+ if only_scan: findings.append({"severity": "MEDIUM", "title": "Solo en escaneo", "detail": ", ".join(only_scan)})
37
+ if only_index: findings.append({"severity": "HIGH", "title": "Solo en índice", "detail": ", ".join(only_index)})
38
+ if mismatches: findings.append({"severity": "HIGH", "title": "Hashes o tamaños divergentes", "detail": f"{len(mismatches)} diferencias."})
39
+ if scan.get("summary", {}).get("total_bytes") != scan_bytes:
40
+ findings.append({"severity": "MEDIUM", "title": "Resumen de bytes del escaneo inconsistente", "detail": f"declarado={scan.get('summary', {}).get('total_bytes')} recalculado={scan_bytes}"})
41
+ if index.get("summary", {}).get("tokens") != token_sum:
42
+ findings.append({"severity": "MEDIUM", "title": "Resumen de tokens inconsistente", "detail": f"declarado={index.get('summary', {}).get('tokens')} recalculado={token_sum}"})
43
+ return {
44
+ "schema_version": "1.0", "module": "result_coherence",
45
+ "status": "FAIL" if any(x["severity"] == "HIGH" for x in findings) else ("WARN" if findings else "PASS"),
46
+ "confidence": "high", "root": str(scan.get("root") or index.get("root") or ""),
47
+ "summary": {
48
+ "scan_files": len(scan_files), "index_documents": len(index_docs),
49
+ "only_scan": len(only_scan), "only_index": len(only_index),
50
+ "mismatches": len(mismatches), "scan_bytes_recalculated": scan_bytes,
51
+ "index_bytes_recalculated": index_bytes, "tokens_recalculated": token_sum
52
+ },
53
+ "findings": findings,
54
+ "evidence": [{"only_scan": only_scan, "only_index": only_index, "mismatches": mismatches}],
55
+ "limitations": ["No valida semántica de tokenización ni conformidad completa con JSON Schema."],
56
+ "started_at": started, "finished_at": datetime.now(timezone.utc).isoformat(),
57
+ }
58
+
59
+ if __name__ == "__main__":
60
+ p = argparse.ArgumentParser(description="Coherencia independiente entre escaneo e índice.")
61
+ p.add_argument("scan_result")
62
+ p.add_argument("index_result")
63
+ p.add_argument("--json-output")
64
+ a = p.parse_args()
65
+ r = audit_result_coherence(a.scan_result, a.index_result)
66
+ s = json.dumps(r, ensure_ascii=False, indent=2)
67
+ Path(a.json_output).write_text(s + "\n", encoding="utf-8") if a.json_output else print(s)
@@ -0,0 +1,46 @@
1
+ from __future__ import annotations
2
+ import argparse
3
+ import json
4
+ from pathlib import Path
5
+ from datetime import datetime, timezone
6
+
7
+ def score_module_result(result_file: str | Path) -> dict:
8
+ """Puntúa un único resultado modular sin requerir el resto de la auditoría."""
9
+ started = datetime.now(timezone.utc).isoformat()
10
+ path = Path(result_file).expanduser().resolve()
11
+ result = json.loads(path.read_text(encoding="utf-8"))
12
+ base = {"PASS": 100, "WARN": 75, "FAIL": 30, "NOT_RUN": 0, "ERROR": 0}.get(result.get("status"), 0)
13
+ penalties = {"CRITICAL": 40, "HIGH": 25, "MEDIUM": 12, "LOW": 4, "OBSERVATION": 0}
14
+ score = base
15
+ breakdown = []
16
+ for finding in result.get("findings", []):
17
+ severity = finding.get("severity", "OBSERVATION")
18
+ penalty = penalties.get(severity, 0)
19
+ score -= penalty
20
+ breakdown.append({"severity": severity, "penalty": penalty, "title": finding.get("title")})
21
+ confidence_factor = {"high": 1.0, "medium": 0.9, "low": 0.75}.get(result.get("confidence"), 0.75)
22
+ score = max(0, min(100, round(score * confidence_factor)))
23
+ return {
24
+ "schema_version": "1.0", "module": "score_module",
25
+ "status": "PASS", "confidence": "high", "root": str(path),
26
+ "summary": {
27
+ "target_module": result.get("module"),
28
+ "raw_status": result.get("status"),
29
+ "target_confidence": result.get("confidence"),
30
+ "score": score,
31
+ "scale": 100
32
+ },
33
+ "findings": [],
34
+ "evidence": [{"penalty_breakdown": breakdown, "confidence_factor": confidence_factor}],
35
+ "limitations": ["La puntuación mide el resultado de un módulo, no la seguridad total del proyecto."],
36
+ "started_at": started, "finished_at": datetime.now(timezone.utc).isoformat(),
37
+ }
38
+
39
+ if __name__ == "__main__":
40
+ p = argparse.ArgumentParser(description="Puntuación independiente de un módulo.")
41
+ p.add_argument("result_file")
42
+ p.add_argument("--json-output")
43
+ a = p.parse_args()
44
+ r = score_module_result(a.result_file)
45
+ s = json.dumps(r, ensure_ascii=False, indent=2)
46
+ Path(a.json_output).write_text(s + "\n", encoding="utf-8") if a.json_output else print(s)
@@ -0,0 +1,73 @@
1
+ from __future__ import annotations
2
+ import argparse
3
+ import json
4
+ import re
5
+ from pathlib import Path
6
+ from datetime import datetime, timezone
7
+
8
+ def audit_sensitive_data(root: str | Path = ".", max_file_bytes: int = 2_000_000) -> dict:
9
+ """Busca secretos, credenciales, correos y rutas locales; siempre redacta el valor."""
10
+ started = datetime.now(timezone.utc).isoformat()
11
+ start = Path(root).expanduser().resolve()
12
+ start = start if start.is_dir() else start.parent
13
+ markers = (".git", ".colmena", "pyproject.toml")
14
+ probe = start
15
+ while probe.parent != probe and not any((probe / m).exists() for m in markers):
16
+ probe = probe.parent
17
+ repo = probe if any((probe / m).exists() for m in markers) else start
18
+ skip = {".git", ".venv", "__pycache__", ".pytest_cache", ".mypy_cache", ".ruff_cache"}
19
+ patterns = [
20
+ ("PRIVATE_KEY", "HIGH", re.compile(r"-----BEGIN (?:[A-Z ]+ )?PRIVATE KEY-----")),
21
+ ("TOKEN", "HIGH", re.compile(r"\b(?:sk-[A-Za-z0-9_-]{16,}|ghp_[A-Za-z0-9]{20,}|github_pat_[A-Za-z0-9_]{20,}|AKIA[0-9A-Z]{16}|AIza[0-9A-Za-z_-]{20,}|xox[baprs]-[A-Za-z0-9-]{10,})\b")),
22
+ ("SENSITIVE_ASSIGNMENT", "MEDIUM", re.compile(r"\b(?:api[_-]?key|client[_-]?secret|access[_-]?token|password|passwd|secret)\b\s*[:=]\s*[\"']?[^\"'\s,}]{4,}", re.I)),
23
+ ("CREDENTIAL_URL", "HIGH", re.compile(r"\b[a-z][a-z0-9+.-]*://[^/\s:@]+:[^/\s@]+@", re.I)),
24
+ ("EMAIL", "LOW", re.compile(r"\b[A-Z0-9._%+-]+@[A-Z0-9.-]+\.[A-Z]{2,}\b", re.I)),
25
+ ("LOCAL_PATH", "LOW", re.compile(r"(?:/home/|/Users/)[^/\s\"']+|[A-Za-z]:\\Users\\[^\\\s\"']+", re.I)),
26
+ ]
27
+ findings, scanned, skipped = [], 0, 0
28
+ for path in sorted(repo.rglob("*")):
29
+ if not path.is_file() or path.is_symlink():
30
+ continue
31
+ rel = path.relative_to(repo)
32
+ if any(part in skip or part.endswith(".egg-info") for part in rel.parts) or path.suffix == ".pyc":
33
+ continue
34
+ try:
35
+ raw = path.read_bytes()
36
+ except OSError:
37
+ skipped += 1
38
+ continue
39
+ if len(raw) > max_file_bytes or b"\x00" in raw:
40
+ skipped += 1
41
+ continue
42
+ scanned += 1
43
+ text = raw.decode("utf-8", errors="replace")
44
+ for lineno, line in enumerate(text.splitlines(), 1):
45
+ for label, severity, pattern in patterns:
46
+ if pattern.search(line):
47
+ findings.append({
48
+ "severity": severity, "title": label,
49
+ "detail": "Coincidencia redactada; revisar contexto y descartar falsos positivos.",
50
+ "path": rel.as_posix(), "line": lineno, "redacted": True
51
+ })
52
+ high = any(x["severity"] in {"CRITICAL", "HIGH"} for x in findings)
53
+ return {
54
+ "schema_version": "1.0", "module": "sensitive_data",
55
+ "status": "FAIL" if high else ("WARN" if findings else "PASS"),
56
+ "confidence": "medium", "root": str(repo),
57
+ "summary": {"files_scanned": scanned, "files_skipped": skipped, "matches": len(findings)},
58
+ "findings": findings,
59
+ "evidence": [{"redaction": "Los valores coincidentes nunca se incluyen en la salida."}],
60
+ "limitations": ["Las expresiones regulares producen falsos positivos y no sustituyen un escáner de secretos con entropía e historial Git."],
61
+ "started_at": started,
62
+ "finished_at": datetime.now(timezone.utc).isoformat(),
63
+ }
64
+
65
+ if __name__ == "__main__":
66
+ p = argparse.ArgumentParser(description="Escaneo redactado de información sensible.")
67
+ p.add_argument("root", nargs="?", default=".")
68
+ p.add_argument("--max-file-bytes", type=int, default=2_000_000)
69
+ p.add_argument("--json-output")
70
+ a = p.parse_args()
71
+ r = audit_sensitive_data(a.root, a.max_file_bytes)
72
+ s = json.dumps(r, ensure_ascii=False, indent=2)
73
+ Path(a.json_output).write_text(s + "\n", encoding="utf-8") if a.json_output else print(s)
@@ -0,0 +1,76 @@
1
+ from __future__ import annotations
2
+ import argparse
3
+ import json
4
+ import os
5
+ import shlex
6
+ import subprocess
7
+ import tempfile
8
+ from pathlib import Path
9
+ from datetime import datetime, timezone
10
+
11
+ def audit_size_limits(
12
+ root: str | Path,
13
+ request_template: str | Path,
14
+ command_template: str,
15
+ limit: int = 1024,
16
+ option_key: str = "options.max_file_bytes",
17
+ timeout_seconds: int = 120
18
+ ) -> dict:
19
+ """Prueba límite-1, límite y límite+1 sin depender del nombre del proyecto."""
20
+ started = datetime.now(timezone.utc).isoformat()
21
+ repo = Path(root).expanduser().resolve()
22
+ template = Path(request_template).expanduser().resolve()
23
+ def set_dotted(obj: dict, dotted: str, value) -> None:
24
+ cur = obj
25
+ parts = dotted.split(".")
26
+ for part in parts[:-1]: cur = cur.setdefault(part, {})
27
+ cur[parts[-1]] = value
28
+ def collect_paths(obj) -> set[str]:
29
+ found = set()
30
+ if isinstance(obj, dict):
31
+ for k, v in obj.items():
32
+ if k == "path" and isinstance(v, str): found.add(v)
33
+ found |= collect_paths(v)
34
+ elif isinstance(obj, list):
35
+ for v in obj: found |= collect_paths(v)
36
+ return found
37
+ with tempfile.TemporaryDirectory(prefix="auditagent-size-") as temp:
38
+ t = Path(temp); workspace = t / "workspace"; workspace.mkdir()
39
+ sizes = {"below.bin": max(0, limit - 1), "equal.bin": limit, "above.bin": limit + 1}
40
+ for name, size in sizes.items(): (workspace / name).write_bytes(b"a" * size)
41
+ req = json.loads(template.read_text(encoding="utf-8"))
42
+ req["root"] = str(workspace); set_dotted(req, option_key, limit)
43
+ request, output = t / "request.json", t / "output.json"
44
+ request.write_text(json.dumps(req, ensure_ascii=False, indent=2), encoding="utf-8")
45
+ cmd = [x.format(request=str(request), output=str(output), root=str(workspace)) for x in shlex.split(command_template)]
46
+ env = os.environ.copy(); env.update({"PYTHONDONTWRITEBYTECODE": "1", "TMPDIR": temp})
47
+ cp = subprocess.run(cmd, cwd=repo, env=env, text=True, capture_output=True, timeout=timeout_seconds, check=False)
48
+ data = json.loads(output.read_text(encoding="utf-8")) if output.exists() else {}
49
+ paths = collect_paths(data)
50
+ findings = []
51
+ if "below.bin" not in paths: findings.append({"severity": "MEDIUM", "title": "Archivo bajo límite ausente", "detail": "below.bin"})
52
+ if "equal.bin" not in paths: findings.append({"severity": "LOW", "title": "Semántica del límite exacto", "detail": "equal.bin no apareció; confirmar si el límite es inclusivo."})
53
+ return {
54
+ "schema_version": "1.0", "module": "size_limits",
55
+ "status": "FAIL" if any(x["severity"] == "MEDIUM" for x in findings) else ("WARN" if findings else "PASS"),
56
+ "confidence": "medium", "root": str(repo),
57
+ "summary": {"limit": limit, "paths_observed": sorted(paths), "returncode": cp.returncode},
58
+ "findings": findings,
59
+ "evidence": [{"sizes": sizes, "command": cmd, "stderr": cp.stderr[-2000:]}],
60
+ "limitations": ["La presencia de la ruta no siempre indica que el contenido haya sido indexado; revisar campos de estado del formato específico."],
61
+ "started_at": started, "finished_at": datetime.now(timezone.utc).isoformat(),
62
+ }
63
+
64
+ if __name__ == "__main__":
65
+ p = argparse.ArgumentParser(description="Prueba de límites de tamaño.")
66
+ p.add_argument("root")
67
+ p.add_argument("--request", required=True)
68
+ p.add_argument("--command", required=True)
69
+ p.add_argument("--limit", type=int, default=1024)
70
+ p.add_argument("--option-key", default="options.max_file_bytes")
71
+ p.add_argument("--timeout", type=int, default=120)
72
+ p.add_argument("--json-output")
73
+ a = p.parse_args()
74
+ r = audit_size_limits(a.root, a.request, a.command, a.limit, a.option_key, a.timeout)
75
+ s = json.dumps(r, ensure_ascii=False, indent=2)
76
+ Path(a.json_output).write_text(s + "\n", encoding="utf-8") if a.json_output else print(s)
@@ -0,0 +1,87 @@
1
+ from __future__ import annotations
2
+ import argparse
3
+ import json
4
+ import os
5
+ import shlex
6
+ import subprocess
7
+ import tempfile
8
+ from pathlib import Path
9
+ from datetime import datetime, timezone
10
+
11
+ def audit_symlink_policy(
12
+ root: str | Path,
13
+ request_template: str | Path,
14
+ command_template: str,
15
+ option_key: str = "options.follow_symlinks",
16
+ timeout_seconds: int = 120
17
+ ) -> dict:
18
+ """Prueba enlaces internos, externos, rotos y de directorio con follow_symlinks=false."""
19
+ started = datetime.now(timezone.utc).isoformat()
20
+ repo = Path(root).expanduser().resolve()
21
+ template = Path(request_template).expanduser().resolve()
22
+ def set_dotted(obj: dict, dotted: str, value) -> None:
23
+ cur = obj
24
+ parts = dotted.split(".")
25
+ for part in parts[:-1]:
26
+ cur = cur.setdefault(part, {})
27
+ cur[parts[-1]] = value
28
+ def collect_paths(obj) -> set[str]:
29
+ found = set()
30
+ if isinstance(obj, dict):
31
+ for k, v in obj.items():
32
+ if k == "path" and isinstance(v, str):
33
+ found.add(v)
34
+ found |= collect_paths(v)
35
+ elif isinstance(obj, list):
36
+ for v in obj:
37
+ found |= collect_paths(v)
38
+ return found
39
+ with tempfile.TemporaryDirectory(prefix="auditagent-symlink-") as temp:
40
+ t = Path(temp)
41
+ workspace, external = t / "workspace", t / "external"
42
+ workspace.mkdir(); external.mkdir()
43
+ (workspace / "inside.txt").write_text("inside\n", encoding="utf-8")
44
+ (external / "outside.txt").write_text("outside\n", encoding="utf-8")
45
+ (workspace / "link-internal").symlink_to(workspace / "inside.txt")
46
+ (workspace / "link-external").symlink_to(external / "outside.txt")
47
+ (workspace / "link-dir-external").symlink_to(external, target_is_directory=True)
48
+ (workspace / "link-broken").symlink_to(t / "missing.txt")
49
+ req = json.loads(template.read_text(encoding="utf-8"))
50
+ req["root"] = str(workspace)
51
+ set_dotted(req, option_key, False)
52
+ request, output = t / "request.json", t / "output.json"
53
+ request.write_text(json.dumps(req, ensure_ascii=False, indent=2), encoding="utf-8")
54
+ cmd = [x.format(request=str(request), output=str(output), root=str(workspace)) for x in shlex.split(command_template)]
55
+ env = os.environ.copy(); env.update({"PYTHONDONTWRITEBYTECODE": "1", "TMPDIR": temp})
56
+ cp = subprocess.run(cmd, cwd=repo, env=env, text=True, capture_output=True, timeout=timeout_seconds, check=False)
57
+ data = json.loads(output.read_text(encoding="utf-8")) if output.exists() else {}
58
+ paths = collect_paths(data)
59
+ escaped = [p for p in paths if "outside.txt" in p or p.startswith("../")]
60
+ findings = []
61
+ if escaped:
62
+ findings.append({"severity": "HIGH", "title": "Contenido externo alcanzado", "detail": ", ".join(sorted(escaped))})
63
+ if cp.returncode != 0:
64
+ findings.append({"severity": "MEDIUM", "title": "Ejecución fallida", "detail": cp.stderr[-2000:]})
65
+ return {
66
+ "schema_version": "1.0", "module": "symlink_policy",
67
+ "status": "FAIL" if escaped else ("WARN" if cp.returncode else "PASS"),
68
+ "confidence": "medium", "root": str(repo),
69
+ "summary": {"returncode": cp.returncode, "paths_observed": len(paths), "external_paths": len(escaped)},
70
+ "findings": findings,
71
+ "evidence": [{"command": cmd, "paths": sorted(paths), "stdout": cp.stdout[-3000:], "stderr": cp.stderr[-3000:]}],
72
+ "limitations": ["La interpretación depende del formato de salida y no demuestra resistencia a sustitución concurrente de rutas."],
73
+ "started_at": started, "finished_at": datetime.now(timezone.utc).isoformat(),
74
+ }
75
+
76
+ if __name__ == "__main__":
77
+ p = argparse.ArgumentParser(description="Prueba de política de symlinks.")
78
+ p.add_argument("root")
79
+ p.add_argument("--request", required=True)
80
+ p.add_argument("--command", required=True)
81
+ p.add_argument("--option-key", default="options.follow_symlinks")
82
+ p.add_argument("--timeout", type=int, default=120)
83
+ p.add_argument("--json-output")
84
+ a = p.parse_args()
85
+ r = audit_symlink_policy(a.root, a.request, a.command, a.option_key, a.timeout)
86
+ s = json.dumps(r, ensure_ascii=False, indent=2)
87
+ Path(a.json_output).write_text(s + "\n", encoding="utf-8") if a.json_output else print(s)
@@ -0,0 +1,70 @@
1
+ from __future__ import annotations
2
+ import argparse
3
+ import json
4
+ import os
5
+ import subprocess
6
+ import tempfile
7
+ from pathlib import Path
8
+ from datetime import datetime, timezone
9
+
10
+ def audit_test_runner(root: str | Path = ".", timeout_seconds: int = 180) -> dict:
11
+ """Ejecuta compilación y pruebas con cachés y temporales fuera del repositorio."""
12
+ started = datetime.now(timezone.utc).isoformat()
13
+ start = Path(root).expanduser().resolve()
14
+ start = start if start.is_dir() else start.parent
15
+ markers = (".git", ".colmena", "pyproject.toml")
16
+ probe = start
17
+ while probe.parent != probe and not any((probe / m).exists() for m in markers):
18
+ probe = probe.parent
19
+ repo = probe if any((probe / m).exists() for m in markers) else start
20
+ commands = []
21
+ python_files = [
22
+ p for p in repo.rglob("*.py")
23
+ if not any(part in {".git", ".venv", "__pycache__"} or part.endswith(".egg-info") for part in p.relative_to(repo).parts)
24
+ ]
25
+ if python_files:
26
+ commands.append([os.sys.executable, "-m", "py_compile", *[str(p) for p in python_files]])
27
+ if (repo / "tests").exists():
28
+ commands.append([os.sys.executable, "-m", "unittest", "discover", "-s", "tests", "-p", "test_*.py", "-v"])
29
+ results = []
30
+ with tempfile.TemporaryDirectory(prefix="auditagent-tests-") as temp:
31
+ env = os.environ.copy()
32
+ env.update({
33
+ "PYTHONDONTWRITEBYTECODE": "1",
34
+ "PYTHONPYCACHEPREFIX": str(Path(temp) / "pycache"),
35
+ "TMPDIR": temp,
36
+ })
37
+ for cmd in commands:
38
+ try:
39
+ cp = subprocess.run(cmd, cwd=repo, env=env, text=True, capture_output=True, timeout=timeout_seconds, check=False)
40
+ results.append({
41
+ "command": cmd, "returncode": cp.returncode,
42
+ "stdout": cp.stdout[-12000:], "stderr": cp.stderr[-12000:]
43
+ })
44
+ except subprocess.TimeoutExpired:
45
+ results.append({"command": cmd, "returncode": None, "timeout": True})
46
+ failed = [x for x in results if x.get("returncode") not in (0,)]
47
+ return {
48
+ "schema_version": "1.0", "module": "test_runner",
49
+ "status": "FAIL" if failed else ("PASS" if results else "NOT_RUN"),
50
+ "confidence": "high", "root": str(repo),
51
+ "summary": {"commands": len(results), "failed": len(failed), "python_files": len(python_files)},
52
+ "findings": [
53
+ {"severity": "HIGH", "title": "Comando de prueba fallido", "detail": " ".join(map(str, x["command"]))}
54
+ for x in failed
55
+ ],
56
+ "evidence": results,
57
+ "limitations": ["No instala dependencias ni ejecuta suites que requieran servicios externos."],
58
+ "started_at": started,
59
+ "finished_at": datetime.now(timezone.utc).isoformat(),
60
+ }
61
+
62
+ if __name__ == "__main__":
63
+ p = argparse.ArgumentParser(description="Compilación y pruebas independientes.")
64
+ p.add_argument("root", nargs="?", default=".")
65
+ p.add_argument("--timeout", type=int, default=180)
66
+ p.add_argument("--json-output")
67
+ a = p.parse_args()
68
+ r = audit_test_runner(a.root, a.timeout)
69
+ s = json.dumps(r, ensure_ascii=False, indent=2)
70
+ Path(a.json_output).write_text(s + "\n", encoding="utf-8") if a.json_output else print(s)
@@ -0,0 +1,80 @@
1
+ from __future__ import annotations
2
+ import argparse
3
+ import ast
4
+ import json
5
+ from pathlib import Path
6
+ from datetime import datetime, timezone
7
+
8
+ def audit_write_surface(root: str | Path = ".") -> dict:
9
+ """Localiza APIs de escritura en Python y las clasifica sin ejecutar el código."""
10
+ started = datetime.now(timezone.utc).isoformat()
11
+ start = Path(root).expanduser().resolve()
12
+ start = start if start.is_dir() else start.parent
13
+ markers = (".git", ".colmena", "pyproject.toml")
14
+ probe = start
15
+ while probe.parent != probe and not any((probe / m).exists() for m in markers):
16
+ probe = probe.parent
17
+ repo = probe if any((probe / m).exists() for m in markers) else start
18
+ write_names = {
19
+ "write_text", "write_bytes", "mkdir", "unlink", "remove", "rename",
20
+ "replace", "rmdir", "chmod", "chown", "symlink_to", "hardlink_to",
21
+ "touch", "mkstemp", "NamedTemporaryFile", "atomic_write_json"
22
+ }
23
+ findings, parsed, parse_errors = [], 0, []
24
+ for path in sorted(repo.rglob("*.py")):
25
+ rel = path.relative_to(repo)
26
+ if any(part in {".git", ".venv", "__pycache__"} or part.endswith(".egg-info") for part in rel.parts):
27
+ continue
28
+ try:
29
+ tree = ast.parse(path.read_text(encoding="utf-8"), filename=str(path))
30
+ parsed += 1
31
+ except (OSError, UnicodeError, SyntaxError) as exc:
32
+ parse_errors.append({"path": rel.as_posix(), "error": type(exc).__name__})
33
+ continue
34
+ for node in ast.walk(tree):
35
+ if isinstance(node, ast.Call):
36
+ name = None
37
+ if isinstance(node.func, ast.Name):
38
+ name = node.func.id
39
+ elif isinstance(node.func, ast.Attribute):
40
+ name = node.func.attr
41
+ if name in write_names:
42
+ in_tests = any(part.lower().startswith("test") or part == "tests" for part in rel.parts)
43
+ findings.append({
44
+ "severity": "OBSERVATION" if in_tests else "MEDIUM",
45
+ "title": f"API de escritura: {name}",
46
+ "detail": "Requiere revisión contextual; una escritura de salida externa puede ser legítima.",
47
+ "path": rel.as_posix(), "line": getattr(node, "lineno", None)
48
+ })
49
+ if name == "open" and len(node.args) >= 2 and isinstance(node.args[1], ast.Constant):
50
+ mode = str(node.args[1].value)
51
+ if any(x in mode for x in ("w", "a", "x", "+")):
52
+ findings.append({
53
+ "severity": "OBSERVATION" if "tests" in rel.parts else "MEDIUM",
54
+ "title": f"open() con modo {mode!r}",
55
+ "detail": "Revisar el destino y la guarda aplicada.",
56
+ "path": rel.as_posix(), "line": getattr(node, "lineno", None)
57
+ })
58
+ return {
59
+ "schema_version": "1.0", "module": "write_surface",
60
+ "status": "WARN" if findings or parse_errors else "PASS",
61
+ "confidence": "high", "root": str(repo),
62
+ "summary": {"python_files_parsed": parsed, "write_sites": len(findings), "parse_errors": len(parse_errors)},
63
+ "findings": findings + [
64
+ {"severity": "LOW", "title": "Archivo no analizado", "detail": x["error"], "path": x["path"]}
65
+ for x in parse_errors
66
+ ],
67
+ "evidence": [{"method": "Python AST", "write_api_names": sorted(write_names)}],
68
+ "limitations": ["No detecta escrituras indirectas por librerías externas, llamadas dinámicas, shell o extensiones nativas."],
69
+ "started_at": started,
70
+ "finished_at": datetime.now(timezone.utc).isoformat(),
71
+ }
72
+
73
+ if __name__ == "__main__":
74
+ p = argparse.ArgumentParser(description="Superficie estática de escritura.")
75
+ p.add_argument("root", nargs="?", default=".")
76
+ p.add_argument("--json-output")
77
+ a = p.parse_args()
78
+ r = audit_write_surface(a.root)
79
+ s = json.dumps(r, ensure_ascii=False, indent=2)
80
+ Path(a.json_output).write_text(s + "\n", encoding="utf-8") if a.json_output else print(s)