zairo 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- zairo/__init__.py +1 -0
- zairo/__main__.py +4 -0
- zairo/_util.py +12 -0
- zairo/analyzer.py +121 -0
- zairo/cli.py +108 -0
- zairo/git_utils.py +131 -0
- zairo/llm_scanner.py +507 -0
- zairo/reporter.py +197 -0
- zairo-0.1.0.dist-info/METADATA +75 -0
- zairo-0.1.0.dist-info/RECORD +14 -0
- zairo-0.1.0.dist-info/WHEEL +5 -0
- zairo-0.1.0.dist-info/entry_points.txt +2 -0
- zairo-0.1.0.dist-info/licenses/LICENSE +21 -0
- zairo-0.1.0.dist-info/top_level.txt +1 -0
zairo/__init__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "0.1.0"
|
zairo/__main__.py
ADDED
zairo/_util.py
ADDED
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
from typing import Any
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
def display_name(name: Any, limit: int = 60) -> str:
|
|
5
|
+
"""Collapses a node name to one short line for log display. Some graph
|
|
6
|
+
nodes (e.g. Trailmark misparsing a chained expression like
|
|
7
|
+
`.map(fn).filter(...)`) end up with a "name" that's actually a chunk of
|
|
8
|
+
raw multi-line source text -- printing that verbatim floods the log."""
|
|
9
|
+
text = " ".join(str(name).split())
|
|
10
|
+
if len(text) > limit:
|
|
11
|
+
text = text[:limit - 1] + "…"
|
|
12
|
+
return text
|
zairo/analyzer.py
ADDED
|
@@ -0,0 +1,121 @@
|
|
|
1
|
+
import os
|
|
2
|
+
from typing import Callable, Dict, List, Optional, Set, Any
|
|
3
|
+
from trailmark.query.api import QueryEngine
|
|
4
|
+
from .git_utils import get_modified_lines
|
|
5
|
+
from ._util import display_name as _display_name
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def analyze_impact(
|
|
9
|
+
repo_path: str,
|
|
10
|
+
depth: int = 1,
|
|
11
|
+
base: str = None,
|
|
12
|
+
target: str = None,
|
|
13
|
+
language: str = "auto",
|
|
14
|
+
log: Optional[Callable[[str], None]] = None,
|
|
15
|
+
) -> Dict[str, Any]:
|
|
16
|
+
"""
|
|
17
|
+
`repo_path` must already be checked out at the state to be indexed: the
|
|
18
|
+
caller is responsible for pointing it at a worktree checked out to
|
|
19
|
+
`target` when diffing two commits, so that node locations/contents line
|
|
20
|
+
up with the line numbers `git diff base target` reports.
|
|
21
|
+
"""
|
|
22
|
+
log = log or (lambda msg: None)
|
|
23
|
+
|
|
24
|
+
analysis_root = os.path.abspath(repo_path)
|
|
25
|
+
|
|
26
|
+
modified_files_lines = get_modified_lines(analysis_root, base, target, log=log)
|
|
27
|
+
log(f"git diff found {len(modified_files_lines)} modified file(s):")
|
|
28
|
+
for f, lines in modified_files_lines.items():
|
|
29
|
+
log(f" {f}: {len(lines)} line(s) changed -> {sorted(lines.keys())}")
|
|
30
|
+
|
|
31
|
+
# Initialize Trailmark
|
|
32
|
+
log(f"Indexing {analysis_root} with Trailmark (language={language})...")
|
|
33
|
+
engine = QueryEngine.from_directory(analysis_root, language=language)
|
|
34
|
+
total_nodes = len(engine._store._graph.nodes)
|
|
35
|
+
total_edges = len(engine._store._graph.edges)
|
|
36
|
+
log(f"Trailmark graph: {total_nodes} node(s), {total_edges} edge(s)")
|
|
37
|
+
|
|
38
|
+
# 1. Identify seed nodes (modified/added)
|
|
39
|
+
seed_nodes = set()
|
|
40
|
+
node_metadata = {}
|
|
41
|
+
|
|
42
|
+
for node_id, node in engine._store._graph.nodes.items():
|
|
43
|
+
node_metadata[node_id] = {
|
|
44
|
+
"id": node_id,
|
|
45
|
+
"name": getattr(node, 'name', node_id),
|
|
46
|
+
"kind": node.kind.value if hasattr(node, 'kind') and hasattr(node.kind, 'value') else str(getattr(node, 'kind', 'unknown')),
|
|
47
|
+
"file": node.location.file_path if getattr(node, 'location', None) else None,
|
|
48
|
+
"start_line": node.location.start_line if getattr(node, 'location', None) else None,
|
|
49
|
+
"end_line": node.location.end_line if getattr(node, 'location', None) else None,
|
|
50
|
+
"complexity": getattr(node, 'cyclomatic_complexity', 0),
|
|
51
|
+
"status": "unchanged" # default
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
if getattr(node, 'location', None) and node.location.file_path in modified_files_lines:
|
|
55
|
+
file_mod_lines = modified_files_lines[node.location.file_path]
|
|
56
|
+
start = node.location.start_line
|
|
57
|
+
end = node.location.end_line
|
|
58
|
+
changed_lines = {ln: text for ln, text in file_mod_lines.items() if start <= ln <= end}
|
|
59
|
+
if changed_lines:
|
|
60
|
+
seed_nodes.add(node_id)
|
|
61
|
+
node_metadata[node_id]["status"] = "modified"
|
|
62
|
+
node_metadata[node_id]["changed_lines"] = changed_lines
|
|
63
|
+
log(f" seed: {_display_name(node_metadata[node_id]['name'])} ({node.location.file_path}:{start}-{end}), {len(changed_lines)} line(s) changed")
|
|
64
|
+
|
|
65
|
+
log(f"Identified {len(seed_nodes)} seed node(s)")
|
|
66
|
+
|
|
67
|
+
# 2. Traverse graph to build subgraph up to `depth`
|
|
68
|
+
subgraph_nodes = set(seed_nodes)
|
|
69
|
+
current_frontier = set(seed_nodes)
|
|
70
|
+
|
|
71
|
+
for hop in range(depth):
|
|
72
|
+
next_frontier = set()
|
|
73
|
+
for edge in engine._store._graph.edges:
|
|
74
|
+
source = edge.source_id
|
|
75
|
+
edge_target = edge.target_id
|
|
76
|
+
|
|
77
|
+
if source in current_frontier and edge_target not in subgraph_nodes:
|
|
78
|
+
next_frontier.add(edge_target)
|
|
79
|
+
subgraph_nodes.add(edge_target)
|
|
80
|
+
elif edge_target in current_frontier and source not in subgraph_nodes:
|
|
81
|
+
next_frontier.add(source)
|
|
82
|
+
subgraph_nodes.add(source)
|
|
83
|
+
|
|
84
|
+
log(f"Hop {hop + 1}/{depth}: added {len(next_frontier)} node(s), frontier now {len(subgraph_nodes)} total")
|
|
85
|
+
current_frontier = next_frontier
|
|
86
|
+
|
|
87
|
+
# Extract edges for subgraph
|
|
88
|
+
final_edges = []
|
|
89
|
+
for edge in engine._store._graph.edges:
|
|
90
|
+
if edge.source_id in subgraph_nodes and edge.target_id in subgraph_nodes:
|
|
91
|
+
final_edges.append({
|
|
92
|
+
"source": edge.source_id,
|
|
93
|
+
"target": edge.target_id,
|
|
94
|
+
"kind": edge.kind.value if hasattr(edge, 'kind') and hasattr(edge.kind, 'value') else str(getattr(edge, 'kind', 'unknown')),
|
|
95
|
+
"confidence": edge.confidence.value if hasattr(edge, 'confidence') and hasattr(edge.confidence, 'value') else "unknown"
|
|
96
|
+
})
|
|
97
|
+
|
|
98
|
+
nodes = []
|
|
99
|
+
for n_id in subgraph_nodes:
|
|
100
|
+
# An edge can reference a node id Trailmark's own graph has no entry
|
|
101
|
+
# for (a dangling/malformed reference -- seen from complex chained
|
|
102
|
+
# expressions like `.map(fn).filter(...)`). The fallback must carry
|
|
103
|
+
# the same fields as a normal node, or downstream code that assumes
|
|
104
|
+
# e.g. 'file' always exists (to read source for LLM context) crashes
|
|
105
|
+
# with a bare KeyError on this one bad node instead of just treating
|
|
106
|
+
# it as having no known location.
|
|
107
|
+
nodes.append(node_metadata.get(n_id, {
|
|
108
|
+
"id": n_id,
|
|
109
|
+
"name": n_id,
|
|
110
|
+
"kind": "unknown",
|
|
111
|
+
"file": None,
|
|
112
|
+
"start_line": None,
|
|
113
|
+
"end_line": None,
|
|
114
|
+
"complexity": 0,
|
|
115
|
+
"status": "unchanged",
|
|
116
|
+
}))
|
|
117
|
+
|
|
118
|
+
return {
|
|
119
|
+
"nodes": nodes,
|
|
120
|
+
"edges": final_edges
|
|
121
|
+
}
|
zairo/cli.py
ADDED
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
import os
|
|
2
|
+
import typer
|
|
3
|
+
from rich.console import Console
|
|
4
|
+
from .analyzer import analyze_impact
|
|
5
|
+
from .reporter import generate_reports
|
|
6
|
+
from .llm_scanner import scan_graph_for_vulnerabilities
|
|
7
|
+
from .git_utils import create_worktree, remove_worktree
|
|
8
|
+
|
|
9
|
+
app = typer.Typer(add_completion=False)
|
|
10
|
+
console = Console()
|
|
11
|
+
|
|
12
|
+
@app.command()
|
|
13
|
+
def analyze(
|
|
14
|
+
repo_path: str = typer.Argument(..., help="Path to the git repository"),
|
|
15
|
+
depth: int = typer.Option(1, "--depth", "-d", help="Depth of connections to traverse from changed nodes"),
|
|
16
|
+
output_dir: str = typer.Option("zairo_out", "--output", "-o", help="Output directory for reports"),
|
|
17
|
+
base: str = typer.Option(None, "--base", "-b", help="Base commit/ref to diff from (e.g. HEAD~3, main, a1b2c3d)"),
|
|
18
|
+
target: str = typer.Option(None, "--target", "-t", help="Target commit/ref to diff to (e.g. HEAD, feature-branch). Requires --base."),
|
|
19
|
+
language: str = typer.Option("auto", "--language", "-l", help="Language for Trailmark parsing (auto, python, typescript, rust, etc.)"),
|
|
20
|
+
llm: bool = typer.Option(False, "--llm", help="Run LLM vulnerability scanning on modified nodes"),
|
|
21
|
+
model: str = typer.Option("gemini/gemini-1.5-pro", "--model", help="LiteLLM model string to use for scanning"),
|
|
22
|
+
concurrency: int = typer.Option(5, "--concurrency", "-c", help="Number of LLM scan requests to run in parallel"),
|
|
23
|
+
cache: bool = typer.Option(True, "--cache/--no-cache", help="Cache LLM findings by content hash in <output>/.llm_cache.json to skip re-scanning unchanged nodes across runs"),
|
|
24
|
+
max_tokens: int = typer.Option(4096, "--max-tokens", help="Max output tokens per LLM scan request. Reasoning models count internal thinking against this budget too — too low can cause empty responses"),
|
|
25
|
+
tokens: bool = typer.Option(False, "--tokens", help="Show total LLM tokens used by the scan (prompt/completion/total, across real API calls -- cache hits don't count)"),
|
|
26
|
+
verbose: bool = typer.Option(False, "--verbose", "-v", help="Print detailed diagnostic output (git commands, worktree setup, node matching, per-node LLM scan progress)")
|
|
27
|
+
):
|
|
28
|
+
def log(msg: str) -> None:
|
|
29
|
+
if verbose:
|
|
30
|
+
console.print(f"[dim] · {msg}[/dim]")
|
|
31
|
+
|
|
32
|
+
if base and target:
|
|
33
|
+
console.print(f"[bold green]Analyzing {repo_path} at depth {depth} — diff {base}..{target}[/bold green]")
|
|
34
|
+
elif base:
|
|
35
|
+
console.print(f"[bold green]Analyzing {repo_path} at depth {depth} — diff {base}..working tree[/bold green]")
|
|
36
|
+
else:
|
|
37
|
+
console.print(f"[bold green]Analyzing {repo_path} at depth {depth} — uncommitted changes[/bold green]")
|
|
38
|
+
# When diffing two commits, all downstream steps (graph analysis AND the
|
|
39
|
+
# LLM scan, which re-reads source files from disk) need to see `target`'s
|
|
40
|
+
# tree — not whatever happens to be checked out in repo_path already.
|
|
41
|
+
# The worktree must stay alive until every step that reads files is done.
|
|
42
|
+
abs_repo = os.path.abspath(repo_path)
|
|
43
|
+
worktree_path = None
|
|
44
|
+
analysis_root = abs_repo
|
|
45
|
+
try:
|
|
46
|
+
if base and target:
|
|
47
|
+
log(f"Checking out '{target}' into a temporary worktree (base+target diff mode)...")
|
|
48
|
+
worktree_path = create_worktree(abs_repo, target)
|
|
49
|
+
analysis_root = worktree_path
|
|
50
|
+
log(f"Worktree ready at {worktree_path}")
|
|
51
|
+
|
|
52
|
+
graph_data = analyze_impact(analysis_root, depth, base, target, language, log=log)
|
|
53
|
+
|
|
54
|
+
num_modified = sum(1 for n in graph_data['nodes'] if n['status'] != 'unchanged')
|
|
55
|
+
console.print(f"[bold blue]Found {num_modified} modified/added nodes.[/bold blue]")
|
|
56
|
+
console.print(f"[bold blue]Total nodes in subgraph: {len(graph_data['nodes'])}[/bold blue]")
|
|
57
|
+
console.print(f"[bold blue]Total edges in subgraph: {len(graph_data['edges'])}[/bold blue]")
|
|
58
|
+
|
|
59
|
+
vulnerabilities = None
|
|
60
|
+
if llm:
|
|
61
|
+
console.print(f"[bold yellow]Running LLM scanner using {model} (concurrency={concurrency})...[/bold yellow]")
|
|
62
|
+
cache_path = os.path.join(output_dir, ".llm_cache.json") if cache else None
|
|
63
|
+
vulnerabilities, token_usage = scan_graph_for_vulnerabilities(
|
|
64
|
+
graph_data, model, log=log, concurrency=concurrency, cache_path=cache_path,
|
|
65
|
+
max_tokens=max_tokens,
|
|
66
|
+
)
|
|
67
|
+
console.print(f"[bold yellow]Found vulnerabilities in {len(vulnerabilities)} nodes.[/bold yellow]")
|
|
68
|
+
|
|
69
|
+
if tokens:
|
|
70
|
+
if token_usage['requests'] == 0:
|
|
71
|
+
console.print("[dim]Token usage: no LLM requests were made (all results came from cache or were skipped).[/dim]")
|
|
72
|
+
elif token_usage['requests'] == token_usage['requests_without_usage']:
|
|
73
|
+
console.print(
|
|
74
|
+
f"[dim]Token usage: unavailable for all {token_usage['requests']} request(s) "
|
|
75
|
+
f"(provider/backend did not report it).[/dim]"
|
|
76
|
+
)
|
|
77
|
+
else:
|
|
78
|
+
counted = token_usage['requests'] - token_usage['requests_without_usage']
|
|
79
|
+
console.print(
|
|
80
|
+
f"[bold magenta]Tokens used:[/bold magenta] "
|
|
81
|
+
f"{token_usage['prompt_tokens']:,} prompt + {token_usage['completion_tokens']:,} completion "
|
|
82
|
+
f"= {token_usage['total_tokens']:,} total across {counted} request(s)"
|
|
83
|
+
)
|
|
84
|
+
if token_usage['requests_without_usage']:
|
|
85
|
+
console.print(
|
|
86
|
+
f"[dim] ({token_usage['requests_without_usage']} additional request(s) had no usage "
|
|
87
|
+
f"data reported by the provider — not counted above)[/dim]"
|
|
88
|
+
)
|
|
89
|
+
|
|
90
|
+
j_path, h_path = generate_reports(graph_data, output_dir, vulnerabilities)
|
|
91
|
+
|
|
92
|
+
console.print(f"[bold green]Success![/bold green] Reports generated:")
|
|
93
|
+
console.print(f" - {j_path}")
|
|
94
|
+
console.print(f" - {h_path}")
|
|
95
|
+
|
|
96
|
+
except Exception as e:
|
|
97
|
+
console.print(f"[bold red]Error:[/bold red] {e}")
|
|
98
|
+
raise typer.Exit(1)
|
|
99
|
+
finally:
|
|
100
|
+
if worktree_path:
|
|
101
|
+
log(f"Removing temporary worktree {worktree_path}")
|
|
102
|
+
remove_worktree(abs_repo, worktree_path)
|
|
103
|
+
|
|
104
|
+
def main():
|
|
105
|
+
app()
|
|
106
|
+
|
|
107
|
+
if __name__ == "__main__":
|
|
108
|
+
main()
|
zairo/git_utils.py
ADDED
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
import subprocess
|
|
2
|
+
import re
|
|
3
|
+
import os
|
|
4
|
+
import tempfile
|
|
5
|
+
from collections import defaultdict
|
|
6
|
+
from typing import Dict, List, Optional
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def create_worktree(repo_path: str, ref: str) -> str:
|
|
10
|
+
"""
|
|
11
|
+
Checks out `ref` into a new temporary git worktree and returns its path.
|
|
12
|
+
|
|
13
|
+
Used so that node locations/contents indexed by Trailmark line up with the
|
|
14
|
+
line numbers reported by `git diff base target` — those line numbers refer
|
|
15
|
+
to `target`'s tree, which may differ arbitrarily from whatever happens to
|
|
16
|
+
be checked out in the caller's working directory.
|
|
17
|
+
"""
|
|
18
|
+
worktree_path = tempfile.mkdtemp(prefix="zairo-worktree-")
|
|
19
|
+
result = subprocess.run(
|
|
20
|
+
["git", "worktree", "add", "--detach", "--force", worktree_path, ref],
|
|
21
|
+
cwd=repo_path,
|
|
22
|
+
capture_output=True,
|
|
23
|
+
text=True,
|
|
24
|
+
)
|
|
25
|
+
if result.returncode != 0:
|
|
26
|
+
raise RuntimeError(f"Failed to check out '{ref}' into a worktree: {result.stderr.strip()}")
|
|
27
|
+
return worktree_path
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def remove_worktree(repo_path: str, worktree_path: str) -> None:
|
|
31
|
+
subprocess.run(
|
|
32
|
+
["git", "worktree", "remove", "--force", worktree_path],
|
|
33
|
+
cwd=repo_path,
|
|
34
|
+
capture_output=True,
|
|
35
|
+
text=True,
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
def get_modified_lines(
|
|
39
|
+
repo_path: str,
|
|
40
|
+
base: str = None,
|
|
41
|
+
target: str = None,
|
|
42
|
+
log: Optional[callable] = None,
|
|
43
|
+
) -> Dict[str, Dict[int, str]]:
|
|
44
|
+
"""
|
|
45
|
+
Parses `git diff -U0` to find which lines have been added/modified.
|
|
46
|
+
|
|
47
|
+
- No base/target: compares working tree vs HEAD (uncommitted changes).
|
|
48
|
+
- base only: compares working tree vs that commit.
|
|
49
|
+
- base + target: compares two commits (e.g. HEAD~3..HEAD).
|
|
50
|
+
|
|
51
|
+
Returns a dict mapping absolute file paths to a dict of
|
|
52
|
+
{target line number: representative changed text}. The text is used to
|
|
53
|
+
cheaply filter out non-substantive changes (comments, blank lines)
|
|
54
|
+
before spending an LLM call on them, and to build a windowed view of
|
|
55
|
+
large functions instead of sending their full body.
|
|
56
|
+
|
|
57
|
+
A hunk with zero added lines (a pure deletion, e.g. `@@ -11 +10,0 @@`)
|
|
58
|
+
has no "+" line to anchor to in the target tree, but the enclosing node
|
|
59
|
+
still changed — a deleted validation check or sanitization call is
|
|
60
|
+
exactly the kind of change a security scan most needs to catch. Those
|
|
61
|
+
are recorded under a synthetic marker at the deletion's boundary line
|
|
62
|
+
in the target file, with the removed text as its value, so the
|
|
63
|
+
enclosing node is still found instead of silently skipped.
|
|
64
|
+
"""
|
|
65
|
+
log = log or (lambda msg: None)
|
|
66
|
+
|
|
67
|
+
# Build the git diff command
|
|
68
|
+
cmd = ["git", "diff", "-U0"]
|
|
69
|
+
if base and target:
|
|
70
|
+
cmd += [base, target]
|
|
71
|
+
elif base:
|
|
72
|
+
cmd += [base]
|
|
73
|
+
log(f"Running: {' '.join(cmd)} (cwd={repo_path})")
|
|
74
|
+
result = subprocess.run(cmd, cwd=repo_path, capture_output=True, text=True)
|
|
75
|
+
|
|
76
|
+
if result.returncode != 0:
|
|
77
|
+
log(f"git diff failed (exit {result.returncode}): {result.stderr.strip()}")
|
|
78
|
+
return {}
|
|
79
|
+
|
|
80
|
+
diff_output = result.stdout
|
|
81
|
+
|
|
82
|
+
modified_lines = defaultdict(dict)
|
|
83
|
+
current_file = None
|
|
84
|
+
next_line_num = None
|
|
85
|
+
pending_deletion_line = None
|
|
86
|
+
pending_deletion_text = []
|
|
87
|
+
|
|
88
|
+
def flush_pending_deletion():
|
|
89
|
+
if current_file and pending_deletion_line is not None and pending_deletion_text:
|
|
90
|
+
modified_lines[current_file][pending_deletion_line] = "\n".join(pending_deletion_text)
|
|
91
|
+
|
|
92
|
+
for line in diff_output.splitlines():
|
|
93
|
+
if line.startswith("+++ "):
|
|
94
|
+
flush_pending_deletion()
|
|
95
|
+
pending_deletion_line, pending_deletion_text = None, []
|
|
96
|
+
if line.startswith("+++ b/"):
|
|
97
|
+
# New file path — resolve to absolute so it matches Trailmark's locations
|
|
98
|
+
rel_path = line[6:]
|
|
99
|
+
current_file = os.path.abspath(os.path.join(repo_path, rel_path))
|
|
100
|
+
else:
|
|
101
|
+
# "+++ /dev/null": the whole file was deleted in the target.
|
|
102
|
+
# There's no target-side file to attribute this hunk to, and
|
|
103
|
+
# without resetting this, a stale current_file from the
|
|
104
|
+
# PREVIOUS file section in the diff would silently absorb
|
|
105
|
+
# this file's content -- a genuine cross-file data leak.
|
|
106
|
+
current_file = None
|
|
107
|
+
next_line_num = None
|
|
108
|
+
elif line.startswith("@@ ") and current_file:
|
|
109
|
+
flush_pending_deletion()
|
|
110
|
+
pending_deletion_line, pending_deletion_text = None, []
|
|
111
|
+
# Parse the + part of the hunk header
|
|
112
|
+
match = re.search(r'\+([0-9]+)(?:,([0-9]+))?', line)
|
|
113
|
+
if match:
|
|
114
|
+
start_line = int(match.group(1))
|
|
115
|
+
count = match.group(2)
|
|
116
|
+
count = int(count) if count is not None else 1
|
|
117
|
+
if count > 0:
|
|
118
|
+
next_line_num = start_line
|
|
119
|
+
else:
|
|
120
|
+
next_line_num = None
|
|
121
|
+
pending_deletion_line = max(1, start_line)
|
|
122
|
+
elif current_file and next_line_num is not None and line.startswith("+") and not line.startswith("+++"):
|
|
123
|
+
# With -U0 there are no context lines, so every "+" line after a
|
|
124
|
+
# hunk header maps to the next line number in the added range.
|
|
125
|
+
modified_lines[current_file][next_line_num] = line[1:]
|
|
126
|
+
next_line_num += 1
|
|
127
|
+
elif current_file and pending_deletion_line is not None and line.startswith("-") and not line.startswith("---"):
|
|
128
|
+
pending_deletion_text.append(line[1:])
|
|
129
|
+
|
|
130
|
+
flush_pending_deletion()
|
|
131
|
+
return dict(modified_lines)
|
zairo/llm_scanner.py
ADDED
|
@@ -0,0 +1,507 @@
|
|
|
1
|
+
import hashlib
|
|
2
|
+
import json
|
|
3
|
+
import os
|
|
4
|
+
import re
|
|
5
|
+
import threading
|
|
6
|
+
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
7
|
+
from functools import lru_cache
|
|
8
|
+
from typing import Callable, Dict, List, Optional, Tuple, Any
|
|
9
|
+
|
|
10
|
+
# `litellm` transitively imports the openai/anthropic SDKs and their full
|
|
11
|
+
# Pydantic type trees (~3s). Import it lazily, only once actual scanning
|
|
12
|
+
# happens, so `--help` and non-`--llm` runs don't pay that cost.
|
|
13
|
+
litellm = None
|
|
14
|
+
|
|
15
|
+
def _ensure_litellm():
|
|
16
|
+
global litellm
|
|
17
|
+
if litellm is None:
|
|
18
|
+
import litellm as _litellm
|
|
19
|
+
litellm = _litellm
|
|
20
|
+
return litellm
|
|
21
|
+
|
|
22
|
+
from ._util import display_name as _display_name
|
|
23
|
+
|
|
24
|
+
# Comment/blank-only diffs (docs, version bumps, log messages) can't produce a
|
|
25
|
+
# real vulnerability finding — skip them before spending an LLM call.
|
|
26
|
+
_COMMENT_PREFIXES = ("//", "#", "*", "/*", "<!--", "-->", "--", "'''", '"""')
|
|
27
|
+
|
|
28
|
+
# Test files aren't part of the shipped attack surface. Scanning them tends
|
|
29
|
+
# to produce either a duplicate of a finding already attached to the real
|
|
30
|
+
# implementation they exercise, or a category error (treating mock/test
|
|
31
|
+
# scaffolding as if it were exploitable production code) -- wasted LLM calls
|
|
32
|
+
# for low-value output either way.
|
|
33
|
+
_TEST_DIR_NAMES = {'test', 'tests', '__tests__', 'spec', 'specs'}
|
|
34
|
+
_TEST_STEM_PREFIXES = ('test_', 'test-')
|
|
35
|
+
_TEST_STEM_SUFFIXES = ('_test', '-test', '.test', '_spec', '-spec', '.spec')
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _is_test_file(file_path: Optional[str]) -> bool:
|
|
39
|
+
if not file_path:
|
|
40
|
+
return False
|
|
41
|
+
parts = re.split(r'[/\\]', file_path)
|
|
42
|
+
if any(p.lower() in _TEST_DIR_NAMES for p in parts[:-1]):
|
|
43
|
+
return True
|
|
44
|
+
# Strip exactly one extension so "index.test.ts" -> "index.test" (still
|
|
45
|
+
# matches the ".test" suffix) without over-stripping "test_utils.py".
|
|
46
|
+
stem = re.sub(r'\.[a-zA-Z0-9]+$', '', parts[-1]).lower()
|
|
47
|
+
return stem.startswith(_TEST_STEM_PREFIXES) or stem.endswith(_TEST_STEM_SUFFIXES)
|
|
48
|
+
|
|
49
|
+
# Functions larger than this get a windowed view around the changed lines
|
|
50
|
+
# instead of their full body, so a one-line change in a 600-line function
|
|
51
|
+
# doesn't cost 600 lines of prompt. Padding is generous on purpose: a tight
|
|
52
|
+
# window can hide the guard clause or sanitization that makes a line safe,
|
|
53
|
+
# which turns "efficient" into "wrong" for a security review specifically.
|
|
54
|
+
_LARGE_FUNCTION_LINES = 100
|
|
55
|
+
_WINDOW_PADDING = 20
|
|
56
|
+
# Signature + early guard clauses are usually here — always include them
|
|
57
|
+
# even when the diff itself is much further down the function.
|
|
58
|
+
_GUARD_HEAD_LINES = 15
|
|
59
|
+
|
|
60
|
+
# Neighbor (caller/callee) context is for orientation, not full audit — cap it.
|
|
61
|
+
_NEIGHBOR_MAX_LINES = 30
|
|
62
|
+
|
|
63
|
+
# Reasoning ("thinking") models count their internal reasoning tokens against
|
|
64
|
+
# this same budget. Too low a cap can make the model exhaust it mid-thought
|
|
65
|
+
# and return empty content before ever writing the JSON answer — so this
|
|
66
|
+
# needs real headroom, not just enough for the expected output size.
|
|
67
|
+
_DEFAULT_MAX_OUTPUT_TOKENS = 4096
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
@lru_cache(maxsize=4096)
|
|
71
|
+
def get_source_code(file_path: str, start_line: Optional[int], end_line: Optional[int]) -> str:
|
|
72
|
+
if not file_path or not os.path.exists(file_path):
|
|
73
|
+
return ""
|
|
74
|
+
try:
|
|
75
|
+
with open(file_path, 'r', encoding='utf-8') as f:
|
|
76
|
+
lines = f.readlines()
|
|
77
|
+
|
|
78
|
+
if start_line is None or end_line is None:
|
|
79
|
+
return "".join(lines)
|
|
80
|
+
|
|
81
|
+
start_idx = max(0, start_line - 1)
|
|
82
|
+
end_idx = min(len(lines), end_line)
|
|
83
|
+
return "".join(lines[start_idx:end_idx])
|
|
84
|
+
except Exception as e:
|
|
85
|
+
return f"// Error reading file: {e}"
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _is_trivial_change(changed_lines: Optional[Dict[int, str]]) -> bool:
|
|
89
|
+
"""True if every added/removed line is blank or a comment — not worth an
|
|
90
|
+
LLM call. A pure-deletion marker can bundle several removed lines into
|
|
91
|
+
one multi-line value, so check each physical line within it, not just
|
|
92
|
+
the value as a whole."""
|
|
93
|
+
if not changed_lines:
|
|
94
|
+
return False # no diff info available; don't risk a false skip
|
|
95
|
+
for text in changed_lines.values():
|
|
96
|
+
for physical_line in (text.splitlines() or [text]):
|
|
97
|
+
stripped = physical_line.strip()
|
|
98
|
+
if not stripped:
|
|
99
|
+
continue
|
|
100
|
+
if stripped.startswith(_COMMENT_PREFIXES):
|
|
101
|
+
continue
|
|
102
|
+
return False
|
|
103
|
+
return True
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def _windowed_source(file_path: str, start_line: int, end_line: int, changed_lines: Dict[int, str]) -> str:
|
|
107
|
+
"""Full body for small functions; a padded window around changed lines for
|
|
108
|
+
large ones, plus the function's head (signature + early guard clauses)
|
|
109
|
+
unconditionally — a check made there determines whether a flagged line
|
|
110
|
+
further down is actually reachable/dangerous."""
|
|
111
|
+
if (end_line - start_line + 1) <= _LARGE_FUNCTION_LINES or not changed_lines:
|
|
112
|
+
return get_source_code(file_path, start_line, end_line)
|
|
113
|
+
|
|
114
|
+
ranges = []
|
|
115
|
+
|
|
116
|
+
def add_range(lo, hi):
|
|
117
|
+
if ranges and lo <= ranges[-1][1] + 1:
|
|
118
|
+
ranges[-1] = (ranges[-1][0], max(ranges[-1][1], hi))
|
|
119
|
+
else:
|
|
120
|
+
ranges.append((lo, hi))
|
|
121
|
+
|
|
122
|
+
add_range(start_line, min(end_line, start_line + _GUARD_HEAD_LINES - 1))
|
|
123
|
+
for ln in sorted(changed_lines.keys()):
|
|
124
|
+
lo = max(start_line, ln - _WINDOW_PADDING)
|
|
125
|
+
hi = min(end_line, ln + _WINDOW_PADDING)
|
|
126
|
+
add_range(lo, hi)
|
|
127
|
+
|
|
128
|
+
chunks = [f"# lines {lo}-{hi}\n{get_source_code(file_path, lo, hi)}" for lo, hi in ranges]
|
|
129
|
+
return "\n...\n".join(chunks)
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def _sibling_outline(mod_node: Dict[str, Any], nodes: Dict[str, Dict[str, Any]]) -> str:
|
|
133
|
+
"""For a module/file-level node, a windowed snippet alone loses all
|
|
134
|
+
orientation — the model can't tell what else the file contains. List the
|
|
135
|
+
other definitions in the same file (name + line range only, no bodies)
|
|
136
|
+
so it has that context without paying to send them in full."""
|
|
137
|
+
siblings = [
|
|
138
|
+
n for n in nodes.values()
|
|
139
|
+
if n.get('file') == mod_node.get('file')
|
|
140
|
+
and n['id'] != mod_node['id']
|
|
141
|
+
and n.get('kind') in ('function', 'class', 'method')
|
|
142
|
+
]
|
|
143
|
+
if not siblings:
|
|
144
|
+
return ""
|
|
145
|
+
siblings.sort(key=lambda n: n.get('start_line') or 0)
|
|
146
|
+
lines = [f"- {n['name']} ({n.get('kind')}, lines {n.get('start_line')}-{n.get('end_line')})" for n in siblings]
|
|
147
|
+
return "Other definitions in this file (not shown in full):\n" + "\n".join(lines)
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def _collapse_nested_definitions(
|
|
151
|
+
file_path: str, start_line: int, end_line: int, nested: List[Dict[str, Any]]
|
|
152
|
+
) -> str:
|
|
153
|
+
"""For a module/file-level node, replace the body of each nested
|
|
154
|
+
top-level function/class with a one-line placeholder instead of sending
|
|
155
|
+
it in full. A nested definition that changed is already covered by its
|
|
156
|
+
own, more specific seed node — including its full body here too means
|
|
157
|
+
the same vulnerable line gets independently re-flagged under the
|
|
158
|
+
enclosing module as well: wasted cost, and a confusing/duplicate
|
|
159
|
+
attribution in the report (e.g. a vulnerability inside `parse` showing
|
|
160
|
+
up as "module X is vulnerable" instead of "parse is vulnerable")."""
|
|
161
|
+
if not nested:
|
|
162
|
+
return get_source_code(file_path, start_line, end_line)
|
|
163
|
+
|
|
164
|
+
skip_ranges = sorted(
|
|
165
|
+
(max(start_line, n['start_line']), min(end_line, n['end_line']), n['name'])
|
|
166
|
+
for n in nested
|
|
167
|
+
if n.get('start_line') is not None and n.get('end_line') is not None
|
|
168
|
+
and n['end_line'] >= start_line and n['start_line'] <= end_line
|
|
169
|
+
)
|
|
170
|
+
|
|
171
|
+
chunks = []
|
|
172
|
+
cursor = start_line
|
|
173
|
+
for lo, hi, name in skip_ranges:
|
|
174
|
+
if lo < cursor:
|
|
175
|
+
continue # nested-within-nested overlap already covered by a prior placeholder
|
|
176
|
+
if lo > cursor:
|
|
177
|
+
chunks.append(get_source_code(file_path, cursor, lo - 1))
|
|
178
|
+
chunks.append(f" # ... body of `{name}` NOT SHOWN (reviewed separately -- do not guess its contents) ...")
|
|
179
|
+
cursor = hi + 1
|
|
180
|
+
|
|
181
|
+
if cursor <= end_line:
|
|
182
|
+
chunks.append(get_source_code(file_path, cursor, end_line))
|
|
183
|
+
|
|
184
|
+
return "".join(chunks)
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def _neighbor_snippet(n: Dict[str, Any]) -> Optional[str]:
|
|
188
|
+
# .get() throughout: a neighbor can be a malformed/dangling graph node
|
|
189
|
+
# missing these fields entirely (see analyzer.py's subgraph-assembly
|
|
190
|
+
# fallback) -- treat it as having no known source rather than crashing.
|
|
191
|
+
code = get_source_code(n.get('file'), n.get('start_line'), n.get('end_line'))
|
|
192
|
+
if not code.strip():
|
|
193
|
+
return None
|
|
194
|
+
lines = code.splitlines()
|
|
195
|
+
if len(lines) > _NEIGHBOR_MAX_LINES:
|
|
196
|
+
truncated = len(lines) - _NEIGHBOR_MAX_LINES
|
|
197
|
+
code = "\n".join(lines[:_NEIGHBOR_MAX_LINES]) + f"\n... ({truncated} more line(s) truncated)"
|
|
198
|
+
return f"Function: {n.get('name', '?')}\n```\n{code}\n```"
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def _hash_prompt(model: str, mod_code: str, neighbor_contexts: List[str]) -> str:
|
|
202
|
+
h = hashlib.sha256()
|
|
203
|
+
h.update(model.encode('utf-8'))
|
|
204
|
+
h.update(b'\x00')
|
|
205
|
+
h.update(mod_code.encode('utf-8', errors='ignore'))
|
|
206
|
+
for c in sorted(neighbor_contexts):
|
|
207
|
+
h.update(b'\x00')
|
|
208
|
+
h.update(c.encode('utf-8', errors='ignore'))
|
|
209
|
+
return h.hexdigest()
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def _load_cache(cache_path: Optional[str]) -> Dict[str, List[Dict]]:
|
|
213
|
+
if not cache_path or not os.path.exists(cache_path):
|
|
214
|
+
return {}
|
|
215
|
+
try:
|
|
216
|
+
with open(cache_path, 'r') as f:
|
|
217
|
+
return json.load(f)
|
|
218
|
+
except Exception:
|
|
219
|
+
return {}
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def _save_cache(cache_path: Optional[str], cache: Dict[str, List[Dict]]) -> None:
|
|
223
|
+
if not cache_path:
|
|
224
|
+
return
|
|
225
|
+
try:
|
|
226
|
+
os.makedirs(os.path.dirname(cache_path) or ".", exist_ok=True)
|
|
227
|
+
with open(cache_path, 'w') as f:
|
|
228
|
+
json.dump(cache, f)
|
|
229
|
+
except Exception:
|
|
230
|
+
pass
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
# Strips a leading/trailing code fence regardless of language tag
|
|
234
|
+
# (```json, ```JSON, or bare ```), which is the only variant the old
|
|
235
|
+
# literal-string `startswith("```json")` check handled.
|
|
236
|
+
_FENCE_RE = re.compile(r'^\s*```[a-zA-Z]*\s*\n?|\n?\s*```\s*$')
|
|
237
|
+
|
|
238
|
+
# Some models embed regex-like text (e.g. \w, \d) directly inside a JSON
|
|
239
|
+
# string without escaping the backslash -- only \", \\, \/, \b, \f, \n, \r,
|
|
240
|
+
# \t, \uXXXX are valid JSON escapes, so a raw \w is a hard parse error.
|
|
241
|
+
# The first alternative below matches (and leaves untouched) any already-
|
|
242
|
+
# valid escape as a single unit; the second only fires on a lone/invalid
|
|
243
|
+
# backslash, so this is a no-op on already-valid JSON.
|
|
244
|
+
_INVALID_ESCAPE_RE = re.compile(r'\\(["\\/bfnrtu])|\\')
|
|
245
|
+
|
|
246
|
+
|
|
247
|
+
def _fix_invalid_escapes(s: str) -> str:
|
|
248
|
+
return _INVALID_ESCAPE_RE.sub(lambda m: m.group(0) if m.group(1) else '\\\\', s)
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
_JSON_DECODER = json.JSONDecoder(strict=False) # strict=False: tolerate raw
|
|
252
|
+
# control characters (e.g. literal newlines) inside string values, which
|
|
253
|
+
# some models emit instead of a proper \n escape.
|
|
254
|
+
|
|
255
|
+
|
|
256
|
+
def _extract_json(content: str) -> dict:
|
|
257
|
+
"""Parses the model's response, tolerating fence variants, stray prose,
|
|
258
|
+
trailing content after the JSON (some smaller/less-aligned models keep
|
|
259
|
+
generating after a complete answer — duplicate output, trailing
|
|
260
|
+
commentary — which plain json.loads() rejects outright as "Extra data"),
|
|
261
|
+
invalid backslash escapes from embedded regex-like text, and a bare
|
|
262
|
+
`[...]` findings array where a `{"vulnerabilities": [...]}` object was
|
|
263
|
+
asked for (normalized back into that shape here, at the parsing
|
|
264
|
+
boundary, so callers can always rely on a dict with a "vulnerabilities"
|
|
265
|
+
key). Uses JSONDecoder.raw_decode, which parses the first complete JSON
|
|
266
|
+
value and stops there instead of requiring the whole string to be one
|
|
267
|
+
value.
|
|
268
|
+
"""
|
|
269
|
+
fixed = _fix_invalid_escapes(content)
|
|
270
|
+
stripped = _FENCE_RE.sub('', fixed).strip()
|
|
271
|
+
|
|
272
|
+
candidates = [stripped]
|
|
273
|
+
first_brace = fixed.find('{')
|
|
274
|
+
if first_brace != -1:
|
|
275
|
+
candidates.append(fixed[first_brace:].strip())
|
|
276
|
+
|
|
277
|
+
last_err = None
|
|
278
|
+
last_candidate = None
|
|
279
|
+
for candidate in candidates:
|
|
280
|
+
if not candidate:
|
|
281
|
+
continue
|
|
282
|
+
try:
|
|
283
|
+
obj, _ = _JSON_DECODER.raw_decode(candidate)
|
|
284
|
+
if isinstance(obj, list):
|
|
285
|
+
obj = {"vulnerabilities": obj}
|
|
286
|
+
return obj
|
|
287
|
+
except json.JSONDecodeError as e:
|
|
288
|
+
last_err, last_candidate = e, candidate
|
|
289
|
+
|
|
290
|
+
# Show the text around the actual failure point, not just the start of
|
|
291
|
+
# the response — a generic head-of-string preview doesn't help diagnose
|
|
292
|
+
# a structural error (e.g. a missing comma) that occurs deep in the doc.
|
|
293
|
+
if last_err is not None:
|
|
294
|
+
pos = last_err.pos
|
|
295
|
+
window = last_candidate[max(0, pos - 80):pos + 80]
|
|
296
|
+
raise ValueError(f"could not find valid JSON ({last_err}); near failure point: {window!r}")
|
|
297
|
+
|
|
298
|
+
raise ValueError(f"could not find valid JSON in model response: {content[:200]!r}")
|
|
299
|
+
|
|
300
|
+
|
|
301
|
+
def _build_prompt(mod_node: Dict[str, Any], mod_code: str, neighbor_contexts: List[str]) -> str:
|
|
302
|
+
kind_label = mod_node.get('kind') or 'function'
|
|
303
|
+
return f"""
|
|
304
|
+
You are an expert security auditor. Analyze the following modified {kind_label} for vulnerabilities.
|
|
305
|
+
Modified {kind_label.capitalize()}: {mod_node['name']}
|
|
306
|
+
```
|
|
307
|
+
{mod_code}
|
|
308
|
+
```
|
|
309
|
+
Context Functions (Callers/Callees):
|
|
310
|
+
{chr(10).join(neighbor_contexts)}
|
|
311
|
+
|
|
312
|
+
Base every finding strictly on the code actually shown above. Do not speculate about the contents of omitted/NOT-SHOWN function bodies, imports, or third-party libraries based on their name alone — if you haven't seen the code, don't report a vulnerability in it.
|
|
313
|
+
|
|
314
|
+
Return ONLY a JSON object with a single key 'vulnerabilities' — no markdown code fence, no prose before or after it. The value should be a list of objects containing 'title', 'description', and 'impact'. Keep each field to 1-2 sentences. If no vulnerabilities are found, return {{"vulnerabilities": []}}.
|
|
315
|
+
"""
|
|
316
|
+
|
|
317
|
+
|
|
318
|
+
def scan_graph_for_vulnerabilities(
|
|
319
|
+
graph_data: Dict[str, Any],
|
|
320
|
+
model: str,
|
|
321
|
+
log: Optional[Callable[[str], None]] = None,
|
|
322
|
+
concurrency: int = 5,
|
|
323
|
+
cache_path: Optional[str] = None,
|
|
324
|
+
max_tokens: int = _DEFAULT_MAX_OUTPUT_TOKENS,
|
|
325
|
+
) -> Tuple[Dict[str, List[Dict]], Dict[str, int]]:
|
|
326
|
+
"""Returns (vulnerabilities, token_usage). token_usage has
|
|
327
|
+
prompt_tokens/completion_tokens/total_tokens summed across every real
|
|
328
|
+
LLM call made (cache hits don't count -- they made no call), plus
|
|
329
|
+
requests/requests_without_usage so a caller can tell whether the token
|
|
330
|
+
totals are complete or partial (e.g. some providers don't report it)."""
|
|
331
|
+
log = log or (lambda msg: None)
|
|
332
|
+
log_lock = threading.Lock()
|
|
333
|
+
|
|
334
|
+
def safe_log(msg: str) -> None:
|
|
335
|
+
with log_lock:
|
|
336
|
+
log(msg)
|
|
337
|
+
|
|
338
|
+
vulnerabilities = {}
|
|
339
|
+
nodes = {n['id']: n for n in graph_data['nodes']}
|
|
340
|
+
edges = graph_data['edges']
|
|
341
|
+
cache = _load_cache(cache_path)
|
|
342
|
+
|
|
343
|
+
model_label = model
|
|
344
|
+
|
|
345
|
+
modified_nodes = [n for n in graph_data['nodes'] if n['status'] in ['modified', 'added']]
|
|
346
|
+
log(f"Scanning {len(modified_nodes)} modified/added node(s) with {model_label}")
|
|
347
|
+
|
|
348
|
+
# Build prompts up front (cheap, local) so trivial/cached nodes never
|
|
349
|
+
# touch the network, and only real work goes into the thread pool.
|
|
350
|
+
jobs = []
|
|
351
|
+
for mod_node in modified_nodes:
|
|
352
|
+
if mod_node.get('kind') == 'proxy':
|
|
353
|
+
# A proxy node represents an external/unresolved call target
|
|
354
|
+
# (e.g. `fs.unlinkSync`), shared across every call site to it in
|
|
355
|
+
# the whole codebase -- it isn't first-party source with a real
|
|
356
|
+
# body of its own. Scanning it pulls in ALL of its callers as
|
|
357
|
+
# "context" (anyone who calls fs.unlinkSync, anywhere), which
|
|
358
|
+
# leaks fully unrelated, unchanged functions from other files
|
|
359
|
+
# into the prompt and produces findings misattributed to code
|
|
360
|
+
# that was never touched by this diff.
|
|
361
|
+
log(f" skip (proxy node, not first-party source): {_display_name(mod_node['name'])}")
|
|
362
|
+
continue
|
|
363
|
+
if _is_test_file(mod_node.get('file')):
|
|
364
|
+
log(f" skip (test file, not shipped attack surface): {_display_name(mod_node['name'])} ({mod_node.get('file')})")
|
|
365
|
+
continue
|
|
366
|
+
changed_lines = mod_node.get('changed_lines')
|
|
367
|
+
# Trivial-skip only applies to module-level edits (e.g. a version
|
|
368
|
+
# bump, a standalone doc comment). Skipping a function-kind node
|
|
369
|
+
# because its own diff happens to be comment-only would also skip
|
|
370
|
+
# scanning whatever pre-existing vulnerable code the rest of that
|
|
371
|
+
# (possibly still-unfixed) function contains -- and small functions
|
|
372
|
+
# cost nothing extra to scan in full anyway, so there's no real
|
|
373
|
+
# savings being traded away by not skipping them.
|
|
374
|
+
if mod_node.get('kind') == 'module' and _is_trivial_change(changed_lines):
|
|
375
|
+
log(f" skip (trivial diff): {_display_name(mod_node['name'])} ({mod_node.get('file')})")
|
|
376
|
+
continue
|
|
377
|
+
|
|
378
|
+
start, end = mod_node.get('start_line'), mod_node.get('end_line')
|
|
379
|
+
if mod_node.get('kind') == 'module' and start is not None and end is not None:
|
|
380
|
+
# A module/file node's own scan shouldn't re-send full bodies of
|
|
381
|
+
# nested functions/classes -- those are already covered by their
|
|
382
|
+
# own, more specific seed node when they change, and including
|
|
383
|
+
# them here too just duplicates cost and mis-attributes their
|
|
384
|
+
# findings to the enclosing module instead of the actual function.
|
|
385
|
+
nested = [
|
|
386
|
+
n for n in nodes.values()
|
|
387
|
+
if n.get('file') == mod_node['file']
|
|
388
|
+
and n['id'] != mod_node['id']
|
|
389
|
+
and n.get('kind') in ('function', 'class')
|
|
390
|
+
and n.get('start_line') is not None and n.get('end_line') is not None
|
|
391
|
+
and n['start_line'] >= start and n['end_line'] <= end
|
|
392
|
+
]
|
|
393
|
+
mod_code = _collapse_nested_definitions(mod_node['file'], start, end, nested)
|
|
394
|
+
elif changed_lines and start is not None and end is not None:
|
|
395
|
+
mod_code = _windowed_source(mod_node['file'], start, end, changed_lines)
|
|
396
|
+
else:
|
|
397
|
+
mod_code = get_source_code(mod_node['file'], start, end)
|
|
398
|
+
|
|
399
|
+
if not mod_code.strip():
|
|
400
|
+
log(f" skip (no source found): {_display_name(mod_node['name'])}")
|
|
401
|
+
continue
|
|
402
|
+
|
|
403
|
+
# A module/file-level node's window is a tiny slice of the whole
|
|
404
|
+
# file — without an outline of what else is there, the model has no
|
|
405
|
+
# idea whether the flagged line is actually reachable in isolation.
|
|
406
|
+
if mod_node.get('kind') == 'module':
|
|
407
|
+
outline = _sibling_outline(mod_node, nodes)
|
|
408
|
+
if outline:
|
|
409
|
+
mod_code = outline + "\n\n" + mod_code
|
|
410
|
+
|
|
411
|
+
# "contains" edges are structural nesting (module -> its functions),
|
|
412
|
+
# not a caller/callee relationship -- pulling a contained child's
|
|
413
|
+
# full body in here as "context" would (a) re-leak exactly the
|
|
414
|
+
# content the collapse step above just excluded from a module's own
|
|
415
|
+
# scan, reintroducing the duplicate-attribution bug, and (b) for a
|
|
416
|
+
# function node, pointlessly pull in its enclosing module's source
|
|
417
|
+
# under a "Callers/Callees" label where it doesn't belong.
|
|
418
|
+
neighbor_ids = set()
|
|
419
|
+
for e in edges:
|
|
420
|
+
if e.get('kind') == 'contains':
|
|
421
|
+
continue
|
|
422
|
+
if e['source'] == mod_node['id']:
|
|
423
|
+
neighbor_ids.add(e['target'])
|
|
424
|
+
elif e['target'] == mod_node['id']:
|
|
425
|
+
neighbor_ids.add(e['source'])
|
|
426
|
+
|
|
427
|
+
neighbor_contexts = []
|
|
428
|
+
for n_id in neighbor_ids:
|
|
429
|
+
if n_id in nodes and n_id != mod_node['id']:
|
|
430
|
+
snippet = _neighbor_snippet(nodes[n_id])
|
|
431
|
+
if snippet:
|
|
432
|
+
neighbor_contexts.append(snippet)
|
|
433
|
+
|
|
434
|
+
prompt_hash = _hash_prompt(model_label, mod_code, neighbor_contexts)
|
|
435
|
+
cached = cache.get(prompt_hash)
|
|
436
|
+
if cached is not None:
|
|
437
|
+
log(f" cache hit: {_display_name(mod_node['name'])} ({len(cached)} finding(s))")
|
|
438
|
+
if cached:
|
|
439
|
+
vulnerabilities[mod_node['id']] = cached
|
|
440
|
+
continue
|
|
441
|
+
|
|
442
|
+
jobs.append((mod_node, prompt_hash, _build_prompt(mod_node, mod_code, neighbor_contexts)))
|
|
443
|
+
|
|
444
|
+
def run_job(job):
|
|
445
|
+
mod_node, prompt_hash, prompt = job
|
|
446
|
+
usage = None # unavailable if the provider doesn't report it
|
|
447
|
+
try:
|
|
448
|
+
response = litellm.completion(
|
|
449
|
+
model=model,
|
|
450
|
+
messages=[{"role": "user", "content": prompt}],
|
|
451
|
+
max_tokens=max_tokens,
|
|
452
|
+
)
|
|
453
|
+
choice = response.choices[0]
|
|
454
|
+
content = choice.message.content or ""
|
|
455
|
+
finish_reason = getattr(choice, 'finish_reason', 'unknown')
|
|
456
|
+
empty_note = (
|
|
457
|
+
f" (finish_reason={finish_reason}) — likely exhausted max_tokens={max_tokens} "
|
|
458
|
+
f"on internal reasoning before writing an answer; try --max-tokens with a higher value"
|
|
459
|
+
)
|
|
460
|
+
resp_usage = getattr(response, 'usage', None)
|
|
461
|
+
if resp_usage is not None:
|
|
462
|
+
usage = {
|
|
463
|
+
'prompt_tokens': getattr(resp_usage, 'prompt_tokens', 0) or 0,
|
|
464
|
+
'completion_tokens': getattr(resp_usage, 'completion_tokens', 0) or 0,
|
|
465
|
+
'total_tokens': getattr(resp_usage, 'total_tokens', 0) or 0,
|
|
466
|
+
}
|
|
467
|
+
|
|
468
|
+
if not content.strip():
|
|
469
|
+
safe_log(f" error scanning {_display_name(mod_node['name'])}: model returned empty content{empty_note}")
|
|
470
|
+
return mod_node['id'], prompt_hash, None, usage
|
|
471
|
+
|
|
472
|
+
try:
|
|
473
|
+
parsed = _extract_json(content)
|
|
474
|
+
except ValueError as e:
|
|
475
|
+
safe_log(f" error scanning {_display_name(mod_node['name'])}: {e}")
|
|
476
|
+
return mod_node['id'], prompt_hash, None, usage
|
|
477
|
+
|
|
478
|
+
findings = parsed.get("vulnerabilities", [])
|
|
479
|
+
safe_log(f" found {len(findings)} vulnerability finding(s): {_display_name(mod_node['name'])}")
|
|
480
|
+
return mod_node['id'], prompt_hash, findings, usage
|
|
481
|
+
except Exception as e:
|
|
482
|
+
safe_log(f" error scanning {_display_name(mod_node['name'])}: {e}")
|
|
483
|
+
return mod_node['id'], prompt_hash, None, usage
|
|
484
|
+
|
|
485
|
+
token_usage = {'prompt_tokens': 0, 'completion_tokens': 0, 'total_tokens': 0, 'requests': 0, 'requests_without_usage': 0}
|
|
486
|
+
|
|
487
|
+
if jobs:
|
|
488
|
+
_ensure_litellm() # deferred until there's actually a request to make
|
|
489
|
+
with ThreadPoolExecutor(max_workers=max(1, concurrency)) as pool:
|
|
490
|
+
futures = [pool.submit(run_job, job) for job in jobs]
|
|
491
|
+
for future in as_completed(futures):
|
|
492
|
+
node_id, prompt_hash, findings, usage = future.result()
|
|
493
|
+
token_usage['requests'] += 1
|
|
494
|
+
if usage:
|
|
495
|
+
token_usage['prompt_tokens'] += usage['prompt_tokens']
|
|
496
|
+
token_usage['completion_tokens'] += usage['completion_tokens']
|
|
497
|
+
token_usage['total_tokens'] += usage['total_tokens']
|
|
498
|
+
else:
|
|
499
|
+
token_usage['requests_without_usage'] += 1
|
|
500
|
+
if findings is None:
|
|
501
|
+
continue # request failed; don't cache a non-result
|
|
502
|
+
cache[prompt_hash] = findings
|
|
503
|
+
if findings:
|
|
504
|
+
vulnerabilities[node_id] = findings
|
|
505
|
+
|
|
506
|
+
_save_cache(cache_path, cache)
|
|
507
|
+
return vulnerabilities, token_usage
|
zairo/reporter.py
ADDED
|
@@ -0,0 +1,197 @@
|
|
|
1
|
+
import json
|
|
2
|
+
import os
|
|
3
|
+
from jinja2 import Template
|
|
4
|
+
|
|
5
|
+
HTML_TEMPLATE = """
|
|
6
|
+
<!DOCTYPE html>
|
|
7
|
+
<html>
|
|
8
|
+
<head>
|
|
9
|
+
<title>Zairo Impact Analysis</title>
|
|
10
|
+
<script src="https://cdnjs.cloudflare.com/ajax/libs/cytoscape/3.28.1/cytoscape.min.js"></script>
|
|
11
|
+
<script src="https://cdnjs.cloudflare.com/ajax/libs/dagre/0.8.5/dagre.min.js"></script>
|
|
12
|
+
<script src="https://cdn.jsdelivr.net/npm/cytoscape-dagre@2.5.0/cytoscape-dagre.min.js"></script>
|
|
13
|
+
<style>
|
|
14
|
+
body { font-family: sans-serif; margin: 0; padding: 0; display: flex; height: 100vh; background-color: #1e1e1e; color: #fff;}
|
|
15
|
+
#cy { width: 80%; height: 100%; }
|
|
16
|
+
#sidebar { width: 20%; height: 100%; background: #252526; padding: 20px; box-sizing: border-box; overflow-y: auto; border-left: 1px solid #3c3c3c;}
|
|
17
|
+
h1 { font-size: 1.2em; border-bottom: 1px solid #3c3c3c; padding-bottom: 10px; }
|
|
18
|
+
.details { margin-top: 20px; font-size: 0.9em; }
|
|
19
|
+
.details strong { color: #4fc1ff; }
|
|
20
|
+
button { background: #0e639c; color: white; border: none; padding: 8px 12px; cursor: pointer; margin-bottom: 10px; width: 100%; }
|
|
21
|
+
button:hover { background: #1177bb; }
|
|
22
|
+
.legend { margin-top: 20px; font-size: 0.9em; border-top: 1px solid #3c3c3c; padding-top: 10px;}
|
|
23
|
+
.legend-item { display: flex; align-items: center; margin-bottom: 5px; }
|
|
24
|
+
.color-box { width: 15px; height: 15px; margin-right: 10px; }
|
|
25
|
+
</style>
|
|
26
|
+
</head>
|
|
27
|
+
<body>
|
|
28
|
+
<div id="cy"></div>
|
|
29
|
+
<div id="sidebar">
|
|
30
|
+
<h1>Zairo Impact</h1>
|
|
31
|
+
<button id="btn-dagre">Layout: Hierarchical</button>
|
|
32
|
+
<button id="btn-cose">Layout: Force-Directed</button>
|
|
33
|
+
<button id="btn-toggle">Toggle Unchanged Nodes</button>
|
|
34
|
+
|
|
35
|
+
<div class="legend">
|
|
36
|
+
<div class="legend-item"><div class="color-box" style="background: #4CAF50;"></div> Added</div>
|
|
37
|
+
<div class="legend-item"><div class="color-box" style="background: #d7ba7d;"></div> Modified</div>
|
|
38
|
+
<div class="legend-item"><div class="color-box" style="background: #808080;"></div> Unchanged</div>
|
|
39
|
+
</div>
|
|
40
|
+
|
|
41
|
+
<div class="details" id="node-details">
|
|
42
|
+
<p>Click a node to see details.</p>
|
|
43
|
+
</div>
|
|
44
|
+
</div>
|
|
45
|
+
|
|
46
|
+
<script>
|
|
47
|
+
const graphData = {{ graph_json }};
|
|
48
|
+
|
|
49
|
+
const elements = [];
|
|
50
|
+
graphData.nodes.forEach(n => {
|
|
51
|
+
let color = '#808080';
|
|
52
|
+
if (n.status === 'modified') color = '#d7ba7d';
|
|
53
|
+
if (n.status === 'added') color = '#4CAF50';
|
|
54
|
+
|
|
55
|
+
// Add red border if vulnerable
|
|
56
|
+
let borderColor = n.vulnerabilities && n.vulnerabilities.length > 0 ? '#ff0000' : 'transparent';
|
|
57
|
+
let borderWidth = n.vulnerabilities && n.vulnerabilities.length > 0 ? 3 : 0;
|
|
58
|
+
|
|
59
|
+
elements.push({
|
|
60
|
+
data: {
|
|
61
|
+
id: n.id,
|
|
62
|
+
name: n.name,
|
|
63
|
+
kind: n.kind || 'unknown',
|
|
64
|
+
file: n.file || 'unknown',
|
|
65
|
+
status: n.status,
|
|
66
|
+
start_line: n.start_line || '?',
|
|
67
|
+
end_line: n.end_line || '?',
|
|
68
|
+
color: color,
|
|
69
|
+
borderColor: borderColor,
|
|
70
|
+
borderWidth: borderWidth,
|
|
71
|
+
vulnerabilities: n.vulnerabilities || []
|
|
72
|
+
}
|
|
73
|
+
});
|
|
74
|
+
});
|
|
75
|
+
|
|
76
|
+
graphData.edges.forEach(e => {
|
|
77
|
+
elements.push({
|
|
78
|
+
data: {
|
|
79
|
+
source: e.source,
|
|
80
|
+
target: e.target,
|
|
81
|
+
kind: e.kind
|
|
82
|
+
}
|
|
83
|
+
});
|
|
84
|
+
});
|
|
85
|
+
|
|
86
|
+
if (typeof cytoscapeDagre !== 'undefined') {
|
|
87
|
+
cytoscape.use( cytoscapeDagre );
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
const cy = cytoscape({
|
|
91
|
+
container: document.getElementById('cy'),
|
|
92
|
+
elements: elements,
|
|
93
|
+
style: [
|
|
94
|
+
{
|
|
95
|
+
selector: 'node',
|
|
96
|
+
style: {
|
|
97
|
+
'background-color': 'data(color)',
|
|
98
|
+
'border-width': 'data(borderWidth)',
|
|
99
|
+
'border-color': 'data(borderColor)',
|
|
100
|
+
'label': 'data(name)',
|
|
101
|
+
'color': '#fff',
|
|
102
|
+
'text-valign': 'center',
|
|
103
|
+
'text-outline-width': 2,
|
|
104
|
+
'text-outline-color': '#222',
|
|
105
|
+
'font-size': '10px'
|
|
106
|
+
}
|
|
107
|
+
},
|
|
108
|
+
{
|
|
109
|
+
selector: 'edge',
|
|
110
|
+
style: {
|
|
111
|
+
'width': 2,
|
|
112
|
+
'line-color': '#555',
|
|
113
|
+
'target-arrow-color': '#555',
|
|
114
|
+
'target-arrow-shape': 'triangle',
|
|
115
|
+
'curve-style': 'bezier',
|
|
116
|
+
'label': 'data(kind)',
|
|
117
|
+
'font-size': '8px',
|
|
118
|
+
'text-rotation': 'autorotate',
|
|
119
|
+
'color': '#888'
|
|
120
|
+
}
|
|
121
|
+
}
|
|
122
|
+
],
|
|
123
|
+
layout: {
|
|
124
|
+
name: 'dagre'
|
|
125
|
+
}
|
|
126
|
+
});
|
|
127
|
+
|
|
128
|
+
cy.on('tap', 'node', function(evt){
|
|
129
|
+
const node = evt.target;
|
|
130
|
+
const d = node.data();
|
|
131
|
+
let vulnHtml = '';
|
|
132
|
+
if (d.vulnerabilities && d.vulnerabilities.length > 0) {
|
|
133
|
+
vulnHtml = '<h3 style="color:#ff4444; margin-top:10px;">Vulnerabilities Found:</h3>';
|
|
134
|
+
d.vulnerabilities.forEach(v => {
|
|
135
|
+
vulnHtml += `
|
|
136
|
+
<div style="background:#440000; padding:10px; margin-bottom:10px; border-left: 3px solid #ff0000;">
|
|
137
|
+
<strong>${v.title}</strong> (Impact: ${v.impact})<br>
|
|
138
|
+
<p style="margin-top:5px; margin-bottom:0;">${v.description}</p>
|
|
139
|
+
</div>
|
|
140
|
+
`;
|
|
141
|
+
});
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
document.getElementById('node-details').innerHTML = `
|
|
145
|
+
<p><strong>Name:</strong> ${d.name}</p>
|
|
146
|
+
<p><strong>ID:</strong> ${d.id}</p>
|
|
147
|
+
<p><strong>Kind:</strong> ${d.kind}</p>
|
|
148
|
+
<p><strong>Status:</strong> ${d.status}</p>
|
|
149
|
+
<p><strong>File:</strong> ${d.file}</p>
|
|
150
|
+
<p><strong>Lines:</strong> ${d.start_line} - ${d.end_line}</p>
|
|
151
|
+
${vulnHtml}
|
|
152
|
+
`;
|
|
153
|
+
});
|
|
154
|
+
|
|
155
|
+
document.getElementById('btn-dagre').addEventListener('click', () => {
|
|
156
|
+
cy.layout({ name: 'dagre' }).run();
|
|
157
|
+
});
|
|
158
|
+
document.getElementById('btn-cose').addEventListener('click', () => {
|
|
159
|
+
cy.layout({ name: 'cose' }).run();
|
|
160
|
+
});
|
|
161
|
+
|
|
162
|
+
let showUnchanged = true;
|
|
163
|
+
document.getElementById('btn-toggle').addEventListener('click', () => {
|
|
164
|
+
showUnchanged = !showUnchanged;
|
|
165
|
+
if (showUnchanged) {
|
|
166
|
+
cy.nodes('[status = "unchanged"]').show();
|
|
167
|
+
} else {
|
|
168
|
+
cy.nodes('[status = "unchanged"]').hide();
|
|
169
|
+
}
|
|
170
|
+
});
|
|
171
|
+
</script>
|
|
172
|
+
</body>
|
|
173
|
+
</html>
|
|
174
|
+
"""
|
|
175
|
+
|
|
176
|
+
def generate_reports(graph_data: dict, output_dir: str, vulnerabilities: dict = None):
|
|
177
|
+
os.makedirs(output_dir, exist_ok=True)
|
|
178
|
+
|
|
179
|
+
# Attach vulnerabilities to graph_data
|
|
180
|
+
if vulnerabilities:
|
|
181
|
+
for node in graph_data['nodes']:
|
|
182
|
+
if node['id'] in vulnerabilities:
|
|
183
|
+
node['vulnerabilities'] = vulnerabilities[node['id']]
|
|
184
|
+
|
|
185
|
+
json_path = os.path.join(output_dir, "report.json")
|
|
186
|
+
html_path = os.path.join(output_dir, "report.html")
|
|
187
|
+
|
|
188
|
+
with open(json_path, 'w') as f:
|
|
189
|
+
json.dump(graph_data, f, indent=2)
|
|
190
|
+
|
|
191
|
+
template = Template(HTML_TEMPLATE)
|
|
192
|
+
html_content = template.render(graph_json=json.dumps(graph_data))
|
|
193
|
+
|
|
194
|
+
with open(html_path, 'w') as f:
|
|
195
|
+
f.write(html_content)
|
|
196
|
+
|
|
197
|
+
return json_path, html_path
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: zairo
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Git-diff-aware security impact analysis: builds a dependency subgraph around changed code and optionally scans it for vulnerabilities with an LLM.
|
|
5
|
+
Author-email: iamavu <imailavantika@gmail.com>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Repository, https://github.com/iamavu/zairo
|
|
8
|
+
Classifier: Development Status :: 3 - Alpha
|
|
9
|
+
Classifier: Environment :: Console
|
|
10
|
+
Classifier: Intended Audience :: Developers
|
|
11
|
+
Classifier: Programming Language :: Python :: 3
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
15
|
+
Classifier: Topic :: Security
|
|
16
|
+
Classifier: Topic :: Software Development :: Quality Assurance
|
|
17
|
+
Requires-Python: >=3.10
|
|
18
|
+
Description-Content-Type: text/markdown
|
|
19
|
+
License-File: LICENSE
|
|
20
|
+
Requires-Dist: typer>=0.27
|
|
21
|
+
Requires-Dist: rich>=15.0
|
|
22
|
+
Requires-Dist: jinja2>=3.1
|
|
23
|
+
Requires-Dist: litellm>=1.98
|
|
24
|
+
Requires-Dist: trailmark>=0.5
|
|
25
|
+
Provides-Extra: dev
|
|
26
|
+
Requires-Dist: pytest>=8.0; extra == "dev"
|
|
27
|
+
Requires-Dist: build>=1.0; extra == "dev"
|
|
28
|
+
Dynamic: license-file
|
|
29
|
+
|
|
30
|
+
# zairo
|
|
31
|
+
|
|
32
|
+
Git-diff-aware security impact analysis. `zairo` diffs a repository, builds a
|
|
33
|
+
dependency subgraph (via [Trailmark](https://pypi.org/project/trailmark/))
|
|
34
|
+
around whatever changed, and can run an LLM vulnerability scan limited to
|
|
35
|
+
just that changed code — instead of re-scanning the whole codebase on every
|
|
36
|
+
change.
|
|
37
|
+
|
|
38
|
+
Output is a `report.json` (raw graph data) and a self-contained
|
|
39
|
+
`report.html` (interactive dependency graph viewer).
|
|
40
|
+
|
|
41
|
+
## Install
|
|
42
|
+
|
|
43
|
+
```bash
|
|
44
|
+
pip install zairo
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
For local development:
|
|
48
|
+
|
|
49
|
+
```bash
|
|
50
|
+
python -m venv venv
|
|
51
|
+
source venv/bin/activate
|
|
52
|
+
pip install -e ".[dev]"
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
## Usage
|
|
56
|
+
|
|
57
|
+
```bash
|
|
58
|
+
# Analyze uncommitted changes in a repo
|
|
59
|
+
zairo /path/to/repo
|
|
60
|
+
|
|
61
|
+
# Diff two refs, traverse 2 hops out from changed nodes, run an LLM scan
|
|
62
|
+
zairo /path/to/repo --base main --target HEAD --depth 2 --llm
|
|
63
|
+
```
|
|
64
|
+
|
|
65
|
+
Run `zairo --help` for the full option list.
|
|
66
|
+
|
|
67
|
+
## Development
|
|
68
|
+
|
|
69
|
+
```bash
|
|
70
|
+
pytest
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
## License
|
|
74
|
+
|
|
75
|
+
MIT — see [LICENSE](LICENSE).
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
zairo/__init__.py,sha256=kUR5RAFc7HCeiqdlX36dZOHkUI5wI6V_43RpEcD8b-0,22
|
|
2
|
+
zairo/__main__.py,sha256=MSmt_5Xg84uHqzTN38JwgseJK8rsJn_11A8WD99VtEo,61
|
|
3
|
+
zairo/_util.py,sha256=4jrbBrP1SedaK5MvaDG7q7yHxZWR7Z7x6O3RX6Y87Bk,489
|
|
4
|
+
zairo/analyzer.py,sha256=qLLz4jHaYcZUuklnUdrWgOzf7pFXPPhCv2b2jDHO2OE,5344
|
|
5
|
+
zairo/cli.py,sha256=e9rALqWbtKsphPPk5Nc8Q1C_FRu12T6_R24W44lmUZc,6252
|
|
6
|
+
zairo/git_utils.py,sha256=yYvnONGz_aHG1CIQtCfg7YnP-I1zeBBXzqu5B0foc90,5506
|
|
7
|
+
zairo/llm_scanner.py,sha256=G81WvZoyFRbrgY1NloOPYqZ8EfperS60Sa1wUNmqVWo,23202
|
|
8
|
+
zairo/reporter.py,sha256=JIcz3-kYfFN1Vpdk-e5AryR_DqD-5yd8Zxn-RNOvJQo,7850
|
|
9
|
+
zairo-0.1.0.dist-info/licenses/LICENSE,sha256=zhDOLhY1QgM2tk7fM3gvJnuxto_KqXF3Rc_LcRUbJhQ,1075
|
|
10
|
+
zairo-0.1.0.dist-info/METADATA,sha256=ZCItOt4b5CuaOuTNN-JMbzXh73_HABlGR87vU8cF3Mw,2038
|
|
11
|
+
zairo-0.1.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
12
|
+
zairo-0.1.0.dist-info/entry_points.txt,sha256=UhpZuE-UJ416sCicqcjKZRkEqSPM_l_-LxPrqi8j8KM,41
|
|
13
|
+
zairo-0.1.0.dist-info/top_level.txt,sha256=d4m_5jscq13hd1lZWevJ9EV6irrAJ-XgS0VoEJIc3Ys,6
|
|
14
|
+
zairo-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Avantika (@iamavu)
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
zairo
|