precommiteu 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- precommiteu/__init__.py +17 -0
- precommiteu/__main__.py +4 -0
- precommiteu/agents/__init__.py +0 -0
- precommiteu/agents/orchestrator.py +472 -0
- precommiteu/agents/prompts.py +249 -0
- precommiteu/agents/react_loop.py +252 -0
- precommiteu/chunking.py +97 -0
- precommiteu/cli.py +528 -0
- precommiteu/config.py +38 -0
- precommiteu/debuglog.py +19 -0
- precommiteu/defaults.py +37 -0
- precommiteu/detector_call.py +46 -0
- precommiteu/direct.py +91 -0
- precommiteu/grammar.py +57 -0
- precommiteu/llama_server.py +309 -0
- precommiteu/model_factory.py +135 -0
- precommiteu/py.typed +0 -0
- precommiteu/regulations/__init__.py +86 -0
- precommiteu/regulations/cra_dma_nis2/__init__.py +0 -0
- precommiteu/regulations/cra_dma_nis2/cases.jsonl +614 -0
- precommiteu/regulations/cra_dma_nis2/detector_summary.md +275 -0
- precommiteu/regulations/cra_dma_nis2/metadata.json +1 -0
- precommiteu/regulations/cra_dma_nis2/regulations_summary.md +245 -0
- precommiteu/regulations/cra_dma_nis2/validator_summary.md +203 -0
- precommiteu/regulations/dora/__init__.py +0 -0
- precommiteu/regulations/dora/cases.jsonl +474 -0
- precommiteu/regulations/dora/detector_summary.md +290 -0
- precommiteu/regulations/dora/metadata.json +1 -0
- precommiteu/regulations/dora/regulations_summary.md +196 -0
- precommiteu/regulations/dora/validator_summary.md +181 -0
- precommiteu/regulations/dsa/__init__.py +0 -0
- precommiteu/regulations/dsa/cases.jsonl +464 -0
- precommiteu/regulations/dsa/detector_summary.md +312 -0
- precommiteu/regulations/dsa/metadata.json +1 -0
- precommiteu/regulations/dsa/regulations_summary.md +372 -0
- precommiteu/regulations/dsa/validator_summary.md +281 -0
- precommiteu/regulations/eu_ai_act/__init__.py +0 -0
- precommiteu/regulations/eu_ai_act/cases.jsonl +461 -0
- precommiteu/regulations/eu_ai_act/detector_summary.md +332 -0
- precommiteu/regulations/eu_ai_act/metadata.json +1 -0
- precommiteu/regulations/eu_ai_act/regulations_summary.md +383 -0
- precommiteu/regulations/eu_ai_act/validator_summary.md +287 -0
- precommiteu/regulations/eu_data_act/__init__.py +0 -0
- precommiteu/regulations/eu_data_act/cases.jsonl +467 -0
- precommiteu/regulations/eu_data_act/detector_summary.md +284 -0
- precommiteu/regulations/eu_data_act/metadata.json +1 -0
- precommiteu/regulations/eu_data_act/regulations_summary.md +270 -0
- precommiteu/regulations/eu_data_act/validator_summary.md +216 -0
- precommiteu/regulations/gdpr/__init__.py +0 -0
- precommiteu/regulations/gdpr/cases.jsonl +450 -0
- precommiteu/regulations/gdpr/detector_summary.md +288 -0
- precommiteu/regulations/gdpr/metadata.json +1 -0
- precommiteu/regulations/gdpr/regulations_summary.md +387 -0
- precommiteu/regulations/gdpr/validator_summary.md +316 -0
- precommiteu/reporting.py +101 -0
- precommiteu/retrieval.py +168 -0
- precommiteu/scan.py +950 -0
- precommiteu/src/__init__.py +0 -0
- precommiteu/src/chunk_view.py +31 -0
- precommiteu/src/eu_ignore_marker.py +104 -0
- precommiteu/src/file_filter.py +519 -0
- precommiteu/src/ignore_directives.py +94 -0
- precommiteu/src/inference.py +16 -0
- precommiteu/src/progress.py +49 -0
- precommiteu/src/regulations.json +1193 -0
- precommiteu/src/regulations.py +146 -0
- precommiteu/src/reporters/__init__.py +6 -0
- precommiteu/src/reporters/comment.py +68 -0
- precommiteu/src/reporters/json_report.py +15 -0
- precommiteu/src/reporters/jsonl_ledger.py +48 -0
- precommiteu/src/reporters/sarif_report.py +16 -0
- precommiteu/src/sarif.py +138 -0
- precommiteu/src/scanner.py +108 -0
- precommiteu/src/schemas.py +99 -0
- precommiteu/src/tools/__init__.py +0 -0
- precommiteu/src/tools/read_tools.py +286 -0
- precommiteu/src/tools/regulation_tools.py +118 -0
- precommiteu/src/tools/sandbox.py +31 -0
- precommiteu/this.py +28 -0
- precommiteu/tools/__init__.py +0 -0
- precommiteu/tools/detector_tool.py +31 -0
- precommiteu/tools/find_references.py +158 -0
- precommiteu/tools/read_tools.py +19 -0
- precommiteu/tools/regulation_tools.py +9 -0
- precommiteu/tools/sandbox.py +5 -0
- precommiteu/tools/validator_tool.py +88 -0
- precommiteu/validator_call.py +127 -0
- precommiteu-0.1.0.dist-info/METADATA +310 -0
- precommiteu-0.1.0.dist-info/RECORD +94 -0
- precommiteu-0.1.0.dist-info/WHEEL +4 -0
- precommiteu-0.1.0.dist-info/entry_points.txt +2 -0
- precommiteu-0.1.0.dist-info/licenses/LICENSE +201 -0
- precommiteu-0.1.0.dist-info/licenses/NOTICE +8 -0
- precommiteu-0.1.0.dist-info/licenses/THIRD_PARTY_NOTICES.md +40 -0
precommiteu/__init__.py
ADDED
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from precommiteu.scan import GitDiffError, scan_diff, scan_paths
|
|
4
|
+
from precommiteu.src.schemas import Advisory, Finding, ScanResult, ScanStatus
|
|
5
|
+
|
|
6
|
+
__version__ = "0.1.0"
|
|
7
|
+
|
|
8
|
+
__all__ = [
|
|
9
|
+
"Advisory",
|
|
10
|
+
"Finding",
|
|
11
|
+
"GitDiffError",
|
|
12
|
+
"ScanResult",
|
|
13
|
+
"ScanStatus",
|
|
14
|
+
"__version__",
|
|
15
|
+
"scan_diff",
|
|
16
|
+
"scan_paths",
|
|
17
|
+
]
|
precommiteu/__main__.py
ADDED
|
File without changes
|
|
@@ -0,0 +1,472 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import logging
|
|
5
|
+
import os
|
|
6
|
+
import pathlib
|
|
7
|
+
import sys
|
|
8
|
+
import time
|
|
9
|
+
from collections.abc import Callable
|
|
10
|
+
from dataclasses import dataclass, field
|
|
11
|
+
from typing import Any
|
|
12
|
+
|
|
13
|
+
from precommiteu.agents import prompts
|
|
14
|
+
from precommiteu.agents.react_loop import run_react_loop
|
|
15
|
+
from precommiteu.chunking import APPROX_TOKENS, CanonicalChunk, token_chunks
|
|
16
|
+
from precommiteu.config import ENRICHMENT_DEPTH_CAP
|
|
17
|
+
from precommiteu.regulations import get_regulation_pack
|
|
18
|
+
from precommiteu.src.chunk_view import ChunkConsultLog
|
|
19
|
+
from precommiteu.src.ignore_directives import apply_prompt_ignore_directives
|
|
20
|
+
from precommiteu.tools import read_tools, regulation_tools
|
|
21
|
+
from precommiteu.tools.detector_tool import build_call_detector_tool
|
|
22
|
+
from precommiteu.tools.find_references import find_references
|
|
23
|
+
from precommiteu.tools.sandbox import Sandbox
|
|
24
|
+
from precommiteu.tools.validator_tool import (
|
|
25
|
+
CandidatesStore,
|
|
26
|
+
EnrichedCodeStore,
|
|
27
|
+
build_call_validator_tool,
|
|
28
|
+
)
|
|
29
|
+
|
|
30
|
+
_DEBUG_ENRICH = bool(os.environ.get("PRECOMMITEU_DEBUG_ENRICH"))
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _debug_emit(payload: dict[str, Any]) -> None:
|
|
34
|
+
if not _DEBUG_ENRICH:
|
|
35
|
+
return
|
|
36
|
+
try:
|
|
37
|
+
sys.stderr.write("PRECOMMITEU_DEBUG_ENRICH " + json.dumps(payload) + "\n")
|
|
38
|
+
sys.stderr.flush()
|
|
39
|
+
except Exception:
|
|
40
|
+
pass
|
|
41
|
+
|
|
42
|
+
__all__ = [
|
|
43
|
+
"OrchestratorRun",
|
|
44
|
+
"run_orchestrator",
|
|
45
|
+
]
|
|
46
|
+
|
|
47
|
+
_LOG = logging.getLogger(__name__)
|
|
48
|
+
|
|
49
|
+
@dataclass
|
|
50
|
+
class OrchestratorRun:
|
|
51
|
+
kept_findings: list[dict[str, Any]] = field(default_factory=list)
|
|
52
|
+
consult_log: ChunkConsultLog = field(default_factory=ChunkConsultLog)
|
|
53
|
+
candidates_store: CandidatesStore = field(default_factory=CandidatesStore)
|
|
54
|
+
tool_call_count: int = 0
|
|
55
|
+
exit_reason: str = ""
|
|
56
|
+
detector_called: bool = False
|
|
57
|
+
validator_called: bool = False
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _initial_user_message(
|
|
61
|
+
file_path: pathlib.Path,
|
|
62
|
+
file_label: str,
|
|
63
|
+
chunks: list[CanonicalChunk],
|
|
64
|
+
) -> str:
|
|
65
|
+
first = chunks[0] if chunks else None
|
|
66
|
+
first_block = (
|
|
67
|
+
f'<chunk id="{first.id}" lines="{first.start_line}-{first.end_line}">\n'
|
|
68
|
+
f"{first.text}\n"
|
|
69
|
+
"</chunk>"
|
|
70
|
+
if first is not None
|
|
71
|
+
else "(file has no chunks)"
|
|
72
|
+
)
|
|
73
|
+
chunk_index = (
|
|
74
|
+
" (no chunks)"
|
|
75
|
+
if not chunks
|
|
76
|
+
else "\n".join(
|
|
77
|
+
f" - {c.id}: lines {c.start_line}-{c.end_line}" for c in chunks
|
|
78
|
+
)
|
|
79
|
+
)
|
|
80
|
+
return (
|
|
81
|
+
f"File: {file_label}\n"
|
|
82
|
+
f"Absolute path: {file_path}\n\n"
|
|
83
|
+
"Chunk index:\n"
|
|
84
|
+
f"{chunk_index}\n\n"
|
|
85
|
+
"First chunk inlined:\n"
|
|
86
|
+
f"{first_block}\n\n"
|
|
87
|
+
"Begin scanning. EMIT when every detector candidate has been routed "
|
|
88
|
+
"through the validator."
|
|
89
|
+
)
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def _build_tools_map(
|
|
93
|
+
*,
|
|
94
|
+
regulation: str,
|
|
95
|
+
chunks: list[CanonicalChunk],
|
|
96
|
+
sandbox: Sandbox,
|
|
97
|
+
regulation_sandbox: Sandbox,
|
|
98
|
+
regulation_docs_dir: pathlib.Path,
|
|
99
|
+
consult_log: ChunkConsultLog,
|
|
100
|
+
enriched_code_store: EnrichedCodeStore,
|
|
101
|
+
target_file_resolved: str,
|
|
102
|
+
target_file_label: str,
|
|
103
|
+
detector_impl: Callable[..., str],
|
|
104
|
+
validator_impl: Callable[..., str],
|
|
105
|
+
counters: dict[str, int],
|
|
106
|
+
todos: list[str],
|
|
107
|
+
) -> dict[str, Callable[..., Any]]:
|
|
108
|
+
repo_root = sandbox.roots[0]
|
|
109
|
+
enrichment_snippets: list[str] = []
|
|
110
|
+
chunk_state = {"text": chunks[0].text if chunks else ""}
|
|
111
|
+
|
|
112
|
+
def _is_non_target(path_str: str) -> bool:
|
|
113
|
+
try:
|
|
114
|
+
resolved = pathlib.Path(path_str).resolve()
|
|
115
|
+
except OSError:
|
|
116
|
+
return False
|
|
117
|
+
return str(resolved) != target_file_resolved
|
|
118
|
+
|
|
119
|
+
def _enrichment_budget_error() -> str:
|
|
120
|
+
return json.dumps(
|
|
121
|
+
{
|
|
122
|
+
"error": (
|
|
123
|
+
"enrichment_depth_cap_reached: at most "
|
|
124
|
+
f"{ENRICHMENT_DEPTH_CAP} external snippet allowed per "
|
|
125
|
+
"target chunk. Call call_detector with what you have."
|
|
126
|
+
)
|
|
127
|
+
}
|
|
128
|
+
)
|
|
129
|
+
|
|
130
|
+
def _try_increment_for_non_target(path_str: str) -> str | None:
|
|
131
|
+
if not _is_non_target(path_str):
|
|
132
|
+
return None
|
|
133
|
+
if counters["enrichment_depth"] >= ENRICHMENT_DEPTH_CAP:
|
|
134
|
+
return _enrichment_budget_error()
|
|
135
|
+
counters["enrichment_depth"] += 1
|
|
136
|
+
return None
|
|
137
|
+
|
|
138
|
+
def _read_file(path: str, start_line: int = 1, end_line: int = 200) -> str:
|
|
139
|
+
result = read_tools.read_file(
|
|
140
|
+
sandbox, path, start_line=start_line, end_line=end_line
|
|
141
|
+
)
|
|
142
|
+
text = result.get("text", "")
|
|
143
|
+
resolved_path = result.get("path", path)
|
|
144
|
+
block = _try_increment_for_non_target(resolved_path)
|
|
145
|
+
if block is not None:
|
|
146
|
+
return block
|
|
147
|
+
if text:
|
|
148
|
+
key = f"{result['path']}:{result['start_line']}-{result['end_line']}"
|
|
149
|
+
consult_log.record(key, text)
|
|
150
|
+
return json.dumps(result)
|
|
151
|
+
|
|
152
|
+
def _read_chunk(path: str, chunk_id: str) -> str:
|
|
153
|
+
result = read_tools.read_chunk(sandbox, chunks, path, chunk_id)
|
|
154
|
+
resolved_path = result.get("path", path)
|
|
155
|
+
block = _try_increment_for_non_target(resolved_path)
|
|
156
|
+
if block is not None:
|
|
157
|
+
return block
|
|
158
|
+
consult_log.record(result["chunk_id"], result["text"])
|
|
159
|
+
chunk_state["text"] = result["text"]
|
|
160
|
+
return json.dumps(result)
|
|
161
|
+
|
|
162
|
+
def _list_chunks(path: str) -> str:
|
|
163
|
+
return json.dumps(read_tools.list_chunks(chunks, path))
|
|
164
|
+
|
|
165
|
+
def _list_dir(path: str = ".", depth: int = 1) -> str:
|
|
166
|
+
return json.dumps(read_tools.list_dir(sandbox, path, depth=depth))
|
|
167
|
+
|
|
168
|
+
def _glob(pattern: str) -> str:
|
|
169
|
+
return json.dumps(read_tools.glob(sandbox, pattern))
|
|
170
|
+
|
|
171
|
+
def _grep(pattern: str, path: str = ".", file_glob: str = "**/*") -> str:
|
|
172
|
+
return json.dumps(
|
|
173
|
+
read_tools.grep(sandbox, pattern, path=path, file_glob=file_glob)
|
|
174
|
+
)
|
|
175
|
+
|
|
176
|
+
def _find_references(symbol: str) -> str:
|
|
177
|
+
hits = find_references(sandbox, symbol, repo_root)
|
|
178
|
+
non_target_hits = [h for h in hits if _is_non_target(h.get("file", ""))]
|
|
179
|
+
_debug_emit(
|
|
180
|
+
{
|
|
181
|
+
"event": "find_references",
|
|
182
|
+
"target_file": target_file_label,
|
|
183
|
+
"symbol": symbol,
|
|
184
|
+
"hits_total": len(hits),
|
|
185
|
+
"hits_cross_file": len(non_target_hits),
|
|
186
|
+
"hit_files": sorted(
|
|
187
|
+
{h.get("file", "") for h in hits if h.get("file")}
|
|
188
|
+
),
|
|
189
|
+
"depth_before": counters["enrichment_depth"],
|
|
190
|
+
"depth_cap": ENRICHMENT_DEPTH_CAP,
|
|
191
|
+
}
|
|
192
|
+
)
|
|
193
|
+
if non_target_hits:
|
|
194
|
+
if counters["enrichment_depth"] >= ENRICHMENT_DEPTH_CAP:
|
|
195
|
+
return _enrichment_budget_error()
|
|
196
|
+
counters["enrichment_depth"] += 1
|
|
197
|
+
for hit in non_target_hits:
|
|
198
|
+
hit_file = hit.get("file", "")
|
|
199
|
+
try:
|
|
200
|
+
rel = (
|
|
201
|
+
pathlib.Path(hit_file)
|
|
202
|
+
.resolve()
|
|
203
|
+
.relative_to(repo_root)
|
|
204
|
+
.as_posix()
|
|
205
|
+
)
|
|
206
|
+
except (OSError, ValueError):
|
|
207
|
+
rel = pathlib.Path(hit_file).name
|
|
208
|
+
span = f"{hit['start_line']}-{hit['end_line']}"
|
|
209
|
+
snippet = hit.get("snippet", "")
|
|
210
|
+
enrichment_snippets.append(
|
|
211
|
+
f"\n\n# --- enriched: {rel}:{span} ---\n{snippet}"
|
|
212
|
+
)
|
|
213
|
+
consult_log.record(f"{hit_file}:{span}", snippet)
|
|
214
|
+
return json.dumps(hits)
|
|
215
|
+
|
|
216
|
+
def _read_article(article_id: str, summary: bool = False) -> str:
|
|
217
|
+
return json.dumps(
|
|
218
|
+
regulation_tools.read_article(
|
|
219
|
+
regulation_sandbox,
|
|
220
|
+
regulation_docs_dir,
|
|
221
|
+
article_id,
|
|
222
|
+
summary=summary,
|
|
223
|
+
regulation=regulation,
|
|
224
|
+
)
|
|
225
|
+
)
|
|
226
|
+
|
|
227
|
+
def _list_articles() -> str:
|
|
228
|
+
return json.dumps(
|
|
229
|
+
regulation_tools.list_articles(
|
|
230
|
+
regulation_sandbox, regulation_docs_dir, regulation=regulation
|
|
231
|
+
)
|
|
232
|
+
)
|
|
233
|
+
|
|
234
|
+
def _grep_regulation(pattern: str) -> str:
|
|
235
|
+
return json.dumps(
|
|
236
|
+
regulation_tools.grep_regulation(
|
|
237
|
+
regulation_sandbox, regulation_docs_dir, pattern, regulation=regulation
|
|
238
|
+
)
|
|
239
|
+
)
|
|
240
|
+
|
|
241
|
+
def _call_detector(enriched_code: str = "", file_label: str = "") -> str:
|
|
242
|
+
label = file_label or target_file_label
|
|
243
|
+
base = chunk_state["text"] or enriched_code
|
|
244
|
+
composed = base + "".join(enrichment_snippets)
|
|
245
|
+
enriched_code_store.put(label, composed)
|
|
246
|
+
if _DEBUG_ENRICH:
|
|
247
|
+
try:
|
|
248
|
+
target_basename = target_file_label.split("/")[-1]
|
|
249
|
+
files_in_blob = sorted(
|
|
250
|
+
{
|
|
251
|
+
p.name
|
|
252
|
+
for p in pathlib.Path(repo_root).glob("*")
|
|
253
|
+
if (
|
|
254
|
+
p.is_file()
|
|
255
|
+
and p.name in composed
|
|
256
|
+
and p.name != target_basename
|
|
257
|
+
)
|
|
258
|
+
}
|
|
259
|
+
)
|
|
260
|
+
except Exception:
|
|
261
|
+
files_in_blob = []
|
|
262
|
+
_debug_emit(
|
|
263
|
+
{
|
|
264
|
+
"event": "call_detector",
|
|
265
|
+
"file_label": label,
|
|
266
|
+
"target_file": target_file_label,
|
|
267
|
+
"enriched_code_chars": len(composed),
|
|
268
|
+
"enriched_code_tokens": APPROX_TOKENS(composed),
|
|
269
|
+
"source_files_referenced": files_in_blob,
|
|
270
|
+
"enrichment_depth": counters["enrichment_depth"],
|
|
271
|
+
"detector_call_index": counters["detector_calls"],
|
|
272
|
+
}
|
|
273
|
+
)
|
|
274
|
+
enrichment_snippets.clear()
|
|
275
|
+
try:
|
|
276
|
+
result = detector_impl(enriched_code=composed, file_label=label)
|
|
277
|
+
except Exception:
|
|
278
|
+
counters["detector_errors"] += 1
|
|
279
|
+
raise
|
|
280
|
+
counters["detector_calls"] += 1
|
|
281
|
+
return result
|
|
282
|
+
|
|
283
|
+
def _call_validator(article_id_hint: str = "") -> str:
|
|
284
|
+
try:
|
|
285
|
+
result = validator_impl(article_id_hint=article_id_hint)
|
|
286
|
+
except Exception:
|
|
287
|
+
counters["validator_errors"] += 1
|
|
288
|
+
raise
|
|
289
|
+
counters["validator_calls"] += 1
|
|
290
|
+
return result
|
|
291
|
+
|
|
292
|
+
todo_items = todos
|
|
293
|
+
|
|
294
|
+
def _write_todos(todos: str = "[]") -> str:
|
|
295
|
+
try:
|
|
296
|
+
parsed = json.loads(todos)
|
|
297
|
+
except json.JSONDecodeError as exc:
|
|
298
|
+
return json.dumps({"error": f"invalid todos JSON: {exc}"})
|
|
299
|
+
if not isinstance(parsed, list):
|
|
300
|
+
return json.dumps({"error": "todos must be a JSON list of strings"})
|
|
301
|
+
todo_items.clear()
|
|
302
|
+
for item in parsed:
|
|
303
|
+
todo_items.append(str(item))
|
|
304
|
+
return json.dumps({"todos": list(todo_items)})
|
|
305
|
+
|
|
306
|
+
return {
|
|
307
|
+
"read_file": _read_file,
|
|
308
|
+
"read_chunk": _read_chunk,
|
|
309
|
+
"list_chunks": _list_chunks,
|
|
310
|
+
"list_dir": _list_dir,
|
|
311
|
+
"glob": _glob,
|
|
312
|
+
"grep": _grep,
|
|
313
|
+
"find_references": _find_references,
|
|
314
|
+
"read_article": _read_article,
|
|
315
|
+
"list_articles": _list_articles,
|
|
316
|
+
"grep_regulation": _grep_regulation,
|
|
317
|
+
"call_detector": _call_detector,
|
|
318
|
+
"call_validator": _call_validator,
|
|
319
|
+
"write_todos": _write_todos,
|
|
320
|
+
}
|
|
321
|
+
|
|
322
|
+
|
|
323
|
+
def _file_label(file_path: pathlib.Path, sandbox: Sandbox) -> str:
|
|
324
|
+
root = sandbox.roots[0]
|
|
325
|
+
try:
|
|
326
|
+
return file_path.resolve().relative_to(root).as_posix()
|
|
327
|
+
except (OSError, ValueError):
|
|
328
|
+
return file_path.as_posix()
|
|
329
|
+
|
|
330
|
+
|
|
331
|
+
def run_orchestrator(
|
|
332
|
+
*,
|
|
333
|
+
file_path: pathlib.Path,
|
|
334
|
+
regulation: str,
|
|
335
|
+
loop_model: Any,
|
|
336
|
+
detector_model: Any,
|
|
337
|
+
validator_model: Any,
|
|
338
|
+
sandbox: Sandbox,
|
|
339
|
+
regulation_sandbox: Sandbox,
|
|
340
|
+
regulation_docs_dir: pathlib.Path,
|
|
341
|
+
max_iterations: int = 12,
|
|
342
|
+
wall_seconds: float = 90.0,
|
|
343
|
+
) -> OrchestratorRun:
|
|
344
|
+
try:
|
|
345
|
+
regulation_pack = get_regulation_pack(regulation)
|
|
346
|
+
except (ModuleNotFoundError, FileNotFoundError) as exc:
|
|
347
|
+
_LOG.warning("orchestrator: regulation %r not packaged: %s", regulation, exc)
|
|
348
|
+
return OrchestratorRun(exit_reason="unknown_regulation")
|
|
349
|
+
|
|
350
|
+
try:
|
|
351
|
+
text = file_path.read_text(encoding="utf-8", errors="replace")
|
|
352
|
+
except OSError as exc:
|
|
353
|
+
_LOG.warning("orchestrator: cannot read %s: %s", file_path, exc)
|
|
354
|
+
return OrchestratorRun(exit_reason="file_unreadable")
|
|
355
|
+
|
|
356
|
+
text = apply_prompt_ignore_directives(text)
|
|
357
|
+
if text is None:
|
|
358
|
+
return OrchestratorRun(exit_reason="file_ignored")
|
|
359
|
+
|
|
360
|
+
chunks = token_chunks(file_path, text)
|
|
361
|
+
consult_log = ChunkConsultLog()
|
|
362
|
+
if chunks:
|
|
363
|
+
consult_log.record(chunks[0].id, chunks[0].text)
|
|
364
|
+
|
|
365
|
+
file_label = _file_label(file_path, sandbox)
|
|
366
|
+
target_file_resolved = str(file_path.resolve())
|
|
367
|
+
|
|
368
|
+
kept_findings: list[dict[str, Any]] = []
|
|
369
|
+
todos: list[str] = []
|
|
370
|
+
counters = {
|
|
371
|
+
"detector_calls": 0,
|
|
372
|
+
"detector_errors": 0,
|
|
373
|
+
"validator_calls": 0,
|
|
374
|
+
"validator_errors": 0,
|
|
375
|
+
"enrichment_depth": 0,
|
|
376
|
+
}
|
|
377
|
+
_debug_emit(
|
|
378
|
+
{
|
|
379
|
+
"event": "orchestrator_start",
|
|
380
|
+
"file_label": file_label,
|
|
381
|
+
"chunks": len(chunks),
|
|
382
|
+
"chunk_token_sizes": [APPROX_TOKENS(c.text) for c in chunks],
|
|
383
|
+
"first_chunk_id": chunks[0].id if chunks else None,
|
|
384
|
+
"first_chunk_tokens": APPROX_TOKENS(chunks[0].text) if chunks else 0,
|
|
385
|
+
}
|
|
386
|
+
)
|
|
387
|
+
enriched_code_store = EnrichedCodeStore()
|
|
388
|
+
candidates_store = CandidatesStore()
|
|
389
|
+
|
|
390
|
+
started = time.monotonic()
|
|
391
|
+
deadline = started + wall_seconds
|
|
392
|
+
|
|
393
|
+
def _remaining_seconds() -> float:
|
|
394
|
+
return deadline - time.monotonic()
|
|
395
|
+
|
|
396
|
+
detector_impl = build_call_detector_tool(
|
|
397
|
+
detector_model=detector_model,
|
|
398
|
+
candidates_store=candidates_store,
|
|
399
|
+
regulation_pack=regulation_pack,
|
|
400
|
+
remaining_seconds=_remaining_seconds,
|
|
401
|
+
)
|
|
402
|
+
validator_impl = build_call_validator_tool(
|
|
403
|
+
validator_model=validator_model,
|
|
404
|
+
enriched_code_store=enriched_code_store,
|
|
405
|
+
candidates_store=candidates_store,
|
|
406
|
+
file_label_provider=lambda: file_label,
|
|
407
|
+
kept_findings=kept_findings,
|
|
408
|
+
regulation_pack=regulation_pack,
|
|
409
|
+
remaining_seconds=_remaining_seconds,
|
|
410
|
+
)
|
|
411
|
+
|
|
412
|
+
tools_map = _build_tools_map(
|
|
413
|
+
regulation=regulation_pack.name,
|
|
414
|
+
chunks=chunks,
|
|
415
|
+
sandbox=sandbox,
|
|
416
|
+
regulation_sandbox=regulation_sandbox,
|
|
417
|
+
regulation_docs_dir=regulation_docs_dir,
|
|
418
|
+
consult_log=consult_log,
|
|
419
|
+
enriched_code_store=enriched_code_store,
|
|
420
|
+
target_file_resolved=target_file_resolved,
|
|
421
|
+
target_file_label=file_label,
|
|
422
|
+
detector_impl=detector_impl,
|
|
423
|
+
validator_impl=validator_impl,
|
|
424
|
+
counters=counters,
|
|
425
|
+
todos=todos,
|
|
426
|
+
)
|
|
427
|
+
|
|
428
|
+
tool_call_count = 0
|
|
429
|
+
|
|
430
|
+
def _on_step(step: dict[str, Any]) -> None:
|
|
431
|
+
nonlocal tool_call_count
|
|
432
|
+
if step.get("kind") == "action":
|
|
433
|
+
tool_call_count += 1
|
|
434
|
+
|
|
435
|
+
orchestrator_system = prompts.build_orchestrator_system(
|
|
436
|
+
regulation=regulation_pack.name,
|
|
437
|
+
sample_article_id=regulation_pack.sample_article_id,
|
|
438
|
+
)
|
|
439
|
+
_, exit_reason = run_react_loop(
|
|
440
|
+
model=loop_model,
|
|
441
|
+
system_prompt=orchestrator_system,
|
|
442
|
+
initial_user_message=_initial_user_message(file_path, file_label, chunks),
|
|
443
|
+
tools_map=tools_map,
|
|
444
|
+
max_iterations=max_iterations,
|
|
445
|
+
wall_seconds=wall_seconds,
|
|
446
|
+
on_step=_on_step,
|
|
447
|
+
)
|
|
448
|
+
elapsed = time.monotonic() - started
|
|
449
|
+
if exit_reason == "budget_exhausted_time":
|
|
450
|
+
_LOG.info(
|
|
451
|
+
"orchestrator: time budget exhausted after %.2fs on %s",
|
|
452
|
+
elapsed,
|
|
453
|
+
file_label,
|
|
454
|
+
)
|
|
455
|
+
|
|
456
|
+
# A tool-level model failure with no later success means the file was
|
|
457
|
+
# never fully analyzed; report it as a failure, not a clean pass.
|
|
458
|
+
analysis_incomplete = (
|
|
459
|
+
counters["detector_errors"] and not counters["detector_calls"]
|
|
460
|
+
) or (counters["validator_errors"] and not counters["validator_calls"])
|
|
461
|
+
if analysis_incomplete:
|
|
462
|
+
exit_reason = "loop_step_failed"
|
|
463
|
+
|
|
464
|
+
return OrchestratorRun(
|
|
465
|
+
kept_findings=kept_findings,
|
|
466
|
+
consult_log=consult_log,
|
|
467
|
+
candidates_store=candidates_store,
|
|
468
|
+
tool_call_count=tool_call_count,
|
|
469
|
+
exit_reason=exit_reason,
|
|
470
|
+
detector_called=counters["detector_calls"] > 0,
|
|
471
|
+
validator_called=counters["validator_calls"] > 0,
|
|
472
|
+
)
|