cgh-codegen 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cgh_codegen/__init__.py +24 -0
- cgh_codegen/backends.py +133 -0
- cgh_codegen/cli.py +295 -0
- cgh_codegen/flow.py +439 -0
- cgh_codegen/gate.py +73 -0
- cgh_codegen/generate.py +193 -0
- cgh_codegen/logview.py +42 -0
- cgh_codegen/mcp_tools.py +94 -0
- cgh_codegen/picker.py +238 -0
- cgh_codegen/py.typed +0 -0
- cgh_codegen-0.1.0.dist-info/METADATA +91 -0
- cgh_codegen-0.1.0.dist-info/RECORD +15 -0
- cgh_codegen-0.1.0.dist-info/WHEEL +5 -0
- cgh_codegen-0.1.0.dist-info/entry_points.txt +2 -0
- cgh_codegen-0.1.0.dist-info/top_level.txt +1 -0
cgh_codegen/flow.py
ADDED
|
@@ -0,0 +1,439 @@
|
|
|
1
|
+
# -#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#
|
|
2
|
+
# __creation__ = 2026-09-15
|
|
3
|
+
# __author__ = "jndjama (Joy Ndjama)"
|
|
4
|
+
# __copyright__ = "Copyright 2026 ALTIKVA."
|
|
5
|
+
# __licence__ = "MIT"
|
|
6
|
+
# -#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#
|
|
7
|
+
# Description: run_generation ties the pieces together: pick the reference,
|
|
8
|
+
# clear it through the egress gate (unless the backend is local),
|
|
9
|
+
# generate against it, and write the target. The safety rails the
|
|
10
|
+
# council asked for live here: a cloud egress is refused on a
|
|
11
|
+
# gated reference, the target is confined to the repo and never
|
|
12
|
+
# clobbered without force, and the generated bytes are a proposal
|
|
13
|
+
# to verify, never trusted because a later lint run is green.
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
from pathlib import Path
|
|
18
|
+
|
|
19
|
+
from .gate import egress_decision
|
|
20
|
+
from .generate import Backend, generate_code
|
|
21
|
+
from .picker import CodegenError, _confine, pick_reference
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _select_reference(
|
|
25
|
+
root: Path, pick: dict, backend: Backend, config: dict
|
|
26
|
+
) -> tuple[str, Path, str, str | None]:
|
|
27
|
+
"""Choose the reference to actually mirror, honoring the egress gate.
|
|
28
|
+
|
|
29
|
+
For a cloud backend the top-ranked pick may carry a confidential or PII
|
|
30
|
+
finding, which the gate refuses. Rather than fail the whole generation,
|
|
31
|
+
walk the ranked candidates and take the first that clears the gate, so an
|
|
32
|
+
auto-pick lands on a sendable reference instead of stopping on the first
|
|
33
|
+
one that happens to hold a secret. A local backend never leaves the
|
|
34
|
+
machine, so it skips the gate and keeps the top pick.
|
|
35
|
+
|
|
36
|
+
Returns ``(ref_rel, ref_path, egress_reason, note)``. ``note`` is a human
|
|
37
|
+
string when the choice fell back off the top pick (so the caller can say
|
|
38
|
+
why the mirrored file is not the one first proposed), else None. Raises
|
|
39
|
+
CodegenError when no candidate is usable.
|
|
40
|
+
"""
|
|
41
|
+
ref_top = pick["reference"]
|
|
42
|
+
if ref_top is None:
|
|
43
|
+
raise CodegenError(pick["reason"])
|
|
44
|
+
|
|
45
|
+
def _readable(rel: str) -> Path | None:
|
|
46
|
+
p = _confine(root, rel)
|
|
47
|
+
return p if (p is not None and p.is_file()) else None
|
|
48
|
+
|
|
49
|
+
if backend.is_local:
|
|
50
|
+
p = _readable(ref_top)
|
|
51
|
+
if p is None:
|
|
52
|
+
raise CodegenError(f"reference {ref_top!r} is not a readable file")
|
|
53
|
+
return ref_top, p, "local backend (no egress)", None
|
|
54
|
+
|
|
55
|
+
# Cloud backend: the reference's bytes leave the machine, so every
|
|
56
|
+
# candidate must clear the gate. Try them in rank order; the top pick is
|
|
57
|
+
# candidates[0], so an ungated top pick is chosen with no fallback note.
|
|
58
|
+
ordered = pick["candidates"] or [ref_top]
|
|
59
|
+
denials: list[str] = []
|
|
60
|
+
for cand in ordered:
|
|
61
|
+
p = _readable(cand)
|
|
62
|
+
if p is None:
|
|
63
|
+
denials.append(f"{cand}: not a readable file")
|
|
64
|
+
continue
|
|
65
|
+
allowed, reason = egress_decision(root, p, config)
|
|
66
|
+
if allowed:
|
|
67
|
+
note = None
|
|
68
|
+
if cand != ref_top:
|
|
69
|
+
note = (
|
|
70
|
+
f"top pick {ref_top} was refused by the egress gate; "
|
|
71
|
+
f"mirrored the next clear candidate {cand} instead"
|
|
72
|
+
)
|
|
73
|
+
return cand, p, reason, note
|
|
74
|
+
denials.append(f"{cand}: {reason}")
|
|
75
|
+
_audit(root, "codegen_egress_denied", f"{cand}: {reason}")
|
|
76
|
+
|
|
77
|
+
if len(denials) == 1:
|
|
78
|
+
raise CodegenError(f"egress refused for reference {denials[0]}")
|
|
79
|
+
raise CodegenError(
|
|
80
|
+
"every candidate reference was refused by the egress gate; "
|
|
81
|
+
+ "; ".join(denials)
|
|
82
|
+
)
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def run_generation(
|
|
86
|
+
repo_root: str | Path,
|
|
87
|
+
spec: str,
|
|
88
|
+
target: str,
|
|
89
|
+
reference: str | None = None,
|
|
90
|
+
*,
|
|
91
|
+
config: dict,
|
|
92
|
+
backend: Backend,
|
|
93
|
+
force: bool = False,
|
|
94
|
+
to_stdout: bool = False,
|
|
95
|
+
verify: str | None = None,
|
|
96
|
+
max_attempts: int = 1,
|
|
97
|
+
extend: bool = False,
|
|
98
|
+
) -> dict:
|
|
99
|
+
"""Generate the code for ``target`` and, unless ``to_stdout``, write it.
|
|
100
|
+
|
|
101
|
+
With ``verify`` (a shell check, exit 0 = pass) and ``max_attempts`` > 1,
|
|
102
|
+
it self-corrects: write, run the check, and on failure feed the check's
|
|
103
|
+
output plus the previous attempt back to the model and regenerate, up to
|
|
104
|
+
``max_attempts``. That is how the loop catches what a first pass misses
|
|
105
|
+
(a generated backend that forgot to commit fails the check, and the
|
|
106
|
+
failure text drives the fix), turning "verify, don't trust" into a
|
|
107
|
+
closed loop instead of a manual review.
|
|
108
|
+
|
|
109
|
+
With ``extend``, the target must already exist and is added to rather than
|
|
110
|
+
written over: the model receives the current file and returns only the
|
|
111
|
+
block to append. That is the safe way to grow a file, since the existing
|
|
112
|
+
content is never in the model's output and so cannot be dropped by a
|
|
113
|
+
careless regeneration. The file is then its own style reference, and no
|
|
114
|
+
sibling is picked unless one is named explicitly.
|
|
115
|
+
|
|
116
|
+
Returns a result dict. Raises CodegenError for a bad target, a gated
|
|
117
|
+
reference, or a refused overwrite.
|
|
118
|
+
"""
|
|
119
|
+
root = Path(repo_root).resolve()
|
|
120
|
+
|
|
121
|
+
tgt_probe = _confine(root, target)
|
|
122
|
+
if tgt_probe is None:
|
|
123
|
+
raise CodegenError(f"target {target!r} is outside the repo")
|
|
124
|
+
|
|
125
|
+
existing_text: str | None = None
|
|
126
|
+
if extend:
|
|
127
|
+
if not tgt_probe.is_file():
|
|
128
|
+
raise CodegenError(
|
|
129
|
+
f"target {target} does not exist; generate it before extending it"
|
|
130
|
+
)
|
|
131
|
+
existing_text = tgt_probe.read_text(encoding="utf-8", errors="replace")
|
|
132
|
+
|
|
133
|
+
refs: list[tuple[str, str]] = []
|
|
134
|
+
if extend and reference is None:
|
|
135
|
+
# The file being extended is its own style example, and it already goes
|
|
136
|
+
# to the model as the text to append to, so no sibling is picked and
|
|
137
|
+
# there is nothing to fall back to. Its bytes still leave the machine,
|
|
138
|
+
# so it still clears the gate.
|
|
139
|
+
pick = {
|
|
140
|
+
"reference": None,
|
|
141
|
+
"reason": "extending the target, which is its own reference",
|
|
142
|
+
"graph_available": False,
|
|
143
|
+
}
|
|
144
|
+
ref_rel, ref_path, ref_note = None, tgt_probe, None
|
|
145
|
+
if backend.is_local:
|
|
146
|
+
egress = "local backend (no egress)"
|
|
147
|
+
else:
|
|
148
|
+
allowed, reason = egress_decision(root, tgt_probe, config)
|
|
149
|
+
if not allowed:
|
|
150
|
+
_audit(root, "codegen_egress_denied", f"{target}: {reason}")
|
|
151
|
+
raise CodegenError(f"egress refused for {target}: {reason}")
|
|
152
|
+
egress = reason
|
|
153
|
+
else:
|
|
154
|
+
# Bound graph-based reference selection so a wedged owner cannot hang
|
|
155
|
+
# the whole call; config can tune the ceiling.
|
|
156
|
+
graph_timeout = config.get("reference_timeout_s")
|
|
157
|
+
pick = pick_reference(
|
|
158
|
+
root,
|
|
159
|
+
target,
|
|
160
|
+
reference,
|
|
161
|
+
graph_timeout=float(graph_timeout) if graph_timeout is not None else None,
|
|
162
|
+
)
|
|
163
|
+
ref_rel, ref_path, egress, ref_note = _select_reference(
|
|
164
|
+
root, pick, backend, config
|
|
165
|
+
)
|
|
166
|
+
refs.append((ref_rel, ref_path.read_text(encoding="utf-8", errors="replace")))
|
|
167
|
+
|
|
168
|
+
# Clobber check once, up front; the retry loop then overwrites its own
|
|
169
|
+
# attempts freely (verify needs the file on disk to check it). Extending
|
|
170
|
+
# never clobbers, so it is exempt.
|
|
171
|
+
tgt = None
|
|
172
|
+
if not to_stdout:
|
|
173
|
+
tgt = tgt_probe
|
|
174
|
+
if tgt.exists() and not extend:
|
|
175
|
+
if not force:
|
|
176
|
+
raise CodegenError(
|
|
177
|
+
f"target {target} already exists; pass force to overwrite"
|
|
178
|
+
)
|
|
179
|
+
# Force refreshes codegen's OWN output freely, but must never
|
|
180
|
+
# silently discard edits a human made since codegen last wrote the
|
|
181
|
+
# file. If we recorded what we wrote and the file no longer matches
|
|
182
|
+
# it, someone changed it by hand, so refuse rather than clobber.
|
|
183
|
+
recorded = _load_provenance(root).get(_rel(root, tgt))
|
|
184
|
+
if recorded is not None and (
|
|
185
|
+
_sha(tgt.read_text(encoding="utf-8", errors="replace")) != recorded
|
|
186
|
+
):
|
|
187
|
+
raise CodegenError(
|
|
188
|
+
f"target {target} was modified since codegen last generated "
|
|
189
|
+
"it (it looks hand-edited); refusing to overwrite and discard "
|
|
190
|
+
"those changes. Use extend to add to it, or delete the file "
|
|
191
|
+
"to regenerate it from scratch."
|
|
192
|
+
)
|
|
193
|
+
|
|
194
|
+
do_verify = bool(verify) and not to_stdout
|
|
195
|
+
prior: tuple[str, str] | None = None
|
|
196
|
+
verified: bool | None = None
|
|
197
|
+
written = False
|
|
198
|
+
attempts = 0
|
|
199
|
+
result = None
|
|
200
|
+
code = ""
|
|
201
|
+
while attempts < max(1, max_attempts):
|
|
202
|
+
attempts += 1
|
|
203
|
+
result = generate_code(
|
|
204
|
+
spec,
|
|
205
|
+
refs,
|
|
206
|
+
backend,
|
|
207
|
+
target=target,
|
|
208
|
+
prior=prior,
|
|
209
|
+
existing=existing_text,
|
|
210
|
+
)
|
|
211
|
+
# Hand the cheap model's output through ruff so what lands on disk is
|
|
212
|
+
# already formatted and lint-clean, rather than leaving trailing
|
|
213
|
+
# whitespace and unused imports for a human to sweep up. Best-effort:
|
|
214
|
+
# a whole new file also gets the safe autofixes; an appended block is
|
|
215
|
+
# only reformatted, since ruff's unused-import view of a fragment out
|
|
216
|
+
# of its file's context is not reliable. No ruff on PATH, a non-Python
|
|
217
|
+
# target, or a syntax error ruff cannot parse leaves the code as-is.
|
|
218
|
+
code = _polish_code(result.code, target, root, full_module=not extend)
|
|
219
|
+
if to_stdout or tgt is None:
|
|
220
|
+
break
|
|
221
|
+
if existing_text is not None:
|
|
222
|
+
# Rebuild from the original every attempt, so a retry appends to
|
|
223
|
+
# the file as it was rather than stacking onto its own last try.
|
|
224
|
+
body = _append_into(existing_text, code)
|
|
225
|
+
else:
|
|
226
|
+
body = code if code.endswith("\n") else code + "\n"
|
|
227
|
+
tgt.parent.mkdir(parents=True, exist_ok=True)
|
|
228
|
+
tgt.write_text(body, encoding="utf-8")
|
|
229
|
+
written = True
|
|
230
|
+
if not do_verify:
|
|
231
|
+
break
|
|
232
|
+
ok, output = _run_verify(root, verify) # type: ignore[arg-type]
|
|
233
|
+
verified = ok
|
|
234
|
+
if ok:
|
|
235
|
+
break
|
|
236
|
+
prior = (code, output)
|
|
237
|
+
|
|
238
|
+
# Extending must never leave the file worse than it was found. A new file
|
|
239
|
+
# that fails its check is a draft worth reading; an existing file that
|
|
240
|
+
# fails one has been damaged, and the last attempt can be truncated or
|
|
241
|
+
# unparsable. So put the original back and report the failure instead.
|
|
242
|
+
rolled_back = False
|
|
243
|
+
if extend and written and do_verify and not verified and existing_text is not None:
|
|
244
|
+
tgt.write_text(existing_text, encoding="utf-8") # type: ignore[union-attr]
|
|
245
|
+
written = False
|
|
246
|
+
rolled_back = True
|
|
247
|
+
|
|
248
|
+
# Remember what codegen wrote (a full generation, not an extend, which
|
|
249
|
+
# carries human content), so a later force-overwrite can tell its own
|
|
250
|
+
# untouched output from a file a human has since edited. Best-effort.
|
|
251
|
+
if written and not extend and tgt is not None:
|
|
252
|
+
_record_provenance(
|
|
253
|
+
root, _rel(root, tgt), tgt.read_text(encoding="utf-8", errors="replace")
|
|
254
|
+
)
|
|
255
|
+
|
|
256
|
+
if result is None: # unreachable: the loop always runs at least once
|
|
257
|
+
raise CodegenError("generation produced no result")
|
|
258
|
+
_audit(
|
|
259
|
+
root,
|
|
260
|
+
"codegen_extended" if extend else "codegen_generated",
|
|
261
|
+
f"{target} <- {ref_rel or '(itself)'} ({backend.name}, "
|
|
262
|
+
f"{attempts} attempt(s), verified={verified})",
|
|
263
|
+
)
|
|
264
|
+
|
|
265
|
+
return {
|
|
266
|
+
"target": target,
|
|
267
|
+
"reference": ref_rel,
|
|
268
|
+
"reason": ref_note or pick["reason"],
|
|
269
|
+
"ref_fallback": ref_note,
|
|
270
|
+
"lines": len(code.splitlines()),
|
|
271
|
+
"cost": result.cost,
|
|
272
|
+
"backend": result.backend,
|
|
273
|
+
"egress": egress,
|
|
274
|
+
"written": written,
|
|
275
|
+
"graph_available": pick["graph_available"],
|
|
276
|
+
"attempts": attempts,
|
|
277
|
+
"verified": verified,
|
|
278
|
+
"extended": extend,
|
|
279
|
+
"rolled_back": rolled_back,
|
|
280
|
+
"code": code if to_stdout else "",
|
|
281
|
+
}
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
def _append_into(existing: str, block: str) -> str:
|
|
285
|
+
"""Put ``block`` at the end of ``existing``, but before a trailing
|
|
286
|
+
`if __name__ == "__main__":` guard.
|
|
287
|
+
|
|
288
|
+
That guard is a runner, not content, and it is conventionally the last
|
|
289
|
+
thing in the file. Appending after it works but reads as a mistake, and
|
|
290
|
+
for a module that does real work under the guard it would leave the new
|
|
291
|
+
code unreachable from a direct run.
|
|
292
|
+
"""
|
|
293
|
+
lines = existing.rstrip("\n").split("\n")
|
|
294
|
+
cut = len(lines)
|
|
295
|
+
for i, line in enumerate(lines):
|
|
296
|
+
if line.startswith('if __name__ == "__main__":') or line.startswith(
|
|
297
|
+
"if __name__ == '__main__':"
|
|
298
|
+
):
|
|
299
|
+
cut = i
|
|
300
|
+
break
|
|
301
|
+
|
|
302
|
+
head = "\n".join(lines[:cut]).rstrip("\n")
|
|
303
|
+
tail = "\n".join(lines[cut:]).strip("\n")
|
|
304
|
+
body = head + "\n\n" + block.strip("\n") + "\n"
|
|
305
|
+
if tail:
|
|
306
|
+
body += "\n" + tail + "\n"
|
|
307
|
+
return body
|
|
308
|
+
|
|
309
|
+
|
|
310
|
+
def _polish_code(code: str, target: str, root: Path, *, full_module: bool) -> str:
|
|
311
|
+
"""Run generated Python through ruff so it lands formatted and lint-clean.
|
|
312
|
+
|
|
313
|
+
A whole new file (``full_module``) is both formatted and given ruff's safe
|
|
314
|
+
autofixes (drop an unused import, normalise quotes); an appended block is
|
|
315
|
+
only formatted, because ruff cannot judge an out-of-context fragment's
|
|
316
|
+
unused imports. A non-Python target, no ruff on PATH, or code ruff cannot
|
|
317
|
+
parse are returned unchanged, so this never blocks or corrupts a run.
|
|
318
|
+
"""
|
|
319
|
+
import shutil
|
|
320
|
+
|
|
321
|
+
if not target.endswith(".py") or shutil.which("ruff") is None:
|
|
322
|
+
return code
|
|
323
|
+
formatted = _ruff_run(["format", "--stdin-filename", target, "-"], code, root, (0,))
|
|
324
|
+
code = formatted or code
|
|
325
|
+
if full_module:
|
|
326
|
+
fixed = _ruff_run(
|
|
327
|
+
["check", "--fix", "--stdin-filename", target, "-"], code, root, (0, 1)
|
|
328
|
+
)
|
|
329
|
+
code = fixed or code
|
|
330
|
+
return code
|
|
331
|
+
|
|
332
|
+
|
|
333
|
+
def _ruff_run(
|
|
334
|
+
args: list[str], code: str, root: Path, ok_codes: tuple[int, ...]
|
|
335
|
+
) -> str | None:
|
|
336
|
+
"""Feed ``code`` to ``ruff <args>`` on stdin and return the transformed
|
|
337
|
+
source, or None when ruff was not usable. ``check --fix`` exits 1 when lint
|
|
338
|
+
issues remain but still emits the fixed source, so its caller passes
|
|
339
|
+
``(0, 1)``; ``format`` passes ``(0,)``. Any other exit (a parse error, ruff
|
|
340
|
+
missing) yields None so the caller keeps the input unchanged."""
|
|
341
|
+
import subprocess
|
|
342
|
+
|
|
343
|
+
from codegraph.plugin_api import quiet_subprocess_kwargs
|
|
344
|
+
|
|
345
|
+
try:
|
|
346
|
+
proc = subprocess.run(
|
|
347
|
+
["ruff", *args],
|
|
348
|
+
input=code,
|
|
349
|
+
capture_output=True,
|
|
350
|
+
text=True,
|
|
351
|
+
cwd=str(root),
|
|
352
|
+
timeout=30,
|
|
353
|
+
check=False,
|
|
354
|
+
**quiet_subprocess_kwargs(),
|
|
355
|
+
)
|
|
356
|
+
except (OSError, subprocess.SubprocessError):
|
|
357
|
+
return None
|
|
358
|
+
if proc.returncode in ok_codes and proc.stdout:
|
|
359
|
+
return proc.stdout
|
|
360
|
+
return None
|
|
361
|
+
|
|
362
|
+
|
|
363
|
+
def _run_verify(root: Path, command: str) -> tuple[bool, str]:
|
|
364
|
+
"""Run the verify command in ``root``; True on exit 0. The captured output
|
|
365
|
+
(capped) is fed back to the model on failure. argv-only, timed, no shell."""
|
|
366
|
+
import shlex
|
|
367
|
+
import subprocess
|
|
368
|
+
|
|
369
|
+
from codegraph.plugin_api import quiet_subprocess_kwargs
|
|
370
|
+
|
|
371
|
+
argv = shlex.split(command)
|
|
372
|
+
if not argv:
|
|
373
|
+
return True, ""
|
|
374
|
+
try:
|
|
375
|
+
proc = subprocess.run(
|
|
376
|
+
argv,
|
|
377
|
+
cwd=str(root),
|
|
378
|
+
capture_output=True,
|
|
379
|
+
text=True,
|
|
380
|
+
timeout=180,
|
|
381
|
+
check=False,
|
|
382
|
+
**quiet_subprocess_kwargs(),
|
|
383
|
+
)
|
|
384
|
+
except (subprocess.TimeoutExpired, FileNotFoundError, OSError) as exc:
|
|
385
|
+
return False, f"the verify command could not run: {exc}"
|
|
386
|
+
return proc.returncode == 0, (proc.stdout + proc.stderr)[-4000:]
|
|
387
|
+
|
|
388
|
+
|
|
389
|
+
def _audit(root: Path, event: str, detail: str) -> None:
|
|
390
|
+
from codegraph.plugin_api import activity_log
|
|
391
|
+
|
|
392
|
+
activity_log(root, event, detail)
|
|
393
|
+
|
|
394
|
+
|
|
395
|
+
def _rel(root: Path, tgt: Path) -> str:
|
|
396
|
+
"""The target's key in the provenance store: its path relative to the repo
|
|
397
|
+
root, so the record survives being read from a different working directory.
|
|
398
|
+
Falls back to the absolute path if the target somehow sits outside root."""
|
|
399
|
+
try:
|
|
400
|
+
return tgt.resolve().relative_to(root.resolve()).as_posix()
|
|
401
|
+
except ValueError:
|
|
402
|
+
return str(tgt.resolve())
|
|
403
|
+
|
|
404
|
+
|
|
405
|
+
def _sha(text: str) -> str:
|
|
406
|
+
import hashlib
|
|
407
|
+
|
|
408
|
+
return hashlib.sha256(text.encode("utf-8")).hexdigest()
|
|
409
|
+
|
|
410
|
+
|
|
411
|
+
def _provenance_path(root: Path) -> Path:
|
|
412
|
+
return root / ".codegraph" / "codegen" / "provenance.json"
|
|
413
|
+
|
|
414
|
+
|
|
415
|
+
def _load_provenance(root: Path) -> dict:
|
|
416
|
+
import json
|
|
417
|
+
|
|
418
|
+
try:
|
|
419
|
+
return json.loads(_provenance_path(root).read_text(encoding="utf-8"))
|
|
420
|
+
except (OSError, ValueError):
|
|
421
|
+
return {}
|
|
422
|
+
|
|
423
|
+
|
|
424
|
+
def _record_provenance(root: Path, rel: str, body: str) -> None:
|
|
425
|
+
"""Record the sha256 of the bytes codegen just wrote for ``rel``. Best-effort:
|
|
426
|
+
a store that cannot be written just means the next force-overwrite falls back
|
|
427
|
+
to the old behaviour, never a failed generation."""
|
|
428
|
+
import json
|
|
429
|
+
|
|
430
|
+
path = _provenance_path(root)
|
|
431
|
+
try:
|
|
432
|
+
data = _load_provenance(root)
|
|
433
|
+
data[rel] = _sha(body)
|
|
434
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
435
|
+
path.write_text(
|
|
436
|
+
json.dumps(data, indent=2, sort_keys=True) + "\n", encoding="utf-8"
|
|
437
|
+
)
|
|
438
|
+
except OSError:
|
|
439
|
+
pass
|
cgh_codegen/gate.py
ADDED
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
# -#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#
|
|
2
|
+
# __creation__ = 2026-09-15
|
|
3
|
+
# __author__ = "jndjama (Joy Ndjama)"
|
|
4
|
+
# __copyright__ = "Copyright 2026 ALTIKVA."
|
|
5
|
+
# __licence__ = "MIT"
|
|
6
|
+
# -#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#-#
|
|
7
|
+
# Description: The egress decision for cgh-codegen. A reference file is
|
|
8
|
+
# source that leaves the machine to reach a cloud model, so it
|
|
9
|
+
# must clear the same confidentiality / PII / severity checks the
|
|
10
|
+
# rest of cgh applies. The whole verdict lives in ONE function,
|
|
11
|
+
# egress_decision, so there is a single place to audit and, later,
|
|
12
|
+
# a single call site to swap for a shared core decision.
|
|
13
|
+
#
|
|
14
|
+
# Consolidation note: this mirrors the egress logic the
|
|
15
|
+
# cgh-summarize plugin also carries. Two copies of a security
|
|
16
|
+
# verdict drift toward the leak, so the intended end state is one
|
|
17
|
+
# shared decision in cgh core that every model-backed plugin
|
|
18
|
+
# calls. Until that lands, keep this a single function, never
|
|
19
|
+
# re-derive the verdict inline elsewhere in the plugin.
|
|
20
|
+
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
from pathlib import Path
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def egress_posture(repo_root: str | Path, config: dict) -> str:
|
|
27
|
+
""" "open" or "strict". An explicit config egress key wins; otherwise the
|
|
28
|
+
global cgh mode decides (secure = strict)."""
|
|
29
|
+
explicit = str(config.get("egress", "")).strip().lower()
|
|
30
|
+
if explicit in ("open", "strict"):
|
|
31
|
+
return explicit
|
|
32
|
+
from codegraph.plugin_api import load_config
|
|
33
|
+
|
|
34
|
+
return "strict" if load_config(repo_root).mode == "secure" else "open"
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def egress_decision(
|
|
38
|
+
repo_root: str | Path, file_path: str | Path, config: dict
|
|
39
|
+
) -> tuple[bool, str]:
|
|
40
|
+
"""May this file's content be sent to a cloud model? Returns
|
|
41
|
+
(allowed, reason); the reason is meant for the audit line on a deny so
|
|
42
|
+
"why was this reference refused" has an answer.
|
|
43
|
+
|
|
44
|
+
This is the single source of the verdict for the plugin. Do not inline
|
|
45
|
+
any part of it elsewhere: a second copy of a security decision is how a
|
|
46
|
+
confidential file eventually slips out while the log still says audited.
|
|
47
|
+
"""
|
|
48
|
+
from codegraph.plugin_api import findings_for_file
|
|
49
|
+
|
|
50
|
+
rows = findings_for_file(repo_root, str(file_path))
|
|
51
|
+
|
|
52
|
+
confidential: bool | None = None
|
|
53
|
+
for row in rows:
|
|
54
|
+
if row["key"] == "confidential":
|
|
55
|
+
confidential = str(row["value"]).strip().lower() in ("true", "yes", "1")
|
|
56
|
+
if confidential:
|
|
57
|
+
return False, "confidential finding"
|
|
58
|
+
|
|
59
|
+
for row in rows:
|
|
60
|
+
if row["severity"] == "block":
|
|
61
|
+
return False, f"block-severity finding {row['key']}"
|
|
62
|
+
|
|
63
|
+
if not config.get("allow_pii", False):
|
|
64
|
+
for row in rows:
|
|
65
|
+
if row["key"].startswith("pii."):
|
|
66
|
+
return False, f"pii finding {row['key']} (allow_pii = false)"
|
|
67
|
+
|
|
68
|
+
if egress_posture(repo_root, config) == "strict":
|
|
69
|
+
if confidential is False:
|
|
70
|
+
return True, "labeled non-confidential"
|
|
71
|
+
return False, "strict posture: file not labeled non-confidential"
|
|
72
|
+
|
|
73
|
+
return True, "gate clear"
|