ai-code-engineer 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ai_code_engineer/__init__.py +2 -0
- ai_code_engineer/catalog.py +143 -0
- ai_code_engineer/chat.py +181 -0
- ai_code_engineer/cli.py +384 -0
- ai_code_engineer/config.py +405 -0
- ai_code_engineer/engine.py +1282 -0
- ai_code_engineer/errors.py +27 -0
- ai_code_engineer/git_integration.py +443 -0
- ai_code_engineer/gui.py +2646 -0
- ai_code_engineer/host.py +81 -0
- ai_code_engineer/ignore.py +269 -0
- ai_code_engineer/intent.py +222 -0
- ai_code_engineer/labels.py +871 -0
- ai_code_engineer/memory.py +91 -0
- ai_code_engineer/modes.py +156 -0
- ai_code_engineer/overrides.py +540 -0
- ai_code_engineer/planbook.py +192 -0
- ai_code_engineer/providers.py +404 -0
- ai_code_engineer/redaction.py +54 -0
- ai_code_engineer/repair.py +564 -0
- ai_code_engineer/report.py +352 -0
- ai_code_engineer/runner.py +854 -0
- ai_code_engineer/setup.py +386 -0
- ai_code_engineer/symbols.py +1286 -0
- ai_code_engineer/verification.py +218 -0
- ai_code_engineer/webapp/__init__.py +1 -0
- ai_code_engineer/webapp/__main__.py +45 -0
- ai_code_engineer/webapp/contract.py +36 -0
- ai_code_engineer/webapp/controller.py +3556 -0
- ai_code_engineer/webapp/fake.py +1141 -0
- ai_code_engineer/webapp/launch.py +108 -0
- ai_code_engineer/webapp/server.py +349 -0
- ai_code_engineer/webapp/static/app.css +780 -0
- ai_code_engineer/webapp/static/app.js +2118 -0
- ai_code_engineer/webapp/static/boot.js +19 -0
- ai_code_engineer/webapp/static/index.html +89 -0
- ai_code_engineer/webapp/static/tokens.css +173 -0
- ai_code_engineer/workspace.py +385 -0
- ai_code_engineer-0.1.0.dist-info/METADATA +7 -0
- ai_code_engineer-0.1.0.dist-info/RECORD +44 -0
- ai_code_engineer-0.1.0.dist-info/WHEEL +5 -0
- ai_code_engineer-0.1.0.dist-info/entry_points.txt +2 -0
- ai_code_engineer-0.1.0.dist-info/licenses/LICENSE +21 -0
- ai_code_engineer-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,192 @@
|
|
|
1
|
+
"""Ordered plan steps with a ledger, so a phase gates on real proof instead of memory.
|
|
2
|
+
|
|
3
|
+
An attached plan used to be reference text: the model was told to do "the phase the user
|
|
4
|
+
asked for", and the sequence lived in whoever typed the next request. This module makes
|
|
5
|
+
the order a fact on disk. A step is complete only when a command run proved that tests
|
|
6
|
+
executed and none failed, and the next step's task is generated from what is already done.
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import json
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
import re
|
|
13
|
+
|
|
14
|
+
from .engine import atomic_json
|
|
15
|
+
from .errors import PolicyError
|
|
16
|
+
from . import memory as memory_store
|
|
17
|
+
from .workspace import Workspace
|
|
18
|
+
|
|
19
|
+
MAX_STEPS = 12
|
|
20
|
+
STEP_HEADING = re.compile(
|
|
21
|
+
r"(?m)^\s{0,3}#{1,6}\s*(?:(?:phase|step|task|stage)s*\s+)?\b(\d+)\b\W*(?P<title>[^\n]*)$", re.I)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def parse_steps(text: str) -> list[dict]:
|
|
25
|
+
"""Numbered headings become steps; anything else is one undivided step."""
|
|
26
|
+
matches = list(STEP_HEADING.finditer(text))
|
|
27
|
+
if len(matches) < 2:
|
|
28
|
+
body = text.strip()
|
|
29
|
+
if not body:
|
|
30
|
+
raise PolicyError("The plan has no steps to run.")
|
|
31
|
+
return [{"id": 1, "title": body.splitlines()[0][:90], "body": body[:6000]}]
|
|
32
|
+
steps = []
|
|
33
|
+
for index, match in enumerate(matches):
|
|
34
|
+
end = matches[index + 1].start() if index + 1 < len(matches) else len(text)
|
|
35
|
+
title = (match.group("title") or "").strip().strip(":—-–") or f"step {index + 1}"
|
|
36
|
+
steps.append({"id": len(steps) + 1, "title": title[:90],
|
|
37
|
+
"body": text[match.end():end].strip()[:6000]})
|
|
38
|
+
if len(steps) == MAX_STEPS:
|
|
39
|
+
break
|
|
40
|
+
return steps
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def key_for(root: str, plan_sha: str) -> str:
|
|
44
|
+
"""Readable, unique per resolved project root, and per plan content.
|
|
45
|
+
|
|
46
|
+
The folder's own name is not enough: two checkouts of one project share the name `demo`,
|
|
47
|
+
and a ledger they both read hands a verified step to the folder that never ran it. The
|
|
48
|
+
root half is exactly the key the project notes use, so the two stores cannot disagree.
|
|
49
|
+
"""
|
|
50
|
+
return memory_store.key_for(root) + "-" + plan_sha[:16]
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def relative_name(ws: Workspace, plan_file: str) -> str:
|
|
54
|
+
"""The plan's workspace-relative name, whether the caller has a path or a full path."""
|
|
55
|
+
candidate = Path(plan_file)
|
|
56
|
+
if candidate.is_absolute():
|
|
57
|
+
try:
|
|
58
|
+
candidate = candidate.relative_to(ws.root)
|
|
59
|
+
except ValueError:
|
|
60
|
+
raise PolicyError("The plan file must be inside the selected project folder.") from None
|
|
61
|
+
return candidate.as_posix()
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _same_root(stored: object, root: Path) -> bool:
|
|
65
|
+
"""A ledger that disagrees about which folder it belongs to is not that folder's ledger."""
|
|
66
|
+
try:
|
|
67
|
+
return (str(Path(str(stored)).expanduser().resolve()).casefold()
|
|
68
|
+
== str(Path(root).expanduser().resolve()).casefold())
|
|
69
|
+
except OSError:
|
|
70
|
+
return False
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def open_book(plans_dir: Path, ws: Workspace, plan_file: str) -> tuple[Path, dict]:
|
|
74
|
+
"""Load the ledger for this plan, or start one. A changed plan starts a new ledger."""
|
|
75
|
+
reference = ws.read(relative_name(ws, plan_file))
|
|
76
|
+
path = plans_dir / (key_for(str(ws.root), reference["sha256"]) + ".json")
|
|
77
|
+
if path.exists():
|
|
78
|
+
try:
|
|
79
|
+
book = json.loads(path.read_text(encoding="utf-8"))
|
|
80
|
+
if (book.get("schema") == 1 and book.get("plan_sha256") == reference["sha256"]
|
|
81
|
+
and _same_root(book.get("root", ""), ws.root)
|
|
82
|
+
and isinstance(book.get("steps"), list)):
|
|
83
|
+
return path, book
|
|
84
|
+
except (OSError, ValueError, TypeError):
|
|
85
|
+
pass
|
|
86
|
+
book = {"schema": 1, "root": str(ws.root), "plan_path": reference["path"],
|
|
87
|
+
"plan_sha256": reference["sha256"],
|
|
88
|
+
"steps": [dict(row, status="pending", session_id=None, verified_at=None)
|
|
89
|
+
for row in parse_steps(reference["content"])]}
|
|
90
|
+
return path, book
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def step(book: dict, step_id: int) -> dict | None:
|
|
94
|
+
return next((row for row in book["steps"] if row["id"] == step_id), None)
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def current(book: dict) -> dict | None:
|
|
98
|
+
return next((row for row in book["steps"] if row["status"] != "verified"), None)
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def done_titles(book: dict) -> list[str]:
|
|
102
|
+
return [row["title"] for row in book["steps"] if row["status"] == "verified"]
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def task_for(book: dict, row: dict, note: str = "") -> str:
|
|
106
|
+
"""The per-turn task the ledger sends instead of a hand-written phase request."""
|
|
107
|
+
lines = [f"Implement step {row['id']} of {len(book['steps'])} from the attached plan: "
|
|
108
|
+
+ row["title"] + "."]
|
|
109
|
+
if row.get("body"):
|
|
110
|
+
lines.append("This step asks for:\n" + row["body"][:3000])
|
|
111
|
+
finished = done_titles(book)
|
|
112
|
+
if finished:
|
|
113
|
+
lines.append("Already applied and verified by a command run — do not redo, rename or "
|
|
114
|
+
"rewrite them: " + ", ".join(finished) + ".")
|
|
115
|
+
lines.append("Touch only the files this step needs. Do not run builds or tests yourself; "
|
|
116
|
+
"the user runs the command and approves each proposal.")
|
|
117
|
+
if note:
|
|
118
|
+
lines.append("User note for this step: " + note[:800])
|
|
119
|
+
return "\n".join(lines)[:4000]
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def proof_reason(session: dict) -> str:
|
|
123
|
+
"""Why this session's last command run does, or does not, prove the step works."""
|
|
124
|
+
if session.get("state") != "CHECKS_PASSED":
|
|
125
|
+
return "the session is not in CHECKS_PASSED"
|
|
126
|
+
runs = session.get("runs") or []
|
|
127
|
+
if not runs:
|
|
128
|
+
return "no command has been recorded for this session"
|
|
129
|
+
run = runs[-1]
|
|
130
|
+
if run.get("status") != "passed":
|
|
131
|
+
return "the last command did not pass"
|
|
132
|
+
proof = run.get("proof")
|
|
133
|
+
if proof:
|
|
134
|
+
if proof.get("failures") or proof.get("errors"):
|
|
135
|
+
return "the test report shows failures or errors"
|
|
136
|
+
if proof.get("tests", 0) - proof.get("skipped", 0) <= 0:
|
|
137
|
+
return "the test report contains no executed test"
|
|
138
|
+
return ""
|
|
139
|
+
if run.get("tests_observed"):
|
|
140
|
+
return ""
|
|
141
|
+
# A compile-only recipe asks for no tests; the build passing is its proof.
|
|
142
|
+
if str(run.get("recipe", "")).endswith("compile"):
|
|
143
|
+
return ""
|
|
144
|
+
return "no test was observed running"
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def complete(path: Path, book: dict, step_id: int, session: dict) -> dict:
|
|
148
|
+
"""Mark a step verified only from a session that proved it, then persist the ledger."""
|
|
149
|
+
row = step(book, step_id)
|
|
150
|
+
if row is None:
|
|
151
|
+
raise PolicyError("Unknown plan step: " + str(step_id)[:20])
|
|
152
|
+
if row["status"] == "verified":
|
|
153
|
+
return book
|
|
154
|
+
if session.get("id") != row.get("session_id"):
|
|
155
|
+
raise PolicyError("That session is not the one recorded for this plan step.")
|
|
156
|
+
reason = proof_reason(session)
|
|
157
|
+
if reason:
|
|
158
|
+
raise PolicyError(f"Plan step {step_id} cannot be marked done: {reason}.")
|
|
159
|
+
row.update(status="verified", verified_at=session.get("created"))
|
|
160
|
+
atomic_json(path, book)
|
|
161
|
+
return book
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def record_session(path: Path, book: dict, step_id: int, session_id: str) -> dict:
|
|
165
|
+
"""Bind the session working on a step; a fix round rebinds it to the newer session."""
|
|
166
|
+
row = step(book, step_id)
|
|
167
|
+
if row is None:
|
|
168
|
+
raise PolicyError("Unknown plan step: " + str(step_id)[:20])
|
|
169
|
+
row["session_id"] = session_id
|
|
170
|
+
if row["status"] != "verified":
|
|
171
|
+
row["status"] = "in_progress"
|
|
172
|
+
atomic_json(path, book)
|
|
173
|
+
return book
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def reopen(path: Path, book: dict, step_id: int) -> dict:
|
|
177
|
+
"""Un-verify a step whose applied files were rolled back, so the next one cannot build on nothing."""
|
|
178
|
+
row = step(book, step_id)
|
|
179
|
+
if row is None or row["status"] != "verified":
|
|
180
|
+
return book
|
|
181
|
+
row.update(status="in_progress", verified_at=None)
|
|
182
|
+
atomic_json(path, book)
|
|
183
|
+
return book
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
def progress_line(book: dict) -> str:
|
|
187
|
+
verified = len(done_titles(book))
|
|
188
|
+
row = current(book)
|
|
189
|
+
if row is None:
|
|
190
|
+
return f"Plan complete — {verified}/{len(book['steps'])} steps verified"
|
|
191
|
+
return (f"Plan step {row['id']}/{len(book['steps'])}: {row['title']} "
|
|
192
|
+
f"({verified} verified)")
|
|
@@ -0,0 +1,404 @@
|
|
|
1
|
+
"""Provider boundaries: no redirects, no proxies, no automatic cloud fallback."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import json
|
|
5
|
+
import math
|
|
6
|
+
import os
|
|
7
|
+
import time
|
|
8
|
+
from typing import Protocol
|
|
9
|
+
from urllib.error import HTTPError, URLError
|
|
10
|
+
from urllib.request import HTTPRedirectHandler, ProxyHandler, Request, build_opener
|
|
11
|
+
|
|
12
|
+
from .config import (Kind, OLLAMA, OPENROUTER, Settings, check_endpoint, kind_for, needs_consent,
|
|
13
|
+
validate)
|
|
14
|
+
from .errors import PolicyError, ProviderError, ProviderUnavailable
|
|
15
|
+
from .redaction import redact
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class NoRedirect(HTTPRedirectHandler):
|
|
19
|
+
def redirect_request(self, req, fp, code, msg, headers, newurl):
|
|
20
|
+
return None
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _refuse(exc: HTTPError) -> ProviderError:
|
|
24
|
+
"""The one translation of a provider that answered with an error code.
|
|
25
|
+
|
|
26
|
+
Never the request, the bearer token or the URL. The response body is a different case: it is the
|
|
27
|
+
only place Ollama writes *why* it refused, and "model requires more system memory (9.2 GiB) than
|
|
28
|
+
is available (6.1 GiB)" is the difference between a dead task and a fixable one. So a short,
|
|
29
|
+
collapsed, redacted excerpt is allowed — capped, not trusted.
|
|
30
|
+
"""
|
|
31
|
+
detail = ""
|
|
32
|
+
try:
|
|
33
|
+
detail = redact(" ".join(exc.read(4096).decode("utf-8", "replace").split()))[:180]
|
|
34
|
+
except Exception: # noqa: BLE001 - a body that will not
|
|
35
|
+
detail = "" # read is not worth losing the code for
|
|
36
|
+
return ProviderError(f"Provider HTTP {exc.code}"
|
|
37
|
+
+ (f": {detail}" if detail else "")
|
|
38
|
+
+ "; no automatic retry or fallback.")
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _opener():
|
|
42
|
+
"""One opener for both reads, so a streaming body cannot inherit different rules than a buffered one."""
|
|
43
|
+
return build_opener(ProxyHandler({}), NoRedirect())
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def request_json(url: str, payload: dict | None = None, *, key: str | None = None,
|
|
47
|
+
timeout: int = 120, max_bytes: int = 2_000_000) -> dict:
|
|
48
|
+
headers = {"Content-Type": "application/json"}
|
|
49
|
+
if key:
|
|
50
|
+
headers["Authorization"] = "Bearer " + key
|
|
51
|
+
request = Request(url, data=json.dumps(payload).encode() if payload is not None else None,
|
|
52
|
+
headers=headers)
|
|
53
|
+
try:
|
|
54
|
+
with _opener().open(request, timeout=timeout) as response:
|
|
55
|
+
raw = response.read(max_bytes + 1)
|
|
56
|
+
if len(raw) > max_bytes:
|
|
57
|
+
raise ProviderError("Provider response exceeds size limit.")
|
|
58
|
+
result = json.loads(raw)
|
|
59
|
+
if not isinstance(result, dict):
|
|
60
|
+
raise ProviderError("Provider returned an invalid response.")
|
|
61
|
+
return result
|
|
62
|
+
except HTTPError as exc:
|
|
63
|
+
raise _refuse(exc) from None
|
|
64
|
+
except (URLError, TimeoutError, OSError):
|
|
65
|
+
raise ProviderUnavailable("Provider connection failed or timed out. A local model needs a "
|
|
66
|
+
"longer request timeout for a reply this large.") from None
|
|
67
|
+
except (ValueError, UnicodeError):
|
|
68
|
+
raise ProviderError("Provider returned invalid JSON.") from None
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
# A cold Ollama daemon answers nothing for a few seconds while it loads, which is the one failure a
|
|
72
|
+
# second ask genuinely fixes. Two attempts and one second is the whole policy; more would only be a
|
|
73
|
+
# slower way of saying the service is not running.
|
|
74
|
+
PROBE_ATTEMPTS = 2
|
|
75
|
+
PROBE_PAUSE = 1.0
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def request_probe_with_retry(url: str, payload: dict | None = None, *, key: str | None = None,
|
|
79
|
+
timeout: int = 10, attempts: int = PROBE_ATTEMPTS,
|
|
80
|
+
pause: float = PROBE_PAUSE) -> dict:
|
|
81
|
+
"""One metadata read, asked twice if the first attempt never reached anything.
|
|
82
|
+
|
|
83
|
+
Only ever for metadata: `/api/tags` and `/api/show` are pure reads, so a repeat costs a second and
|
|
84
|
+
changes nothing. A generation call is deliberately not in here. Re-running `/api/chat` or a stream
|
|
85
|
+
starts another model load — minutes of CPU on a machine that may have just said it has no room —
|
|
86
|
+
and restarting a stream mid-answer breaks the row the user is already watching, which is what
|
|
87
|
+
`tests/test_transport.py` holds the refusal sentence for. So those keep one attempt, and the
|
|
88
|
+
distinction is typed (`ProviderUnavailable`) rather than read back out of an English message.
|
|
89
|
+
"""
|
|
90
|
+
for attempt in range(max(1, attempts)):
|
|
91
|
+
try:
|
|
92
|
+
return request_json(url, payload, key=key, timeout=timeout)
|
|
93
|
+
except ProviderUnavailable:
|
|
94
|
+
if attempt + 1 >= attempts:
|
|
95
|
+
raise
|
|
96
|
+
time.sleep(pause)
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
# A streaming body announces no length to trust, so the cap moves to what has been read. The same
|
|
100
|
+
# figure as the buffered read, because a reply that grows past it is not an answer that arrived late.
|
|
101
|
+
MAX_STREAM_BYTES = 2_000_000
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def ollama_chunk(line: bytes) -> dict | None:
|
|
105
|
+
"""One NDJSON line from `/api/chat`: what was added, to which field, and how it ended."""
|
|
106
|
+
try:
|
|
107
|
+
chunk = json.loads(line)
|
|
108
|
+
except ValueError:
|
|
109
|
+
return None
|
|
110
|
+
if not isinstance(chunk, dict):
|
|
111
|
+
return None
|
|
112
|
+
message = chunk.get("message") if isinstance(chunk.get("message"), dict) else {}
|
|
113
|
+
return {"content": str(message.get("content") or ""),
|
|
114
|
+
"thinking": str(message.get("thinking") or message.get("reasoning") or ""),
|
|
115
|
+
"finish": str(chunk.get("done_reason") or "") or None,
|
|
116
|
+
"model": "", "done": bool(chunk.get("done"))}
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def openai_chunk(line: bytes) -> dict | None:
|
|
120
|
+
"""One SSE frame from `/chat/completions`. Blank frames and comment lines carry no text.
|
|
121
|
+
|
|
122
|
+
DeepSeek's R1 line and OpenRouter's reasoning models put the deliberation in the delta beside the
|
|
123
|
+
content, under the same two names they use in the buffered answer, so the same fields are read
|
|
124
|
+
here — a streaming reply is not a licence to drop half of it.
|
|
125
|
+
"""
|
|
126
|
+
text = line.decode("utf-8", "replace").strip()
|
|
127
|
+
if not text.startswith("data:"):
|
|
128
|
+
return None
|
|
129
|
+
body = text[5:].strip()
|
|
130
|
+
if body == "[DONE]":
|
|
131
|
+
return {"content": "", "thinking": "", "finish": None, "model": "", "done": True}
|
|
132
|
+
try:
|
|
133
|
+
chunk = json.loads(body)
|
|
134
|
+
except ValueError:
|
|
135
|
+
return None
|
|
136
|
+
choices = chunk.get("choices") or []
|
|
137
|
+
first = choices[0] if choices and isinstance(choices[0], dict) else {}
|
|
138
|
+
delta = first.get("delta") if isinstance(first.get("delta"), dict) else {}
|
|
139
|
+
return {"content": str(delta.get("content") or ""),
|
|
140
|
+
"thinking": str(delta.get("reasoning_content") or delta.get("reasoning") or ""),
|
|
141
|
+
"finish": first.get("finish_reason") or None,
|
|
142
|
+
"model": str(chunk.get("model") or ""), "done": False}
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def read_stream(url: str, payload: dict, *, key: str | None = None, timeout: int = 120,
|
|
146
|
+
on_token=None, chunk=ollama_chunk, max_bytes: int = MAX_STREAM_BYTES) -> dict:
|
|
147
|
+
"""Read a streamed reply in pieces, saying each one as it lands, and return the whole of it.
|
|
148
|
+
|
|
149
|
+
The text is assembled here rather than trusted from the client's copy: the caller has to parse an
|
|
150
|
+
envelope, store a record and hash a proposal out of what arrives, and a browser that lost a frame
|
|
151
|
+
would otherwise change what the tool decided. `on_token` is a *display* consumer.
|
|
152
|
+
"""
|
|
153
|
+
headers = {"Content-Type": "application/json", "Accept": "text/event-stream"}
|
|
154
|
+
if key:
|
|
155
|
+
headers["Authorization"] = "Bearer " + key
|
|
156
|
+
request = Request(url, data=json.dumps(payload).encode(), headers=headers)
|
|
157
|
+
content, thought, size, model, finish = [], [], 0, "", None
|
|
158
|
+
try:
|
|
159
|
+
with _opener().open(request, timeout=timeout) as response:
|
|
160
|
+
for line in response:
|
|
161
|
+
size += len(line)
|
|
162
|
+
if size > max_bytes:
|
|
163
|
+
raise ProviderError("Provider response exceeds size limit.")
|
|
164
|
+
part = chunk(line)
|
|
165
|
+
if not part:
|
|
166
|
+
continue
|
|
167
|
+
if part.get("model"):
|
|
168
|
+
model = part["model"]
|
|
169
|
+
if part.get("finish"):
|
|
170
|
+
finish = part["finish"]
|
|
171
|
+
if part["content"]:
|
|
172
|
+
content.append(part["content"])
|
|
173
|
+
if on_token is not None:
|
|
174
|
+
on_token(part["content"])
|
|
175
|
+
thought.append(part["thinking"])
|
|
176
|
+
if part.get("done"):
|
|
177
|
+
break
|
|
178
|
+
except HTTPError as exc:
|
|
179
|
+
raise _refuse(exc) from None
|
|
180
|
+
except (URLError, TimeoutError, OSError):
|
|
181
|
+
raise ProviderUnavailable("Provider connection failed or timed out. A local model needs a "
|
|
182
|
+
"longer request timeout for a reply this large.") from None
|
|
183
|
+
except ValueError:
|
|
184
|
+
raise ProviderError("Provider returned invalid JSON.") from None
|
|
185
|
+
return {"content": "".join(content), "thinking": "".join(thought),
|
|
186
|
+
"finish": finish, "model": model}
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
# Ollama's default context is small enough to cut a real coding prompt in half without saying so.
|
|
190
|
+
# Measured on this machine with `qwen2.5-coder:3b`: a 20 000-character prompt and a 60 000-character
|
|
191
|
+
# prompt both answered with `prompt_eval_count=2050`. The model replied from roughly the first eight
|
|
192
|
+
# thousand characters and nothing reported the loss — and `num_predict` above the window was capped the
|
|
193
|
+
# same quiet way. So the window is asked for explicitly, sized to the request that is being sent.
|
|
194
|
+
MIN_CTX = 2048
|
|
195
|
+
MAX_CTX = 16384
|
|
196
|
+
TEMPLATE_MARKERS = 64
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def estimate_tokens(chars: int) -> int:
|
|
200
|
+
"""The same ÷4 the interface labels as an estimate. It is an estimate: it fits English prose and
|
|
201
|
+
not code or Arabic, which is why it sizes a request rather than reporting one."""
|
|
202
|
+
return max(1, math.ceil(int(chars or 0) / 4))
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def context_window(prompt_chars: int, output_tokens: int) -> int:
|
|
206
|
+
"""A power-of-two `num_ctx` with room for this prompt *and* its reply, clamped to what a CPU
|
|
207
|
+
box can hold — prompt processing, not generation, is what costs minutes here."""
|
|
208
|
+
need = estimate_tokens(prompt_chars) + int(output_tokens or 0) + TEMPLATE_MARKERS
|
|
209
|
+
window = MIN_CTX
|
|
210
|
+
while window < need and window < MAX_CTX:
|
|
211
|
+
window *= 2
|
|
212
|
+
return min(window, MAX_CTX)
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
class ModelProvider(Protocol):
|
|
216
|
+
"""The seam between the loop and a model.
|
|
217
|
+
|
|
218
|
+
`supports_stream` is declared rather than defaulted: a caller asks it before it passes
|
|
219
|
+
`on_token`, which is what lets the scripted providers in the tests keep the two-argument
|
|
220
|
+
signature they have always had.
|
|
221
|
+
"""
|
|
222
|
+
|
|
223
|
+
model: str
|
|
224
|
+
supports_stream: bool
|
|
225
|
+
|
|
226
|
+
def generate(self, messages: list[dict], json_mode: bool = True,
|
|
227
|
+
on_token=None) -> str: ...
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
# A reasoning model answers twice: once in a field nobody asked for and once in `content`. The names
|
|
231
|
+
# differ per provider — `reasoning` on Ollama and OpenRouter, `reasoning_content` on DeepSeek's R1 line,
|
|
232
|
+
# `thinking` on Qwen — and all three have been seen carrying the words while `content` carried the JSON.
|
|
233
|
+
REASONING_KEYS = ("reasoning", "reasoning_content", "thinking")
|
|
234
|
+
REASONING_CHARS = 1200
|
|
235
|
+
|
|
236
|
+
|
|
237
|
+
def read_reasoning(message) -> str:
|
|
238
|
+
"""What the model thought out loud, capped and redacted — or "" when it kept that to itself.
|
|
239
|
+
|
|
240
|
+
This never enters the envelope path: `parse_action` must only ever see what the model was asked to
|
|
241
|
+
return, and reasoning is precisely the field that contains prose, braces and half-finished thoughts.
|
|
242
|
+
It is redacted here rather than at the surface because the raw words are also what a report exports,
|
|
243
|
+
and a chain of thought quoting a connection string is still quoting a connection string.
|
|
244
|
+
"""
|
|
245
|
+
if not isinstance(message, dict):
|
|
246
|
+
return ""
|
|
247
|
+
for key in REASONING_KEYS:
|
|
248
|
+
value = message.get(key)
|
|
249
|
+
if isinstance(value, str) and value.strip():
|
|
250
|
+
return redact(value.strip())[:REASONING_CHARS]
|
|
251
|
+
return ""
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
class OllamaProvider:
|
|
255
|
+
supports_stream = True
|
|
256
|
+
|
|
257
|
+
def __init__(self, settings: Settings, *, allow_cloud: bool = False):
|
|
258
|
+
self.settings = settings
|
|
259
|
+
self.model = settings.model
|
|
260
|
+
self.supports_thinking = False
|
|
261
|
+
self.reasoning = ""
|
|
262
|
+
self.allow_cloud = allow_cloud
|
|
263
|
+
# One rule for every provider, in ``config``: this device only, no credentials in the URL.
|
|
264
|
+
# It used to refuse a URL *path* as well, which is how a relocated Ollama became unusable.
|
|
265
|
+
self.endpoint = check_endpoint(OLLAMA, settings.endpoint)
|
|
266
|
+
|
|
267
|
+
def preflight(self) -> None:
|
|
268
|
+
info = request_json(self.endpoint + "/api/show", {"model": self.model}, timeout=10)
|
|
269
|
+
cloud_backed = bool("cloud" in self.model.casefold() or info.get("remote_host") or info.get("remote_model"))
|
|
270
|
+
if cloud_backed and not self.allow_cloud:
|
|
271
|
+
raise PolicyError("This Ollama model runs in the cloud. Approve cloud processing for public/synthetic code or choose a local model.")
|
|
272
|
+
if not cloud_backed and not info.get("model_info"):
|
|
273
|
+
raise PolicyError("Cannot establish that this is an installed local model.")
|
|
274
|
+
self.supports_thinking = "thinking" in (info.get("capabilities") or [])
|
|
275
|
+
|
|
276
|
+
def generate(self, messages: list[dict], json_mode: bool = True, on_token=None) -> str:
|
|
277
|
+
self.reasoning = ""
|
|
278
|
+
prompt_chars = sum(len(str(message.get("content", ""))) for message in messages or [])
|
|
279
|
+
payload = {
|
|
280
|
+
"model": self.model, "messages": messages,
|
|
281
|
+
# Asked for piece by piece only when somebody is listening. A turn that parses an envelope
|
|
282
|
+
# gains nothing from a stream, and an SSE reader is one more way for a reply to go wrong.
|
|
283
|
+
"stream": on_token is not None,
|
|
284
|
+
# temperature 0 is greedy decoding, and a weak model that is poor at Arabic
|
|
285
|
+
# tokenisation can sit on one token forever; 1.1 is the smallest penalty that
|
|
286
|
+
# breaks the loop without bending the distribution on English or code.
|
|
287
|
+
"options": {"temperature": 0, "num_predict": self.settings.output_tokens,
|
|
288
|
+
"repeat_penalty": 1.1,
|
|
289
|
+
"num_ctx": context_window(prompt_chars, self.settings.output_tokens)},
|
|
290
|
+
}
|
|
291
|
+
if json_mode:
|
|
292
|
+
payload["format"] = "json"
|
|
293
|
+
if self.supports_thinking:
|
|
294
|
+
# Measured on the local endpoint (2026-09-29, qwen3:4b): a prose turn with the switch off
|
|
295
|
+
# puts the deliberation at the *front of `content`* — its closing marker sits inside the
|
|
296
|
+
# text the reader is meant to call the answer — while the same 136 tokens with it on split
|
|
297
|
+
# into a 180-character reply and a 522-character field. An envelope keeps it off: under
|
|
298
|
+
# `format: json` the same request spent its whole budget thinking and answered nothing.
|
|
299
|
+
payload["think"] = not json_mode
|
|
300
|
+
if on_token is None:
|
|
301
|
+
result = request_json(self.endpoint + "/api/chat", payload,
|
|
302
|
+
timeout=self.settings.timeout_seconds)
|
|
303
|
+
message = result.get("message") if isinstance(result.get("message"), dict) else {}
|
|
304
|
+
value, thought = message.get("content"), message
|
|
305
|
+
truncated = result.get("done_reason") == "length"
|
|
306
|
+
else:
|
|
307
|
+
streamed = read_stream(self.endpoint + "/api/chat", payload,
|
|
308
|
+
timeout=self.settings.timeout_seconds, on_token=on_token,
|
|
309
|
+
chunk=ollama_chunk)
|
|
310
|
+
value, thought = streamed["content"], {"thinking": streamed["thinking"]}
|
|
311
|
+
truncated = streamed["finish"] == "length"
|
|
312
|
+
self.reasoning = read_reasoning(thought)
|
|
313
|
+
if truncated:
|
|
314
|
+
raise ProviderError("Model output truncated; reduce the change size.")
|
|
315
|
+
if not isinstance(value, str) or not value:
|
|
316
|
+
raise ProviderError("No model content returned.")
|
|
317
|
+
return value
|
|
318
|
+
|
|
319
|
+
|
|
320
|
+
class OpenAICompatibleProvider:
|
|
321
|
+
"""Every provider that answers ``POST {base}/chat/completions`` with the OpenAI shape.
|
|
322
|
+
|
|
323
|
+
That is OpenAI, Groq, DeepSeek, OpenRouter, LM Studio, vLLM and any custom base a user types:
|
|
324
|
+
one body, one response, a different URL and a different key. Anthropic and Gemini are *not* in
|
|
325
|
+
this class — different auth header, different body, different response — and are not offered.
|
|
326
|
+
The three things that really are OpenRouter-only (the upstream routing block, the echoed
|
|
327
|
+
upstream model, the free/paid rule) hang off ``Kind`` flags rather than a subclass.
|
|
328
|
+
"""
|
|
329
|
+
|
|
330
|
+
supports_stream = True
|
|
331
|
+
|
|
332
|
+
def __init__(self, settings: Settings, api_key: str | None = None, *,
|
|
333
|
+
allow_paid: bool = False, kind: Kind | None = None):
|
|
334
|
+
self.kind = kind or kind_for(settings.provider) or OPENROUTER
|
|
335
|
+
self.reasoning = ""
|
|
336
|
+
self.settings = settings
|
|
337
|
+
self.model = settings.model
|
|
338
|
+
if self.kind.free_only and not allow_paid and settings.model != "openrouter/free" \
|
|
339
|
+
and not str(settings.model).endswith(":free"):
|
|
340
|
+
raise PolicyError("Select the Paid cloud option explicitly to use a paid OpenRouter model.")
|
|
341
|
+
self.endpoint = check_endpoint(self.kind, settings.endpoint)
|
|
342
|
+
env_name = settings.api_key_env or self.kind.key_env
|
|
343
|
+
self.key = api_key or (os.environ.get(env_name) if env_name else "")
|
|
344
|
+
# A custom endpoint may or may not want a key — that is the user's server to decide — so only
|
|
345
|
+
# the rows that are known to require one refuse without it.
|
|
346
|
+
if self.kind.needs_key and not self.key:
|
|
347
|
+
raise ProviderError(f"Set {env_name or 'an API key'} in your environment (never in a file).")
|
|
348
|
+
|
|
349
|
+
def generate(self, messages: list[dict], json_mode: bool = True, on_token=None) -> str:
|
|
350
|
+
self.reasoning = ""
|
|
351
|
+
body = {
|
|
352
|
+
"model": self.settings.model, "messages": messages, "stream": on_token is not None,
|
|
353
|
+
"temperature": 0, "max_tokens": self.settings.output_tokens,
|
|
354
|
+
}
|
|
355
|
+
if json_mode:
|
|
356
|
+
body["response_format"] = {"type": "json_object"}
|
|
357
|
+
if self.kind.routing:
|
|
358
|
+
# Refuse OpenRouter's own failovers: a silent hop to another upstream means the model
|
|
359
|
+
# that answered is not the one that was reviewed.
|
|
360
|
+
body["provider"] = {"allow_fallbacks": False, "require_parameters": True}
|
|
361
|
+
if on_token is None:
|
|
362
|
+
try:
|
|
363
|
+
result = request_json(self.endpoint + "/chat/completions", body,
|
|
364
|
+
key=self.key or None, timeout=self.settings.timeout_seconds)
|
|
365
|
+
choice = result["choices"][0]
|
|
366
|
+
message = choice["message"] if isinstance(choice.get("message"), dict) else {}
|
|
367
|
+
value, thought = message.get("content"), message
|
|
368
|
+
finish, echoed = choice.get("finish_reason"), str(result.get("model") or "")
|
|
369
|
+
except (KeyError, IndexError, TypeError, AttributeError):
|
|
370
|
+
raise ProviderError(f"Invalid {self.kind.label} response.") from None
|
|
371
|
+
else:
|
|
372
|
+
streamed = read_stream(self.endpoint + "/chat/completions", body, key=self.key or None,
|
|
373
|
+
timeout=self.settings.timeout_seconds, on_token=on_token,
|
|
374
|
+
chunk=openai_chunk)
|
|
375
|
+
value, thought = streamed["content"], {"thinking": streamed["thinking"]}
|
|
376
|
+
finish, echoed = streamed["finish"], streamed["model"]
|
|
377
|
+
self.reasoning = read_reasoning(thought)
|
|
378
|
+
# Proposals must be complete; chat tolerates a missing finish_reason.
|
|
379
|
+
if not (finish == "stop" or (not json_mode and finish is None)):
|
|
380
|
+
raise ProviderError("Model did not finish normally; output discarded.")
|
|
381
|
+
if self.kind.routing and echoed:
|
|
382
|
+
self.model = echoed
|
|
383
|
+
if not isinstance(value, str) or not value:
|
|
384
|
+
raise ProviderError(f"Invalid {self.kind.label} response.")
|
|
385
|
+
return value
|
|
386
|
+
|
|
387
|
+
|
|
388
|
+
def make_provider(settings: Settings, *, allow_cloud: bool, data_class: str,
|
|
389
|
+
api_key: str | None = None, allow_paid: bool = False) -> ModelProvider:
|
|
390
|
+
validate(settings)
|
|
391
|
+
kind = kind_for(settings.provider) or OLLAMA
|
|
392
|
+
# Judged on the address that will actually be used, not on the string the settings happened to
|
|
393
|
+
# carry: an empty endpoint resolves through the profile, the environment and the table, and a
|
|
394
|
+
# consent decision made before that question is answered is made about a host nobody chose.
|
|
395
|
+
endpoint = check_endpoint(kind, settings.endpoint)
|
|
396
|
+
if kind.shape == "ollama":
|
|
397
|
+
# A cloud-backed Ollama model is discovered in preflight, not assumed from the endpoint.
|
|
398
|
+
provider = OllamaProvider(settings, allow_cloud=allow_cloud and data_class in {"public", "synthetic"})
|
|
399
|
+
provider.preflight()
|
|
400
|
+
return provider
|
|
401
|
+
if needs_consent(kind, endpoint) and (
|
|
402
|
+
not allow_cloud or data_class not in {"public", "synthetic"}):
|
|
403
|
+
raise PolicyError("Cloud requires --allow-cloud and --data-class public or synthetic.")
|
|
404
|
+
return OpenAICompatibleProvider(settings, api_key=api_key, allow_paid=allow_paid, kind=kind)
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
"""Keep credentials out of the text a build run leaves behind.
|
|
2
|
+
|
|
3
|
+
Command output is stored on the session and fed back to the model on the next repair
|
|
4
|
+
turn, so anything the project printed lands in a log file and possibly in a cloud
|
|
5
|
+
request. Projects print their own connection strings often enough that the guarantee
|
|
6
|
+
cannot be "nobody printed a secret"; the value is removed where it is captured.
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import re
|
|
11
|
+
|
|
12
|
+
# Placeholder text is not a secret and redacting it would only confuse the model.
|
|
13
|
+
SKIP_VALUES = {"null", "none", "true", "false", "undefined", "***", "[redacted]", "string",
|
|
14
|
+
"change-me", "changeme", "secret", "password", "your-key-here", "<token>"}
|
|
15
|
+
|
|
16
|
+
PATTERNS = (
|
|
17
|
+
# Whole PEM private keys, across lines.
|
|
18
|
+
re.compile(r"(?s)-----BEGIN[A-Z ]*PRIVATE KEY-----.*?-----END[A-Z ]*PRIVATE KEY-----"),
|
|
19
|
+
re.compile(r"\b(?:AKIA|ASIA)[0-9A-Z]{16}\b"), # AWS access key id
|
|
20
|
+
re.compile(r"\bgh[pousr]_[A-Za-z0-9]{20,}\b"), # GitHub token
|
|
21
|
+
re.compile(r"\bxox[baprs]-[A-Za-z0-9-]{10,}\b"), # Slack token
|
|
22
|
+
re.compile(r"\bAIza[0-9A-Za-z_\-]{35}\b"), # Google API key
|
|
23
|
+
re.compile(r"\bsk-(?:ant-|proj-)?[A-Za-z0-9_\-]{20,}\b"), # OpenAI/Anthropic key
|
|
24
|
+
re.compile(r"\beyJ[A-Za-z0-9_\-]{8,}\.[A-Za-z0-9_\-]{4,}\.[A-Za-z0-9_\-]{4,}\b"), # JWT
|
|
25
|
+
re.compile(r"(?i)\bglpat-[A-Za-z0-9_\-]{20,}\b"), # GitLab token
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
# user:password inside any URL or JDBC-style connection string.
|
|
29
|
+
URL_CREDENTIALS = re.compile(r"(?i)\b([a-z][a-z0-9+.\-]*://)[^\s/@:]+:[^\s/@]+@")
|
|
30
|
+
# "password = hunter2", "API_KEY: 'abcd'", -Dtoken=abcd, and friends. The optional qualifier in
|
|
31
|
+
# front exists because `\b` does not fire after an underscore: without it the two shapes that build
|
|
32
|
+
# logs really print — SPRING_DATASOURCE_PASSWORD=… and AWS_SECRET_ACCESS_KEY=… — passed straight
|
|
33
|
+
# through, into the session file and on to the next model request.
|
|
34
|
+
ASSIGNMENT = re.compile(
|
|
35
|
+
r"(?i)\b([\w.]*_)?(password|passwd|pwd|secret|api[_-]?key|access[_-]?key|secret[_-]?key|"
|
|
36
|
+
r"client[_-]?secret|auth[_-]?token|token)(\s*[=:]\s*)([\"']?)([^\s,;\"']{6,})(\4)")
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _assignment(match: re.Match) -> str:
|
|
40
|
+
value = match.group(5)
|
|
41
|
+
if value.casefold().strip("'\"") in SKIP_VALUES or value.startswith(("${", "$(", "%{")):
|
|
42
|
+
return match.group(0)
|
|
43
|
+
return ((match.group(1) or "") + match.group(2) + match.group(3) + match.group(4)
|
|
44
|
+
+ "[redacted]" + match.group(6))
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def redact(text: str) -> str:
|
|
48
|
+
"""Replace credential-shaped runs of characters; leave the surrounding log intact."""
|
|
49
|
+
if not isinstance(text, str) or not text:
|
|
50
|
+
return text
|
|
51
|
+
for pattern in PATTERNS:
|
|
52
|
+
text = pattern.sub("[redacted]", text)
|
|
53
|
+
text = URL_CREDENTIALS.sub(lambda m: m.group(1) + "[redacted]@", text)
|
|
54
|
+
return ASSIGNMENT.sub(_assignment, text)
|