graphban-fleet 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- gbagent/__init__.py +6 -0
- gbagent/cli.py +463 -0
- gbagent/compact.py +284 -0
- gbagent/config.py +222 -0
- gbagent/coord.py +264 -0
- gbagent/heartbeat.py +124 -0
- gbagent/llm.py +250 -0
- gbagent/loop.py +504 -0
- gbagent/orient.py +237 -0
- gbagent/tools.py +359 -0
- gbagent/toolset.py +356 -0
- gbagent/verify.py +162 -0
- gbagent/workspace.py +94 -0
- gbfleet/__init__.py +25 -0
- gbfleet/adapters/__init__.py +418 -0
- gbfleet/adapters/claude.py +111 -0
- gbfleet/adapters/codex.py +27 -0
- gbfleet/adapters/cursor.py +108 -0
- gbfleet/adapters/cursor_stream.py +86 -0
- gbfleet/adapters/gbagent.py +192 -0
- gbfleet/adapters/grok.py +148 -0
- gbfleet/adapters/qwen_code.py +120 -0
- gbfleet/adopt.py +310 -0
- gbfleet/cli.py +603 -0
- gbfleet/client.py +275 -0
- gbfleet/doctor.py +445 -0
- gbfleet/hostos.py +551 -0
- gbfleet/lock.py +222 -0
- gbfleet/matrix.py +621 -0
- gbfleet/matrix.toml +146 -0
- gbfleet/mcp.py +645 -0
- gbfleet/observe.py +147 -0
- gbfleet/progress.py +141 -0
- gbfleet/record.py +50 -0
- gbfleet/seat.py +364 -0
- gbfleet/spawn.py +450 -0
- gbfleet/state.py +120 -0
- gbfleet/supervisor.py +1255 -0
- gbfleet/tiers.py +60 -0
- gbfleet/touchpoints.py +132 -0
- gbfleet/until.py +716 -0
- gbfleet/waits.py +35 -0
- gbfleet/worktree.py +694 -0
- graphban_fleet-0.1.0.dist-info/METADATA +205 -0
- graphban_fleet-0.1.0.dist-info/RECORD +48 -0
- graphban_fleet-0.1.0.dist-info/WHEEL +4 -0
- graphban_fleet-0.1.0.dist-info/entry_points.txt +3 -0
- graphban_fleet-0.1.0.dist-info/licenses/LICENSE +201 -0
gbagent/__init__.py
ADDED
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
"""gbagent — a first-party coding agent for local models (PRD-24).
|
|
2
|
+
|
|
3
|
+
Ships inside `graphban-fleet` and is versioned with it (D8): the version range machinery in
|
|
4
|
+
`gbfleet.adapters` exists because three vendors ship on their own schedules, and the
|
|
5
|
+
first-party adapter is the one case where that problem does not arise.
|
|
6
|
+
"""
|
gbagent/cli.py
ADDED
|
@@ -0,0 +1,463 @@
|
|
|
1
|
+
"""`gbagent` — the command the supervisor launches (PRD-24 D8, S5).
|
|
2
|
+
|
|
3
|
+
Three commands, and the small one matters most: `--version` is what
|
|
4
|
+
`adapters/gbagent.py` resolves against, and because the pin is exact it is also how a
|
|
5
|
+
`gbagent` from a different install gets caught before a process does any work.
|
|
6
|
+
|
|
7
|
+
**`models` exists so a typo is refused at spawn.** GRPH-485 was this failure the long way
|
|
8
|
+
round — a model name that did not exist, found days later as a grill that would not converge.
|
|
9
|
+
The adapter cannot ask an endpoint itself (only two modules in this package may open a socket),
|
|
10
|
+
so it shells out to here.
|
|
11
|
+
|
|
12
|
+
**`run` can pick up its own work.** `--item` is optional from S7 on: without one the model
|
|
13
|
+
calls `claim_cluster` itself, which is in `coord.WORKER_TOOLS` along with the rest of
|
|
14
|
+
`COORDINATION_TOOLS` (P30 D3). `claim_next` is not advertised: it reserves no files.
|
|
15
|
+
|
|
16
|
+
This paragraph used to say the opposite, and was true when written — a later slice wired the
|
|
17
|
+
thing it described as unwired, and the prose did not follow (GRPH-562). Corrected rather than
|
|
18
|
+
deleted, because the mistake is worth not repeating: **a tool set is a declaration of intent,
|
|
19
|
+
not an enforcement boundary**, so a docstring reasoning about authority from set membership is
|
|
20
|
+
describing the wrong object. The file even disagreed with itself — `assignment_for` below has
|
|
21
|
+
said all along that the model claims for itself.
|
|
22
|
+
|
|
23
|
+
What actually stops a worker overreaching is on the server: `TOOL_ROLES` gates what a
|
|
24
|
+
credential may call, `independent()` refuses a sign-off from the author whatever tools it
|
|
25
|
+
holds, and D5 clamps a worker at `review` — done is not the agent's word. `assert not
|
|
26
|
+
WORKER_TOOLS & ALLOWED_TOOLS` pins that a worker is not a supervisor, which is the one thing
|
|
27
|
+
the set itself is good for.
|
|
28
|
+
"""
|
|
29
|
+
from __future__ import annotations
|
|
30
|
+
|
|
31
|
+
import argparse
|
|
32
|
+
import json
|
|
33
|
+
import os
|
|
34
|
+
import re
|
|
35
|
+
import sys
|
|
36
|
+
from pathlib import Path
|
|
37
|
+
|
|
38
|
+
import gbfleet
|
|
39
|
+
|
|
40
|
+
from . import loop
|
|
41
|
+
from .config import ConfigRefused, load, prepare
|
|
42
|
+
from .coord import (
|
|
43
|
+
MERGED_COORDINATION, MERGED_TOOLS, REVIEWER_COORDINATION, REVIEWER_TOOLS,
|
|
44
|
+
WORKER_TOOLS, Coordinator,
|
|
45
|
+
)
|
|
46
|
+
from .heartbeat import Heartbeat
|
|
47
|
+
from .llm import ModelUnreachable, OllamaSession
|
|
48
|
+
from .orient import (
|
|
49
|
+
COORDINATION_TOOLS,
|
|
50
|
+
INSTRUCTION as ORIENT_INSTRUCTION,
|
|
51
|
+
OrientationUnavailable,
|
|
52
|
+
build as build_orientation,
|
|
53
|
+
)
|
|
54
|
+
from .toolset import Toolset
|
|
55
|
+
|
|
56
|
+
#: Where the model endpoint lives. Named, never discovered — the same argument D3 makes
|
|
57
|
+
#: about the test command.
|
|
58
|
+
BASE_URL_ENV = "GBAGENT_BASE_URL"
|
|
59
|
+
#: A bearer for the model endpoint, when it wants one. Environment only, never argv: the
|
|
60
|
+
#: fleet's rule is that nothing carrying a credential goes on a command line (`ps` shows it),
|
|
61
|
+
#: and the supervisor's child inherits the operator's environment (spawn.py). Unset means an
|
|
62
|
+
#: unauthenticated endpoint — a local Ollama — which is what every walk so far has used.
|
|
63
|
+
API_KEY_ENV = "GBAGENT_API_KEY"
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def endpoint_key() -> str:
|
|
67
|
+
"""What `GBAGENT_API_KEY` holds, or "" — the only way a model credential reaches gbagent."""
|
|
68
|
+
return os.environ.get(API_KEY_ENV, "")
|
|
69
|
+
|
|
70
|
+
#: What the model is told before anything else. Two jobs: say what it cannot do, so it does
|
|
71
|
+
#: not spend 30-second turns finding out, and say what to reach for FIRST (S6).
|
|
72
|
+
SYSTEM = (
|
|
73
|
+
"You are gbagent, an unattended coding agent working inside one git worktree.\n"
|
|
74
|
+
"Use the tools. Do not narrate what you are about to do — do it.\n"
|
|
75
|
+
"Paths are relative to the worktree root. You cannot write outside it and you have no "
|
|
76
|
+
"shell; run_tests runs the command this repository declares.\n"
|
|
77
|
+
"\n" + ORIENT_INSTRUCTION + "\n"
|
|
78
|
+
"\nWhen the tests pass, say DONE and stop calling tools."
|
|
79
|
+
)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def assignment_for(item: str, role: str = "worker") -> str:
|
|
83
|
+
"""What the model is told to work on.
|
|
84
|
+
|
|
85
|
+
S6 (PRD-39 D-h): one loop — try `claim_review`, fall through to `claim_cluster`,
|
|
86
|
+
exit when both are empty. The role parameter is kept for backward compatibility
|
|
87
|
+
but no longer changes the assignment: a worker now claims, builds, AND reviews.
|
|
88
|
+
|
|
89
|
+
`--item` is optional from S7 on. Without one the model calls `claim_review` then
|
|
90
|
+
`claim_cluster` itself. With one it works the item it was handed.
|
|
91
|
+
"""
|
|
92
|
+
if item:
|
|
93
|
+
return (
|
|
94
|
+
f"You are working on {item}. Do not claim other build work. When {item} is in "
|
|
95
|
+
"review, call claim_review with wait_seconds=0 and review what you did not build "
|
|
96
|
+
"— sign_off, or bounce with a reason — until it answers nothing; then say DONE "
|
|
97
|
+
"and stop."
|
|
98
|
+
)
|
|
99
|
+
return (
|
|
100
|
+
"Call claim_review with wait_seconds=0. If there is nothing to review, "
|
|
101
|
+
"call claim_cluster with wait_seconds=0 to take the next ready non-colliding "
|
|
102
|
+
"cluster. If both are empty, say DONE and stop — exiting on an empty queue "
|
|
103
|
+
"is the normal end of your run, not a failure. You may sign_off work you "
|
|
104
|
+
"did not build. If the work is not ready — tests fail, the change is wrong — "
|
|
105
|
+
"bounce it with a reason naming what is wrong, then say DONE."
|
|
106
|
+
)
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
#: How the enrolment code is read back out of the instruction file.
|
|
110
|
+
#:
|
|
111
|
+
#: `spawn` writes the code there and deliberately never into the MCP config — the code is an
|
|
112
|
+
#: argument to `register_agent`, not a config value (`seat.mcp_config`). The format is OURS
|
|
113
|
+
#: (`seat.INSTRUCTION`), and a test pins this pattern against that constant, so a reworded
|
|
114
|
+
#: instruction fails a test rather than producing a child that silently never registers.
|
|
115
|
+
ENROLMENT = re.compile(r"enrolment_code=['\"]([^'\"]+)['\"]")
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
class NotRegistered(RuntimeError):
|
|
119
|
+
"""Registration failed, so the supervisor is going to kill this child anyway.
|
|
120
|
+
|
|
121
|
+
Refusing here names the cause. `await_registration` can only report that nothing appeared
|
|
122
|
+
on the roster, which reads as a broken adapter — the misattribution PRD-22 S2 exists to
|
|
123
|
+
prevent.
|
|
124
|
+
"""
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def register(client, *, code: str, model: str, worktree: str, branch: str) -> tuple[str, str, dict, str]:
|
|
128
|
+
"""Redeem the seat and come back with this child's server-side identity.
|
|
129
|
+
|
|
130
|
+
Takes the client rather than building one, so the wiring is testable without a server —
|
|
131
|
+
the property that matters is that the id the SERVER returned is the one the run uses, and
|
|
132
|
+
a helper that made its own connection could only be checked by reading the source.
|
|
133
|
+
|
|
134
|
+
`worktree` is what `spawn.await_registration` matches on (D-g: one worker, one worktree).
|
|
135
|
+
`capabilities.vendor` is what drives review diversity, so a local tier is distinguishable
|
|
136
|
+
from a frontier one on the roster.
|
|
137
|
+
"""
|
|
138
|
+
try:
|
|
139
|
+
me = client.call(
|
|
140
|
+
"register_agent",
|
|
141
|
+
enrolment_code=code,
|
|
142
|
+
label=f"gbagent/{model}",
|
|
143
|
+
worktree=worktree,
|
|
144
|
+
branch=branch,
|
|
145
|
+
capabilities={"vendor": "gbagent", "model": model, "tier": "local"},
|
|
146
|
+
)
|
|
147
|
+
except Exception as exc: # noqa: BLE001 — every failure here has the same consequence
|
|
148
|
+
raise NotRegistered(f"could not register: {exc}") from None
|
|
149
|
+
agent_id = str(me.get("agent_id") or "")
|
|
150
|
+
if not agent_id:
|
|
151
|
+
raise NotRegistered("register_agent returned no agent_id")
|
|
152
|
+
role = str(me.get("active_role") or "")
|
|
153
|
+
off = me.get("tools_off_limits") or []
|
|
154
|
+
if "create_item" in off:
|
|
155
|
+
# P30 D11. A worker that cannot create cannot file a typed human wait.
|
|
156
|
+
# That seat is a mis-mint, not a child that should limp on with free-text
|
|
157
|
+
# `blocker`. S6: reviewer merged into worker, so every child is a worker.
|
|
158
|
+
raise NotRegistered(
|
|
159
|
+
"this seat cannot create_item — a worker that cannot file a human wait "
|
|
160
|
+
"is a mis-mint (P30 D11)"
|
|
161
|
+
)
|
|
162
|
+
# PRD-36 D4: what a BOUND seat handed this child. `none` on an unbound seat; a server
|
|
163
|
+
# that predates PRD-36 sends no key, which reads the same as `none` here.
|
|
164
|
+
assigned = me.get("assigned") if isinstance(me.get("assigned"), dict) else {}
|
|
165
|
+
# GRPH-719: the project this child landed on. Named on every later call, so a credential
|
|
166
|
+
# spanning several projects does not send the child's reads to its default project.
|
|
167
|
+
# A server that predates the field sends none, and the client then names nothing.
|
|
168
|
+
project = str(me.get("project_id") or "")
|
|
169
|
+
return agent_id, role, {"item": assigned.get("item"), "state": assigned.get("state") or "none",
|
|
170
|
+
"reason": assigned.get("reason"), "held_by": assigned.get("held_by")}, project
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def enrolment_code(instruction: str) -> str:
|
|
174
|
+
"""The seat out of the instruction the supervisor wrote. "" when there is none."""
|
|
175
|
+
found = ENROLMENT.search(instruction or "")
|
|
176
|
+
return found.group(1) if found else ""
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def task_from(instruction: str) -> str:
|
|
180
|
+
"""The instruction with the REGISTRATION sentence removed.
|
|
181
|
+
|
|
182
|
+
**FOUND BY THE FIRST SUPERVISOR-SPAWNED BUILD.** `spawn` writes one instruction for every
|
|
183
|
+
adapter and it opens by telling the child to call `register_agent` — correct for a vendor
|
|
184
|
+
harness, which registers by being prompted to. gbagent registers in `_run` before the model
|
|
185
|
+
exists, and `register_agent` is deliberately not among the tools it advertises. So the
|
|
186
|
+
model was being told, as its first instruction, to call a tool it does not have. It spent
|
|
187
|
+
thirty turns on it and claimed nothing.
|
|
188
|
+
|
|
189
|
+
Only that sentence goes. Everything after it is still exactly right for this agent: it IS a
|
|
190
|
+
separate process, it must NOT declare parentage, and exiting on an empty queue is the
|
|
191
|
+
normal end of its run (D-b, D-c).
|
|
192
|
+
|
|
193
|
+
Keyed on the sentence rather than on line 1, so a reordered instruction loses the right
|
|
194
|
+
line — and a test renders `seat.INSTRUCTION` and asserts what survives.
|
|
195
|
+
"""
|
|
196
|
+
kept = [line for line in (instruction or "").splitlines()
|
|
197
|
+
if "register_agent" not in line]
|
|
198
|
+
return "\n".join(kept).strip()
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
class SeatUnreadable(RuntimeError):
|
|
202
|
+
"""The MCP config the supervisor wrote is not one this agent can use."""
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def read_seat(path: Path) -> tuple[str, str]:
|
|
206
|
+
"""Pull the server URL and credential out of the seat file `spawn` wrote.
|
|
207
|
+
|
|
208
|
+
Refuses rather than degrading. A missing key here means an agent that starts, cannot
|
|
209
|
+
reach the server, and burns its whole turn budget discovering it — the expensive shape
|
|
210
|
+
of the same mistake `config.load` refuses at spawn.
|
|
211
|
+
"""
|
|
212
|
+
try:
|
|
213
|
+
data = json.loads(Path(path).read_text(encoding="utf-8"))
|
|
214
|
+
server = data["mcpServers"]["graphban"]
|
|
215
|
+
url = str(server["url"])
|
|
216
|
+
key = str(server["headers"]["X-API-Key"])
|
|
217
|
+
except (OSError, ValueError, KeyError, TypeError) as exc:
|
|
218
|
+
raise SeatUnreadable(f"{path}: not a Graphban MCP config ({exc})") from None
|
|
219
|
+
if not url or not key:
|
|
220
|
+
raise SeatUnreadable(f"{path}: the Graphban entry has no url or no X-API-Key")
|
|
221
|
+
# `mcp_config` writes the endpoint, and the client appends it again.
|
|
222
|
+
return url[: -len("/api/mcp")] if url.endswith("/api/mcp") else url, key
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
def _models(base_url: str) -> list[str]:
|
|
226
|
+
"""What the endpoint serves, one per line. Empty when there is nothing to ask."""
|
|
227
|
+
if not base_url:
|
|
228
|
+
return []
|
|
229
|
+
session = OllamaSession(base_url, "", system="", task="", api_key=endpoint_key())
|
|
230
|
+
try:
|
|
231
|
+
return session.list_models()
|
|
232
|
+
finally:
|
|
233
|
+
session.close()
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
def _trace(event: "loop.Trace") -> None:
|
|
237
|
+
"""One line per thing that happened, to the child's own stderr (GRPH-506).
|
|
238
|
+
|
|
239
|
+
stderr because that is what `spawn` captures to a file the supervisor can read, and
|
|
240
|
+
because a fleet child has nowhere else to say anything. One line each, bounded upstream —
|
|
241
|
+
a forty-turn run should be readable, not re-livable.
|
|
242
|
+
"""
|
|
243
|
+
if event.kind == "turn":
|
|
244
|
+
said = f" {event.text}" if event.text else ""
|
|
245
|
+
print(f"gbagent: [{event.turn:>2}] model:{said}", file=sys.stderr, flush=True)
|
|
246
|
+
else:
|
|
247
|
+
mark = "ok " if event.ok else "ERR"
|
|
248
|
+
print(f"gbagent: [{event.turn:>2}] {mark} {event.name}: {event.text}",
|
|
249
|
+
file=sys.stderr, flush=True)
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
def _run(args: argparse.Namespace) -> int:
|
|
253
|
+
root = Path(args.worktree).resolve()
|
|
254
|
+
|
|
255
|
+
try:
|
|
256
|
+
base_url, api_key = read_seat(Path(args.mcp_config))
|
|
257
|
+
except SeatUnreadable as exc:
|
|
258
|
+
print(f"gbagent: {exc}", file=sys.stderr)
|
|
259
|
+
return 78
|
|
260
|
+
|
|
261
|
+
written = Path(args.instruction_file).read_text(encoding="utf-8") if args.instruction_file else ""
|
|
262
|
+
# The registration sentence is the harness's job and names a tool the model does not have.
|
|
263
|
+
task = task_from(written)
|
|
264
|
+
|
|
265
|
+
# REGISTER BEFORE PREPARE (P30 D8 / GRPH-503). `spawn.await_registration` polls the
|
|
266
|
+
# roster for 90s and kills an unregistered child, blaming the adapter. `prepare()`
|
|
267
|
+
# can run `uv pip install` for 900s. Counting that against the 90s window makes a
|
|
268
|
+
# cold worktree look like a broken adapter. Presence-only heartbeats (no item id)
|
|
269
|
+
# keep the roster alive during setup. Do not stretch registration to 900s.
|
|
270
|
+
agent_id = args.agent_id
|
|
271
|
+
project = ""
|
|
272
|
+
role = ""
|
|
273
|
+
if not agent_id:
|
|
274
|
+
code = enrolment_code(written)
|
|
275
|
+
if not code:
|
|
276
|
+
print(
|
|
277
|
+
"gbagent: no enrolment code in the instruction file and no --agent-id. A child "
|
|
278
|
+
"that does not register is one the supervisor kills for looking like a broken "
|
|
279
|
+
"adapter, so this refuses instead and says which it was.",
|
|
280
|
+
file=sys.stderr,
|
|
281
|
+
)
|
|
282
|
+
return 78
|
|
283
|
+
try:
|
|
284
|
+
agent_id, role, assigned, project = register(
|
|
285
|
+
Coordinator.connect(base_url, api_key, item_id="").client,
|
|
286
|
+
code=code, model=args.model, worktree=str(root), branch=args.branch,
|
|
287
|
+
)
|
|
288
|
+
except NotRegistered as exc:
|
|
289
|
+
print(f"gbagent: {exc}", file=sys.stderr)
|
|
290
|
+
return 78
|
|
291
|
+
print(f"gbagent: registered {agent_id} as {role!r}", file=sys.stderr)
|
|
292
|
+
# PRD-36 D3/D4: the server's answer outranks --item. `claimed` means this child
|
|
293
|
+
# already HOLDS the seat's item — no claim_cluster, and the heartbeat carries it from
|
|
294
|
+
# the first beat. `taken` means somebody else holds it: exit, the normal end of a
|
|
295
|
+
# run with nothing to do, and say who.
|
|
296
|
+
if assigned["state"] == "claimed" and assigned["item"]:
|
|
297
|
+
args.item = str(assigned["item"])
|
|
298
|
+
print(f"gbagent: this seat handed me {args.item}", file=sys.stderr)
|
|
299
|
+
elif assigned["state"] == "taken":
|
|
300
|
+
print(
|
|
301
|
+
f"gbagent: this seat was bound to {assigned['item']} but it is {assigned['reason']}"
|
|
302
|
+
+ (f" by {assigned['held_by']}" if assigned.get("held_by") else "")
|
|
303
|
+
+ " — nothing to do, exiting",
|
|
304
|
+
file=sys.stderr,
|
|
305
|
+
)
|
|
306
|
+
return 0
|
|
307
|
+
assignment = assignment_for(args.item, role=role)
|
|
308
|
+
# S6 (PRD-39 D-h): merged worker gets both build and review tools.
|
|
309
|
+
tools = MERGED_TOOLS
|
|
310
|
+
coordinator = Coordinator.connect(base_url, api_key, item_id=args.item,
|
|
311
|
+
agent_id=agent_id, allowed=tools, project_id=project)
|
|
312
|
+
heartbeat = Heartbeat(coordinator)
|
|
313
|
+
heartbeat.start()
|
|
314
|
+
session = None
|
|
315
|
+
try:
|
|
316
|
+
try:
|
|
317
|
+
# AFTER register. The executable check inside `load` is what an unbuilt
|
|
318
|
+
# worktree fails; a fresh `git worktree` is what PRD-22 hands every child
|
|
319
|
+
# (GRPH-502). The heartbeat above is presence-only until a claim lands.
|
|
320
|
+
built = prepare(root)
|
|
321
|
+
for command in built:
|
|
322
|
+
print(f"gbagent: setup ran {command!r}", file=sys.stderr)
|
|
323
|
+
cfg = load(root)
|
|
324
|
+
except ConfigRefused as exc:
|
|
325
|
+
print(f"gbagent: {exc}", file=sys.stderr)
|
|
326
|
+
return 78 # EX_CONFIG. Distinct from a crash, and from giving up.
|
|
327
|
+
|
|
328
|
+
try:
|
|
329
|
+
# S6 (PRD-39 D-h): merged worker orientation covers both build and review.
|
|
330
|
+
orientation = build_orientation(
|
|
331
|
+
coordinator.client, extra=MERGED_COORDINATION, agent_id=agent_id,
|
|
332
|
+
)
|
|
333
|
+
except OrientationUnavailable as exc:
|
|
334
|
+
print(f"gbagent: {exc}", file=sys.stderr)
|
|
335
|
+
return 78
|
|
336
|
+
toolset = Toolset(root=root, cfg=cfg, orientation=orientation)
|
|
337
|
+
# The heartbeat thread was started before the toolset existed; from here on it
|
|
338
|
+
# reports what the model is doing (PRD-34 D12).
|
|
339
|
+
coordinator.status_source = toolset.activity
|
|
340
|
+
session = OllamaSession(
|
|
341
|
+
args.base_url, args.model,
|
|
342
|
+
system=SYSTEM,
|
|
343
|
+
task=f"{task}\n\n{assignment}".strip(),
|
|
344
|
+
api_key=endpoint_key(),
|
|
345
|
+
)
|
|
346
|
+
try:
|
|
347
|
+
outcome = loop.run(session, toolset, coordinator=coordinator,
|
|
348
|
+
window=args.window, budget=args.turns, heartbeat=heartbeat,
|
|
349
|
+
trace=_trace)
|
|
350
|
+
except ModelUnreachable as exc:
|
|
351
|
+
print(f"gbagent: {exc}", file=sys.stderr)
|
|
352
|
+
return 69 # EX_UNAVAILABLE. The endpoint, not this agent, and not a give-up.
|
|
353
|
+
|
|
354
|
+
print(_summary(outcome, graph_calls=orientation.calls, beats=heartbeat.beats),
|
|
355
|
+
file=sys.stderr)
|
|
356
|
+
# The RESULT RECORD, on stdout, one line, machine-readable (PRD-38 D3). Every other
|
|
357
|
+
# vendor has one — qwen's `-o json`, claude's `--output-format json` — and the
|
|
358
|
+
# supervisor's exit report reads it to say what a run cost. gbagent had none, so its
|
|
359
|
+
# cells read "not comparable: 0 of N attempts reported tokens" while the endpoint was
|
|
360
|
+
# reporting the numbers on every turn and the loop was dropping them.
|
|
361
|
+
#
|
|
362
|
+
# stdout, not stderr: stderr is the human trace and it interleaves with the model's
|
|
363
|
+
# own chatter. A record a machine has to find inside that is a record that will
|
|
364
|
+
# eventually be mis-parsed.
|
|
365
|
+
print(json.dumps({"gbagent": _result_record(outcome)}), flush=True)
|
|
366
|
+
return outcome.exit_code
|
|
367
|
+
finally:
|
|
368
|
+
heartbeat.stop()
|
|
369
|
+
if session is not None:
|
|
370
|
+
session.close()
|
|
371
|
+
|
|
372
|
+
|
|
373
|
+
def _result_record(outcome) -> dict:
|
|
374
|
+
"""What a run cost, in the terms `attempt_telemetry` records.
|
|
375
|
+
|
|
376
|
+
`tokens_in`/`tokens_out` are null when the endpoint never reported usage — not zero. A
|
|
377
|
+
zero would say "this run was free", and the ledger's whole cost story rests on telling
|
|
378
|
+
"nobody said" apart from "nothing was spent" (PRD-38 D3, D11).
|
|
379
|
+
"""
|
|
380
|
+
reported = outcome.tokens_in or outcome.tokens_out
|
|
381
|
+
return {
|
|
382
|
+
"status": outcome.status,
|
|
383
|
+
"exit": outcome.exit_code,
|
|
384
|
+
"turns": outcome.turns,
|
|
385
|
+
"tokens_in": outcome.tokens_in if reported else None,
|
|
386
|
+
"tokens_out": outcome.tokens_out if reported else None,
|
|
387
|
+
"compactions": outcome.compactions,
|
|
388
|
+
}
|
|
389
|
+
|
|
390
|
+
|
|
391
|
+
def _summary(outcome, *, graph_calls: int, beats: int) -> str:
|
|
392
|
+
"""The one line a human reads about a run.
|
|
393
|
+
|
|
394
|
+
**`NEVER`, not 0, when the agent never wrote** — see docs/orientation-metric-prd24.md.
|
|
395
|
+
`Outcome.turns_to_first_write` is `None` in that case and four tests pin it, but this
|
|
396
|
+
line is what anybody actually sees, and it was pinned by nothing: rendering it as
|
|
397
|
+
`{first or 0}` left `Outcome` carrying `None`, every value-layer assertion holding, and
|
|
398
|
+
the reader told "first write on turn 0" (GRPH-533).
|
|
399
|
+
|
|
400
|
+
That matters more here than the value does. The S7 walk's run 1 claimed an item, ran the
|
|
401
|
+
suite, passed BECAUSE IT HAD CHANGED NOTHING, and moved the item to review with "Ran all
|
|
402
|
+
tests and verified the fix". Nothing else in the stack noticed — the server does not know
|
|
403
|
+
worktrees exist, and an item arriving in review with a receipt looks like finished work.
|
|
404
|
+
Somebody reading THIS LINE is how it was caught, and averaged in as 0 that run scores as
|
|
405
|
+
the best one ever recorded.
|
|
406
|
+
|
|
407
|
+
Extracted from `_run` so it can be asserted at all. Inline in a function that opens a
|
|
408
|
+
model session and a heartbeat thread, it was unreachable from a test — which is why the
|
|
409
|
+
value grew four guards and the sentence grew none.
|
|
410
|
+
"""
|
|
411
|
+
first = outcome.turns_to_first_write
|
|
412
|
+
return (f"gbagent: {outcome.status} after {outcome.turns} turns "
|
|
413
|
+
f"({outcome.compactions} compaction(s), {graph_calls} graph call(s), "
|
|
414
|
+
f"{beats} heartbeat(s), "
|
|
415
|
+
f"first write on turn {first if first is not None else 'NEVER'})"
|
|
416
|
+
f" — {outcome.meaning}")
|
|
417
|
+
|
|
418
|
+
|
|
419
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
420
|
+
parser = argparse.ArgumentParser(prog="gbagent", description=__doc__.splitlines()[0])
|
|
421
|
+
parser.add_argument("--version", action="version",
|
|
422
|
+
version=f"gbagent {gbfleet.__version__}")
|
|
423
|
+
sub = parser.add_subparsers(dest="command")
|
|
424
|
+
|
|
425
|
+
run = sub.add_parser("run", help="build one item in one worktree")
|
|
426
|
+
run.add_argument("--worktree", required=True)
|
|
427
|
+
run.add_argument("--mcp-config", required=True)
|
|
428
|
+
run.add_argument("--instruction-file", default="")
|
|
429
|
+
run.add_argument("--item", default="")
|
|
430
|
+
# An OVERRIDE, not the normal path: a walk re-running a stuck item should not have to mint
|
|
431
|
+
# a seat. Given one, registration is skipped entirely.
|
|
432
|
+
run.add_argument("--agent-id", default="")
|
|
433
|
+
run.add_argument("--branch", default="")
|
|
434
|
+
run.add_argument("--model", required=True)
|
|
435
|
+
# No defaults. `loop.run` refuses to guess either of these and so does this.
|
|
436
|
+
run.add_argument("--turns", type=int, required=True)
|
|
437
|
+
run.add_argument("--window", type=int, required=True)
|
|
438
|
+
run.add_argument("--base-url", default="")
|
|
439
|
+
|
|
440
|
+
sub.add_parser("models", help=f"list what {BASE_URL_ENV} serves ({API_KEY_ENV} is sent as a bearer when set)")
|
|
441
|
+
return parser
|
|
442
|
+
|
|
443
|
+
|
|
444
|
+
def main(argv: list[str] | None = None) -> int:
|
|
445
|
+
parser = build_parser()
|
|
446
|
+
args = parser.parse_args(argv)
|
|
447
|
+
if args.command is None:
|
|
448
|
+
parser.error("no command given — try `gbagent --help`")
|
|
449
|
+
if args.command == "models":
|
|
450
|
+
for name in _models(os.environ.get(BASE_URL_ENV, "")):
|
|
451
|
+
print(name)
|
|
452
|
+
return 0
|
|
453
|
+
if not args.base_url:
|
|
454
|
+
args.base_url = os.environ.get(BASE_URL_ENV, "")
|
|
455
|
+
if not args.base_url:
|
|
456
|
+
print(f"gbagent: no model endpoint. Set {BASE_URL_ENV} or pass --base-url.",
|
|
457
|
+
file=sys.stderr)
|
|
458
|
+
return 78
|
|
459
|
+
return _run(args)
|
|
460
|
+
|
|
461
|
+
|
|
462
|
+
if __name__ == "__main__": # pragma: no cover
|
|
463
|
+
raise SystemExit(main())
|