graphban-fleet 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
gbagent/__init__.py ADDED
@@ -0,0 +1,6 @@
1
+ """gbagent — a first-party coding agent for local models (PRD-24).
2
+
3
+ Ships inside `graphban-fleet` and is versioned with it (D8): the version range machinery in
4
+ `gbfleet.adapters` exists because three vendors ship on their own schedules, and the
5
+ first-party adapter is the one case where that problem does not arise.
6
+ """
gbagent/cli.py ADDED
@@ -0,0 +1,463 @@
1
+ """`gbagent` — the command the supervisor launches (PRD-24 D8, S5).
2
+
3
+ Three commands, and the small one matters most: `--version` is what
4
+ `adapters/gbagent.py` resolves against, and because the pin is exact it is also how a
5
+ `gbagent` from a different install gets caught before a process does any work.
6
+
7
+ **`models` exists so a typo is refused at spawn.** GRPH-485 was this failure the long way
8
+ round — a model name that did not exist, found days later as a grill that would not converge.
9
+ The adapter cannot ask an endpoint itself (only two modules in this package may open a socket),
10
+ so it shells out to here.
11
+
12
+ **`run` can pick up its own work.** `--item` is optional from S7 on: without one the model
13
+ calls `claim_cluster` itself, which is in `coord.WORKER_TOOLS` along with the rest of
14
+ `COORDINATION_TOOLS` (P30 D3). `claim_next` is not advertised: it reserves no files.
15
+
16
+ This paragraph used to say the opposite, and was true when written — a later slice wired the
17
+ thing it described as unwired, and the prose did not follow (GRPH-562). Corrected rather than
18
+ deleted, because the mistake is worth not repeating: **a tool set is a declaration of intent,
19
+ not an enforcement boundary**, so a docstring reasoning about authority from set membership is
20
+ describing the wrong object. The file even disagreed with itself — `assignment_for` below has
21
+ said all along that the model claims for itself.
22
+
23
+ What actually stops a worker overreaching is on the server: `TOOL_ROLES` gates what a
24
+ credential may call, `independent()` refuses a sign-off from the author whatever tools it
25
+ holds, and D5 clamps a worker at `review` — done is not the agent's word. `assert not
26
+ WORKER_TOOLS & ALLOWED_TOOLS` pins that a worker is not a supervisor, which is the one thing
27
+ the set itself is good for.
28
+ """
29
+ from __future__ import annotations
30
+
31
+ import argparse
32
+ import json
33
+ import os
34
+ import re
35
+ import sys
36
+ from pathlib import Path
37
+
38
+ import gbfleet
39
+
40
+ from . import loop
41
+ from .config import ConfigRefused, load, prepare
42
+ from .coord import (
43
+ MERGED_COORDINATION, MERGED_TOOLS, REVIEWER_COORDINATION, REVIEWER_TOOLS,
44
+ WORKER_TOOLS, Coordinator,
45
+ )
46
+ from .heartbeat import Heartbeat
47
+ from .llm import ModelUnreachable, OllamaSession
48
+ from .orient import (
49
+ COORDINATION_TOOLS,
50
+ INSTRUCTION as ORIENT_INSTRUCTION,
51
+ OrientationUnavailable,
52
+ build as build_orientation,
53
+ )
54
+ from .toolset import Toolset
55
+
56
+ #: Where the model endpoint lives. Named, never discovered — the same argument D3 makes
57
+ #: about the test command.
58
+ BASE_URL_ENV = "GBAGENT_BASE_URL"
59
+ #: A bearer for the model endpoint, when it wants one. Environment only, never argv: the
60
+ #: fleet's rule is that nothing carrying a credential goes on a command line (`ps` shows it),
61
+ #: and the supervisor's child inherits the operator's environment (spawn.py). Unset means an
62
+ #: unauthenticated endpoint — a local Ollama — which is what every walk so far has used.
63
+ API_KEY_ENV = "GBAGENT_API_KEY"
64
+
65
+
66
+ def endpoint_key() -> str:
67
+ """What `GBAGENT_API_KEY` holds, or "" — the only way a model credential reaches gbagent."""
68
+ return os.environ.get(API_KEY_ENV, "")
69
+
70
+ #: What the model is told before anything else. Two jobs: say what it cannot do, so it does
71
+ #: not spend 30-second turns finding out, and say what to reach for FIRST (S6).
72
+ SYSTEM = (
73
+ "You are gbagent, an unattended coding agent working inside one git worktree.\n"
74
+ "Use the tools. Do not narrate what you are about to do — do it.\n"
75
+ "Paths are relative to the worktree root. You cannot write outside it and you have no "
76
+ "shell; run_tests runs the command this repository declares.\n"
77
+ "\n" + ORIENT_INSTRUCTION + "\n"
78
+ "\nWhen the tests pass, say DONE and stop calling tools."
79
+ )
80
+
81
+
82
+ def assignment_for(item: str, role: str = "worker") -> str:
83
+ """What the model is told to work on.
84
+
85
+ S6 (PRD-39 D-h): one loop — try `claim_review`, fall through to `claim_cluster`,
86
+ exit when both are empty. The role parameter is kept for backward compatibility
87
+ but no longer changes the assignment: a worker now claims, builds, AND reviews.
88
+
89
+ `--item` is optional from S7 on. Without one the model calls `claim_review` then
90
+ `claim_cluster` itself. With one it works the item it was handed.
91
+ """
92
+ if item:
93
+ return (
94
+ f"You are working on {item}. Do not claim other build work. When {item} is in "
95
+ "review, call claim_review with wait_seconds=0 and review what you did not build "
96
+ "— sign_off, or bounce with a reason — until it answers nothing; then say DONE "
97
+ "and stop."
98
+ )
99
+ return (
100
+ "Call claim_review with wait_seconds=0. If there is nothing to review, "
101
+ "call claim_cluster with wait_seconds=0 to take the next ready non-colliding "
102
+ "cluster. If both are empty, say DONE and stop — exiting on an empty queue "
103
+ "is the normal end of your run, not a failure. You may sign_off work you "
104
+ "did not build. If the work is not ready — tests fail, the change is wrong — "
105
+ "bounce it with a reason naming what is wrong, then say DONE."
106
+ )
107
+
108
+
109
+ #: How the enrolment code is read back out of the instruction file.
110
+ #:
111
+ #: `spawn` writes the code there and deliberately never into the MCP config — the code is an
112
+ #: argument to `register_agent`, not a config value (`seat.mcp_config`). The format is OURS
113
+ #: (`seat.INSTRUCTION`), and a test pins this pattern against that constant, so a reworded
114
+ #: instruction fails a test rather than producing a child that silently never registers.
115
+ ENROLMENT = re.compile(r"enrolment_code=['\"]([^'\"]+)['\"]")
116
+
117
+
118
+ class NotRegistered(RuntimeError):
119
+ """Registration failed, so the supervisor is going to kill this child anyway.
120
+
121
+ Refusing here names the cause. `await_registration` can only report that nothing appeared
122
+ on the roster, which reads as a broken adapter — the misattribution PRD-22 S2 exists to
123
+ prevent.
124
+ """
125
+
126
+
127
+ def register(client, *, code: str, model: str, worktree: str, branch: str) -> tuple[str, str, dict, str]:
128
+ """Redeem the seat and come back with this child's server-side identity.
129
+
130
+ Takes the client rather than building one, so the wiring is testable without a server —
131
+ the property that matters is that the id the SERVER returned is the one the run uses, and
132
+ a helper that made its own connection could only be checked by reading the source.
133
+
134
+ `worktree` is what `spawn.await_registration` matches on (D-g: one worker, one worktree).
135
+ `capabilities.vendor` is what drives review diversity, so a local tier is distinguishable
136
+ from a frontier one on the roster.
137
+ """
138
+ try:
139
+ me = client.call(
140
+ "register_agent",
141
+ enrolment_code=code,
142
+ label=f"gbagent/{model}",
143
+ worktree=worktree,
144
+ branch=branch,
145
+ capabilities={"vendor": "gbagent", "model": model, "tier": "local"},
146
+ )
147
+ except Exception as exc: # noqa: BLE001 — every failure here has the same consequence
148
+ raise NotRegistered(f"could not register: {exc}") from None
149
+ agent_id = str(me.get("agent_id") or "")
150
+ if not agent_id:
151
+ raise NotRegistered("register_agent returned no agent_id")
152
+ role = str(me.get("active_role") or "")
153
+ off = me.get("tools_off_limits") or []
154
+ if "create_item" in off:
155
+ # P30 D11. A worker that cannot create cannot file a typed human wait.
156
+ # That seat is a mis-mint, not a child that should limp on with free-text
157
+ # `blocker`. S6: reviewer merged into worker, so every child is a worker.
158
+ raise NotRegistered(
159
+ "this seat cannot create_item — a worker that cannot file a human wait "
160
+ "is a mis-mint (P30 D11)"
161
+ )
162
+ # PRD-36 D4: what a BOUND seat handed this child. `none` on an unbound seat; a server
163
+ # that predates PRD-36 sends no key, which reads the same as `none` here.
164
+ assigned = me.get("assigned") if isinstance(me.get("assigned"), dict) else {}
165
+ # GRPH-719: the project this child landed on. Named on every later call, so a credential
166
+ # spanning several projects does not send the child's reads to its default project.
167
+ # A server that predates the field sends none, and the client then names nothing.
168
+ project = str(me.get("project_id") or "")
169
+ return agent_id, role, {"item": assigned.get("item"), "state": assigned.get("state") or "none",
170
+ "reason": assigned.get("reason"), "held_by": assigned.get("held_by")}, project
171
+
172
+
173
+ def enrolment_code(instruction: str) -> str:
174
+ """The seat out of the instruction the supervisor wrote. "" when there is none."""
175
+ found = ENROLMENT.search(instruction or "")
176
+ return found.group(1) if found else ""
177
+
178
+
179
+ def task_from(instruction: str) -> str:
180
+ """The instruction with the REGISTRATION sentence removed.
181
+
182
+ **FOUND BY THE FIRST SUPERVISOR-SPAWNED BUILD.** `spawn` writes one instruction for every
183
+ adapter and it opens by telling the child to call `register_agent` — correct for a vendor
184
+ harness, which registers by being prompted to. gbagent registers in `_run` before the model
185
+ exists, and `register_agent` is deliberately not among the tools it advertises. So the
186
+ model was being told, as its first instruction, to call a tool it does not have. It spent
187
+ thirty turns on it and claimed nothing.
188
+
189
+ Only that sentence goes. Everything after it is still exactly right for this agent: it IS a
190
+ separate process, it must NOT declare parentage, and exiting on an empty queue is the
191
+ normal end of its run (D-b, D-c).
192
+
193
+ Keyed on the sentence rather than on line 1, so a reordered instruction loses the right
194
+ line — and a test renders `seat.INSTRUCTION` and asserts what survives.
195
+ """
196
+ kept = [line for line in (instruction or "").splitlines()
197
+ if "register_agent" not in line]
198
+ return "\n".join(kept).strip()
199
+
200
+
201
+ class SeatUnreadable(RuntimeError):
202
+ """The MCP config the supervisor wrote is not one this agent can use."""
203
+
204
+
205
+ def read_seat(path: Path) -> tuple[str, str]:
206
+ """Pull the server URL and credential out of the seat file `spawn` wrote.
207
+
208
+ Refuses rather than degrading. A missing key here means an agent that starts, cannot
209
+ reach the server, and burns its whole turn budget discovering it — the expensive shape
210
+ of the same mistake `config.load` refuses at spawn.
211
+ """
212
+ try:
213
+ data = json.loads(Path(path).read_text(encoding="utf-8"))
214
+ server = data["mcpServers"]["graphban"]
215
+ url = str(server["url"])
216
+ key = str(server["headers"]["X-API-Key"])
217
+ except (OSError, ValueError, KeyError, TypeError) as exc:
218
+ raise SeatUnreadable(f"{path}: not a Graphban MCP config ({exc})") from None
219
+ if not url or not key:
220
+ raise SeatUnreadable(f"{path}: the Graphban entry has no url or no X-API-Key")
221
+ # `mcp_config` writes the endpoint, and the client appends it again.
222
+ return url[: -len("/api/mcp")] if url.endswith("/api/mcp") else url, key
223
+
224
+
225
+ def _models(base_url: str) -> list[str]:
226
+ """What the endpoint serves, one per line. Empty when there is nothing to ask."""
227
+ if not base_url:
228
+ return []
229
+ session = OllamaSession(base_url, "", system="", task="", api_key=endpoint_key())
230
+ try:
231
+ return session.list_models()
232
+ finally:
233
+ session.close()
234
+
235
+
236
+ def _trace(event: "loop.Trace") -> None:
237
+ """One line per thing that happened, to the child's own stderr (GRPH-506).
238
+
239
+ stderr because that is what `spawn` captures to a file the supervisor can read, and
240
+ because a fleet child has nowhere else to say anything. One line each, bounded upstream —
241
+ a forty-turn run should be readable, not re-livable.
242
+ """
243
+ if event.kind == "turn":
244
+ said = f" {event.text}" if event.text else ""
245
+ print(f"gbagent: [{event.turn:>2}] model:{said}", file=sys.stderr, flush=True)
246
+ else:
247
+ mark = "ok " if event.ok else "ERR"
248
+ print(f"gbagent: [{event.turn:>2}] {mark} {event.name}: {event.text}",
249
+ file=sys.stderr, flush=True)
250
+
251
+
252
+ def _run(args: argparse.Namespace) -> int:
253
+ root = Path(args.worktree).resolve()
254
+
255
+ try:
256
+ base_url, api_key = read_seat(Path(args.mcp_config))
257
+ except SeatUnreadable as exc:
258
+ print(f"gbagent: {exc}", file=sys.stderr)
259
+ return 78
260
+
261
+ written = Path(args.instruction_file).read_text(encoding="utf-8") if args.instruction_file else ""
262
+ # The registration sentence is the harness's job and names a tool the model does not have.
263
+ task = task_from(written)
264
+
265
+ # REGISTER BEFORE PREPARE (P30 D8 / GRPH-503). `spawn.await_registration` polls the
266
+ # roster for 90s and kills an unregistered child, blaming the adapter. `prepare()`
267
+ # can run `uv pip install` for 900s. Counting that against the 90s window makes a
268
+ # cold worktree look like a broken adapter. Presence-only heartbeats (no item id)
269
+ # keep the roster alive during setup. Do not stretch registration to 900s.
270
+ agent_id = args.agent_id
271
+ project = ""
272
+ role = ""
273
+ if not agent_id:
274
+ code = enrolment_code(written)
275
+ if not code:
276
+ print(
277
+ "gbagent: no enrolment code in the instruction file and no --agent-id. A child "
278
+ "that does not register is one the supervisor kills for looking like a broken "
279
+ "adapter, so this refuses instead and says which it was.",
280
+ file=sys.stderr,
281
+ )
282
+ return 78
283
+ try:
284
+ agent_id, role, assigned, project = register(
285
+ Coordinator.connect(base_url, api_key, item_id="").client,
286
+ code=code, model=args.model, worktree=str(root), branch=args.branch,
287
+ )
288
+ except NotRegistered as exc:
289
+ print(f"gbagent: {exc}", file=sys.stderr)
290
+ return 78
291
+ print(f"gbagent: registered {agent_id} as {role!r}", file=sys.stderr)
292
+ # PRD-36 D3/D4: the server's answer outranks --item. `claimed` means this child
293
+ # already HOLDS the seat's item — no claim_cluster, and the heartbeat carries it from
294
+ # the first beat. `taken` means somebody else holds it: exit, the normal end of a
295
+ # run with nothing to do, and say who.
296
+ if assigned["state"] == "claimed" and assigned["item"]:
297
+ args.item = str(assigned["item"])
298
+ print(f"gbagent: this seat handed me {args.item}", file=sys.stderr)
299
+ elif assigned["state"] == "taken":
300
+ print(
301
+ f"gbagent: this seat was bound to {assigned['item']} but it is {assigned['reason']}"
302
+ + (f" by {assigned['held_by']}" if assigned.get("held_by") else "")
303
+ + " — nothing to do, exiting",
304
+ file=sys.stderr,
305
+ )
306
+ return 0
307
+ assignment = assignment_for(args.item, role=role)
308
+ # S6 (PRD-39 D-h): merged worker gets both build and review tools.
309
+ tools = MERGED_TOOLS
310
+ coordinator = Coordinator.connect(base_url, api_key, item_id=args.item,
311
+ agent_id=agent_id, allowed=tools, project_id=project)
312
+ heartbeat = Heartbeat(coordinator)
313
+ heartbeat.start()
314
+ session = None
315
+ try:
316
+ try:
317
+ # AFTER register. The executable check inside `load` is what an unbuilt
318
+ # worktree fails; a fresh `git worktree` is what PRD-22 hands every child
319
+ # (GRPH-502). The heartbeat above is presence-only until a claim lands.
320
+ built = prepare(root)
321
+ for command in built:
322
+ print(f"gbagent: setup ran {command!r}", file=sys.stderr)
323
+ cfg = load(root)
324
+ except ConfigRefused as exc:
325
+ print(f"gbagent: {exc}", file=sys.stderr)
326
+ return 78 # EX_CONFIG. Distinct from a crash, and from giving up.
327
+
328
+ try:
329
+ # S6 (PRD-39 D-h): merged worker orientation covers both build and review.
330
+ orientation = build_orientation(
331
+ coordinator.client, extra=MERGED_COORDINATION, agent_id=agent_id,
332
+ )
333
+ except OrientationUnavailable as exc:
334
+ print(f"gbagent: {exc}", file=sys.stderr)
335
+ return 78
336
+ toolset = Toolset(root=root, cfg=cfg, orientation=orientation)
337
+ # The heartbeat thread was started before the toolset existed; from here on it
338
+ # reports what the model is doing (PRD-34 D12).
339
+ coordinator.status_source = toolset.activity
340
+ session = OllamaSession(
341
+ args.base_url, args.model,
342
+ system=SYSTEM,
343
+ task=f"{task}\n\n{assignment}".strip(),
344
+ api_key=endpoint_key(),
345
+ )
346
+ try:
347
+ outcome = loop.run(session, toolset, coordinator=coordinator,
348
+ window=args.window, budget=args.turns, heartbeat=heartbeat,
349
+ trace=_trace)
350
+ except ModelUnreachable as exc:
351
+ print(f"gbagent: {exc}", file=sys.stderr)
352
+ return 69 # EX_UNAVAILABLE. The endpoint, not this agent, and not a give-up.
353
+
354
+ print(_summary(outcome, graph_calls=orientation.calls, beats=heartbeat.beats),
355
+ file=sys.stderr)
356
+ # The RESULT RECORD, on stdout, one line, machine-readable (PRD-38 D3). Every other
357
+ # vendor has one — qwen's `-o json`, claude's `--output-format json` — and the
358
+ # supervisor's exit report reads it to say what a run cost. gbagent had none, so its
359
+ # cells read "not comparable: 0 of N attempts reported tokens" while the endpoint was
360
+ # reporting the numbers on every turn and the loop was dropping them.
361
+ #
362
+ # stdout, not stderr: stderr is the human trace and it interleaves with the model's
363
+ # own chatter. A record a machine has to find inside that is a record that will
364
+ # eventually be mis-parsed.
365
+ print(json.dumps({"gbagent": _result_record(outcome)}), flush=True)
366
+ return outcome.exit_code
367
+ finally:
368
+ heartbeat.stop()
369
+ if session is not None:
370
+ session.close()
371
+
372
+
373
+ def _result_record(outcome) -> dict:
374
+ """What a run cost, in the terms `attempt_telemetry` records.
375
+
376
+ `tokens_in`/`tokens_out` are null when the endpoint never reported usage — not zero. A
377
+ zero would say "this run was free", and the ledger's whole cost story rests on telling
378
+ "nobody said" apart from "nothing was spent" (PRD-38 D3, D11).
379
+ """
380
+ reported = outcome.tokens_in or outcome.tokens_out
381
+ return {
382
+ "status": outcome.status,
383
+ "exit": outcome.exit_code,
384
+ "turns": outcome.turns,
385
+ "tokens_in": outcome.tokens_in if reported else None,
386
+ "tokens_out": outcome.tokens_out if reported else None,
387
+ "compactions": outcome.compactions,
388
+ }
389
+
390
+
391
+ def _summary(outcome, *, graph_calls: int, beats: int) -> str:
392
+ """The one line a human reads about a run.
393
+
394
+ **`NEVER`, not 0, when the agent never wrote** — see docs/orientation-metric-prd24.md.
395
+ `Outcome.turns_to_first_write` is `None` in that case and four tests pin it, but this
396
+ line is what anybody actually sees, and it was pinned by nothing: rendering it as
397
+ `{first or 0}` left `Outcome` carrying `None`, every value-layer assertion holding, and
398
+ the reader told "first write on turn 0" (GRPH-533).
399
+
400
+ That matters more here than the value does. The S7 walk's run 1 claimed an item, ran the
401
+ suite, passed BECAUSE IT HAD CHANGED NOTHING, and moved the item to review with "Ran all
402
+ tests and verified the fix". Nothing else in the stack noticed — the server does not know
403
+ worktrees exist, and an item arriving in review with a receipt looks like finished work.
404
+ Somebody reading THIS LINE is how it was caught, and averaged in as 0 that run scores as
405
+ the best one ever recorded.
406
+
407
+ Extracted from `_run` so it can be asserted at all. Inline in a function that opens a
408
+ model session and a heartbeat thread, it was unreachable from a test — which is why the
409
+ value grew four guards and the sentence grew none.
410
+ """
411
+ first = outcome.turns_to_first_write
412
+ return (f"gbagent: {outcome.status} after {outcome.turns} turns "
413
+ f"({outcome.compactions} compaction(s), {graph_calls} graph call(s), "
414
+ f"{beats} heartbeat(s), "
415
+ f"first write on turn {first if first is not None else 'NEVER'})"
416
+ f" — {outcome.meaning}")
417
+
418
+
419
+ def build_parser() -> argparse.ArgumentParser:
420
+ parser = argparse.ArgumentParser(prog="gbagent", description=__doc__.splitlines()[0])
421
+ parser.add_argument("--version", action="version",
422
+ version=f"gbagent {gbfleet.__version__}")
423
+ sub = parser.add_subparsers(dest="command")
424
+
425
+ run = sub.add_parser("run", help="build one item in one worktree")
426
+ run.add_argument("--worktree", required=True)
427
+ run.add_argument("--mcp-config", required=True)
428
+ run.add_argument("--instruction-file", default="")
429
+ run.add_argument("--item", default="")
430
+ # An OVERRIDE, not the normal path: a walk re-running a stuck item should not have to mint
431
+ # a seat. Given one, registration is skipped entirely.
432
+ run.add_argument("--agent-id", default="")
433
+ run.add_argument("--branch", default="")
434
+ run.add_argument("--model", required=True)
435
+ # No defaults. `loop.run` refuses to guess either of these and so does this.
436
+ run.add_argument("--turns", type=int, required=True)
437
+ run.add_argument("--window", type=int, required=True)
438
+ run.add_argument("--base-url", default="")
439
+
440
+ sub.add_parser("models", help=f"list what {BASE_URL_ENV} serves ({API_KEY_ENV} is sent as a bearer when set)")
441
+ return parser
442
+
443
+
444
+ def main(argv: list[str] | None = None) -> int:
445
+ parser = build_parser()
446
+ args = parser.parse_args(argv)
447
+ if args.command is None:
448
+ parser.error("no command given — try `gbagent --help`")
449
+ if args.command == "models":
450
+ for name in _models(os.environ.get(BASE_URL_ENV, "")):
451
+ print(name)
452
+ return 0
453
+ if not args.base_url:
454
+ args.base_url = os.environ.get(BASE_URL_ENV, "")
455
+ if not args.base_url:
456
+ print(f"gbagent: no model endpoint. Set {BASE_URL_ENV} or pass --base-url.",
457
+ file=sys.stderr)
458
+ return 78
459
+ return _run(args)
460
+
461
+
462
+ if __name__ == "__main__": # pragma: no cover
463
+ raise SystemExit(main())