handcode 0.3.0rc1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (67) hide show
  1. agentctl/__init__.py +0 -0
  2. agentctl/adapters/__init__.py +0 -0
  3. agentctl/adapters/litellm/__init__.py +9 -0
  4. agentctl/adapters/litellm/hook.py +49 -0
  5. agentctl/adapters/litellm/recorder.py +187 -0
  6. agentctl/adapters/openhands/__init__.py +169 -0
  7. agentctl/adapters/openhands/handoff.py +155 -0
  8. agentctl/adapters/openhands/seam_b.py +259 -0
  9. agentctl/adapters/openhands/seam_c.py +209 -0
  10. agentctl/cli.py +1450 -0
  11. agentctl/control/__init__.py +0 -0
  12. agentctl/control/cost/__init__.py +4 -0
  13. agentctl/control/cost/ledger.py +210 -0
  14. agentctl/control/dash.py +697 -0
  15. agentctl/control/keys.py +440 -0
  16. agentctl/control/matrix/__init__.py +0 -0
  17. agentctl/control/matrix/data/tools.yaml +149 -0
  18. agentctl/control/policy/__init__.py +10 -0
  19. agentctl/control/policy/compile.py +258 -0
  20. agentctl/control/policy/data/policy.compiled.json +38 -0
  21. agentctl/control/policy/data/policy.yaml +46 -0
  22. agentctl/control/probe.py +399 -0
  23. agentctl/control/providers.py +293 -0
  24. agentctl/control/proxy.py +536 -0
  25. agentctl/control/proxyenv.py +309 -0
  26. agentctl/control/replay/__init__.py +14 -0
  27. agentctl/control/replay/cassette.py +281 -0
  28. agentctl/control/replay/server.py +109 -0
  29. agentctl/demo/__init__.py +214 -0
  30. agentctl/demo/child.py +84 -0
  31. agentctl/demo/mock.py +79 -0
  32. agentctl/demo/tool.py +62 -0
  33. agentctl/gha.py +488 -0
  34. agentctl/kernel/__init__.py +0 -0
  35. agentctl/kernel/classify.py +170 -0
  36. agentctl/kernel/gate.py +391 -0
  37. agentctl/kernel/hook.py +229 -0
  38. agentctl/kernel/ledger/__init__.py +0 -0
  39. agentctl/kernel/ledger/models.py +160 -0
  40. agentctl/kernel/ledger/schema.sql +62 -0
  41. agentctl/kernel/ledger/store.py +596 -0
  42. agentctl/kernel/paths.py +203 -0
  43. agentctl/kernel/policy.py +160 -0
  44. agentctl/kernel/reconcile/__init__.py +31 -0
  45. agentctl/kernel/reconcile/base.py +106 -0
  46. agentctl/kernel/reconcile/external.py +137 -0
  47. agentctl/kernel/reconcile/filesystem.py +162 -0
  48. agentctl/kernel/reconcile/git.py +162 -0
  49. agentctl/runtime/__init__.py +20 -0
  50. agentctl/runtime/citations.py +179 -0
  51. agentctl/runtime/config.py +97 -0
  52. agentctl/runtime/doctor.py +335 -0
  53. agentctl/runtime/init.py +148 -0
  54. agentctl/runtime/lease.py +143 -0
  55. agentctl/runtime/orchestrate.py +187 -0
  56. agentctl/runtime/plugins.py +130 -0
  57. agentctl/runtime/report.py +361 -0
  58. agentctl/runtime/runner.py +787 -0
  59. agentctl/runtime/runs.py +191 -0
  60. agentctl/runtime/subagent.py +274 -0
  61. agentctl/runtime/tools.py +350 -0
  62. handcode-0.3.0rc1.dist-info/METADATA +659 -0
  63. handcode-0.3.0rc1.dist-info/RECORD +67 -0
  64. handcode-0.3.0rc1.dist-info/WHEEL +5 -0
  65. handcode-0.3.0rc1.dist-info/entry_points.txt +3 -0
  66. handcode-0.3.0rc1.dist-info/licenses/LICENSE +21 -0
  67. handcode-0.3.0rc1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,787 @@
1
+ """Run a real agent on a real workspace, behind the gate.
2
+
3
+ This is the thing to use. Everything in `experiments/` is a test harness; this
4
+ does actual work.
5
+
6
+ agentctl run "add type hints to utils.py" --workspace ./myproject
7
+
8
+ What it wires up:
9
+
10
+ real tools bash, read, write -- classified by the capability matrix
11
+ the gate Seams B and C, so a crash cannot duplicate an effect
12
+ a real model via OpenRouter, or any OpenAI-compatible base URL
13
+ the ledgers effect ledger + Seam A telemetry for `agentctl cost`
14
+
15
+ **Two different safety properties, and they are not the same thing:**
16
+
17
+ * The **gate** stops an effect happening *twice*. That is replay safety.
18
+ * `--confirm-destructive` stops a dangerous effect happening *at all* without
19
+ a human saying yes. That is authorization.
20
+
21
+ A first-time `rm -rf` is not a replay, so the gate admits it. If you point this
22
+ at a repository you care about, keep the confirmation on — it is the default.
23
+
24
+ The confirmation asks on two independent grounds: the effect is `DESTRUCTIVE`,
25
+ or it writes somewhere outside the workspace. The second is not a lesser form
26
+ of the first — `echo x > ~/.bashrc` is an ordinary idempotent write that simply
27
+ is not the agent's business (`docs/0027`).
28
+ """
29
+ from __future__ import annotations
30
+
31
+ import os
32
+ import sys
33
+ import time
34
+ import uuid
35
+ from pathlib import Path
36
+
37
+ DEFAULT_MODEL = "openrouter/nvidia/nemotron-3-super-120b-a12b:free"
38
+
39
+ #: What `<workspace>/.agentctl/.gitignore` says. The directory holds the ledger
40
+ #: and every conversation -- task text, file contents, command output -- and
41
+ #: sat untracked in the user's repository, one `git add -A` from a commit
42
+ #: (`docs/0044` N11). `*` ignores all of it, this file included, so it never
43
+ #: appears in `git status`. `agents/` stays visible: subagent definitions are
44
+ #: written by the user and may belong in the repository.
45
+ STATE_GITIGNORE = ("# Written by agentctl: runtime state, never source.\n"
46
+ "*\n!agents/\n!agents/**\n")
47
+
48
+
49
+ SHOW_SYSTEM_PROMPT_ENV = "AGENTCTL_SHOW_SYSTEM_PROMPT"
50
+
51
+
52
+ def _visualizer():
53
+ """The SDK's own visualizer, minus the system prompt.
54
+
55
+ The default prints the whole system prompt and tool schemas before the
56
+ first turn: a 38-second run produced 687 lines, most of them text the user
57
+ did not write and cannot act on (`docs/0044` N10). It is replaced by one
58
+ line saying it was hidden and how to see it -- never silently dropped.
59
+ """
60
+ from openhands.sdk.conversation.visualizer import DefaultConversationVisualizer
61
+ from openhands.sdk.event import SystemPromptEvent
62
+
63
+ class _Quiet(DefaultConversationVisualizer):
64
+ def on_event(self, event):
65
+ if (isinstance(event, SystemPromptEvent)
66
+ and not os.environ.get(SHOW_SYSTEM_PROMPT_ENV)):
67
+ print(f" (system prompt hidden; set {SHOW_SYSTEM_PROMPT_ENV}=1 "
68
+ f"to show it)\n")
69
+ return
70
+ super().on_event(event)
71
+
72
+ return _Quiet()
73
+
74
+
75
+ def state_dir(ws: Path) -> Path:
76
+ """`<ws>/.agentctl`, created with its own `.gitignore`.
77
+
78
+ An existing `.gitignore` is left alone: if someone wrote one, it is theirs.
79
+ """
80
+ d = ws / ".agentctl"
81
+ d.mkdir(parents=True, exist_ok=True)
82
+ ignore = d / ".gitignore"
83
+ if not ignore.exists():
84
+ ignore.write_text(STATE_GITIGNORE, encoding="utf-8")
85
+ return d
86
+
87
+
88
+ def _build_agent(llm, tools):
89
+ """Construct the Agent with `tool_concurrency_limit` pinned to 1.
90
+
91
+ `agentctl/runtime/tools.py` implements `declared_resources()` on none of
92
+ its three tools, so the SDK's `ParallelToolExecutor` falls back to a
93
+ per-*tool-name* mutex (`declared=False` -> lock key `f"tool:{name}"`).
94
+ That lets `bash` -- which can `git commit` -- run concurrently with
95
+ `write_file` the moment the limit rises above 1, racing a ledger and a
96
+ git probe that both assume a single writer, and racing them *silently*:
97
+ every ledger write happens on the agent thread, so there is no exception
98
+ to notice (`docs/0038` §4.3).
99
+
100
+ Pinned explicitly here rather than relying on the SDK's own default
101
+ (currently 1, but not ours to depend on), and checked immediately after
102
+ construction so a future SDK upgrade, or an edit that drops the kwarg,
103
+ fails loudly instead of letting tools race unnoticed. This is the only
104
+ place in `agentctl/` allowed to construct an `Agent` -- see
105
+ `tests/test_tool_concurrency_pin.py`.
106
+ """
107
+ from openhands.sdk import Agent, Tool
108
+
109
+ agent = Agent(llm=llm, tools=[Tool(name=n) for n in tools],
110
+ tool_concurrency_limit=1, include_default_tools=[])
111
+ if agent.tool_concurrency_limit != 1:
112
+ raise RuntimeError(
113
+ "tool_concurrency_limit did not pin to 1. agentctl's tools do "
114
+ "not implement declared_resources(), so running them "
115
+ "concurrently can silently corrupt the effect ledger and the "
116
+ "git probe (docs/0038 §4.3). Refusing to start rather than run "
117
+ "unsafely.")
118
+ return agent
119
+
120
+
121
+ def _key_for(model: str) -> tuple[str | None, str | None]:
122
+ """Find a key for this model without ever printing it."""
123
+ for prefix, env in (("openrouter/", "OPENROUTER_API_KEY"),
124
+ ("anthropic/", "ANTHROPIC_API_KEY"),
125
+ ("openai/", "OPENAI_API_KEY"),
126
+ ("gemini/", "GEMINI_API_KEY"),
127
+ ("mistral/", "MISTRAL_API_KEY")):
128
+ if model.startswith(prefix):
129
+ return os.environ.get(env), env
130
+ for env in ("OPENROUTER_API_KEY", "OPENAI_API_KEY", "ANTHROPIC_API_KEY"):
131
+ if os.environ.get(env):
132
+ return os.environ[env], env
133
+ return None, None
134
+
135
+
136
+ def _env_var_for(model: str) -> str:
137
+ """The environment variable this model's provider actually reads."""
138
+ for prefix, env in (("openrouter/", "OPENROUTER_API_KEY"),
139
+ ("anthropic/", "ANTHROPIC_API_KEY"),
140
+ ("openai/", "OPENAI_API_KEY"),
141
+ ("gemini/", "GEMINI_API_KEY"),
142
+ ("mistral/", "MISTRAL_API_KEY"),
143
+ ("cerebras/", "CEREBRAS_API_KEY"),
144
+ ("groq/", "GROQ_API_KEY")):
145
+ if model.startswith(prefix):
146
+ return env
147
+ return "the API key variable for your provider"
148
+
149
+
150
+ def run(
151
+ task: str,
152
+ workspace: str | Path = ".",
153
+ *,
154
+ model: str = DEFAULT_MODEL,
155
+ base_url: str | None = None,
156
+ ledger: str | Path | None = None,
157
+ confirm_destructive: bool = True,
158
+ max_iterations: int = 30,
159
+ max_budget_usd: float | None = None,
160
+ resume: str | None = None,
161
+ record: str | Path | None = None,
162
+ replay: str | Path | None = None,
163
+ policy: str | Path | None = None,
164
+ cost_ledger: str | Path | None = None,
165
+ verbose: bool = True,
166
+ takeover: bool = False,
167
+ accept: str | None = None,
168
+ ) -> dict:
169
+ """Run one agent task. Returns a summary dict.
170
+
171
+ `record` writes every completion to a cassette; `replay` serves them back
172
+ from disk with no API key, no network and no sampling (M6, `docs/0029`).
173
+
174
+ `accept` is a command agentctl runs itself once the agent has finished --
175
+ the test suite, usually. Its exit code is the run's outcome; without it
176
+ the outcome is "not checked", never an implied success (`docs/0048`).
177
+ """
178
+ from agentctl.adapters.openhands import protect
179
+ from agentctl.runtime import tools as rt
180
+ from openhands.sdk import LLM, Conversation
181
+
182
+ from agentctl.runtime import report as rp
183
+
184
+ ws = Path(workspace).resolve()
185
+ ws.mkdir(parents=True, exist_ok=True)
186
+ start = rp.snapshot(ws)
187
+ state = state_dir(ws)
188
+ ledger_path = Path(ledger) if ledger else state / "ledger.db"
189
+
190
+ os.environ[rt.WORKSPACE_ENV] = str(ws)
191
+ os.environ.setdefault("AGENTCTL_TELEMETRY",
192
+ str(state / "hook_telemetry.json"))
193
+ # Do NOT register the plain tools here: protect() registers gated versions
194
+ # under the same names, and doing both only produces duplicate warnings.
195
+
196
+ # Replay stands in for the provider entirely: the cassette decides the
197
+ # model, so a `--model` flag cannot silently invalidate every fingerprint.
198
+ replay_server = None
199
+ if replay is not None:
200
+ from agentctl.control.replay import Cassette, ReplayServer
201
+
202
+ cassette = Cassette.load(replay)
203
+ if not cassette.turns:
204
+ raise SystemExit(f"cassette {replay} has no turns")
205
+ first = cassette.turns[0]
206
+ model = first.provider_model or first.model or model
207
+ replay_server = ReplayServer(cassette).start()
208
+ base_url = replay_server.base_url
209
+ if verbose:
210
+ print(f" replay {replay} ({len(cassette)} turns)")
211
+
212
+ api_key, key_env = _key_for(model)
213
+ if replay_server is not None:
214
+ api_key, key_env = "replay-no-key-needed", None
215
+ elif base_url and not api_key:
216
+ # Pointing at a proxy: the credentials live in the proxy, not here.
217
+ # litellm still requires *something* in the field and errors with
218
+ # "Missing credentials ... set OPENAI_API_KEY" if it is None -- which
219
+ # sends you looking for a key you deliberately do not have. The
220
+ # `agentctl proxy` instructions promised no client key was needed and
221
+ # this is what makes that true (`docs/0036`).
222
+ api_key, key_env = "proxy-holds-the-credentials", None
223
+ if not api_key and not base_url:
224
+ raise SystemExit(
225
+ f"No API key found for {model}.\n"
226
+ # Name the variable THIS model needs. Telling someone to set
227
+ # OPENROUTER_API_KEY for an anthropic/ model is advice that
228
+ # cannot work.
229
+ f" Set {_env_var_for(model)} in your environment.\n"
230
+ f" It is read from the environment and never logged.\n"
231
+ f" Check what you have: agentctl doctor")
232
+
233
+ recorder = None
234
+ if record is not None:
235
+ from agentctl.adapters.litellm.recorder import attach
236
+
237
+ recorder = attach(record, configured_model=model)
238
+ if verbose:
239
+ print(f" recording {record}")
240
+
241
+ pol = _load_policy(policy)
242
+ if pol:
243
+ _enforce_before_spending(pol, model, cost_ledger, max_budget_usd,
244
+ replay is not None, verbose)
245
+ if max_budget_usd is None:
246
+ max_budget_usd = pol.limit("per_task")
247
+ # `docs/0002` §5: never silently spend. A policy that names DESTRUCTIVE
248
+ # as requiring approval turns the confirmation on regardless of flags.
249
+ if pol.requires_approval("DESTRUCTIVE"):
250
+ confirm_destructive = True
251
+
252
+ cid = uuid.UUID(resume) if resume else uuid.uuid4()
253
+
254
+ decisions: list[dict] = []
255
+
256
+ def on_decision(call, decision):
257
+ decisions.append({"tool": call.tool_name,
258
+ "verdict": decision.verdict.value,
259
+ "class": (decision.effect_class.value
260
+ if decision.effect_class else None),
261
+ "reason": decision.reason})
262
+ if verbose and decision.verdict.value != "EXECUTE":
263
+ # The user's words, not the gate's (`docs/0044` F10).
264
+ word = {"SUBSTITUTE": "reused", "BLOCK": "paused",
265
+ "ESCALATE": "paused"}.get(decision.verdict.value,
266
+ decision.verdict.value)
267
+ print(f" [agentctl] {word} {call.tool_name}: "
268
+ f"{decision.reason}", file=sys.stderr)
269
+
270
+ # One driver per conversation (`docs/0042` I-02). This used to be
271
+ # `takeover=bool(resume)`: every resume stole the lease, including from a
272
+ # run still working in another terminal. Now a dead holder is taken over
273
+ # without asking, and a live one is refused unless `--takeover`.
274
+ from agentctl.kernel.ledger.store import LeaseHeld
275
+ from agentctl.runtime.lease import claim, holder_id
276
+
277
+ steal = False
278
+ if resume:
279
+ c = claim(ledger_path, str(cid), force=takeover)
280
+ steal = c.takeover
281
+ if c.note and verbose:
282
+ print(f" lease {c.note}")
283
+ try:
284
+ guard = protect(
285
+ ledger=ledger_path,
286
+ conversation_id=str(cid),
287
+ tools=rt.TOOLS,
288
+ repo_root=ws,
289
+ holder=holder_id(),
290
+ takeover=steal,
291
+ on_decision=on_decision,
292
+ )
293
+ except LeaseHeld as e:
294
+ # Taken between the check and the acquire. The same refusal.
295
+ raise SystemExit(f"{e}\n If that run is gone: add --takeover") from None
296
+
297
+ if confirm_destructive:
298
+ _install_confirmation(guard, verbose, workspace=ws)
299
+
300
+ llm = LLM(model=model, api_key=api_key, base_url=base_url,
301
+ service_id="agentctl-run", temperature=0.0,
302
+ num_retries=2, max_output_tokens=4096)
303
+ agent = _build_agent(llm, rt.TOOLS)
304
+
305
+ conv = Conversation(
306
+ agent=agent, workspace=str(ws),
307
+ persistence_dir=str(state / "conversations"),
308
+ conversation_id=cid, delete_on_close=False,
309
+ max_iteration_per_run=max_iterations,
310
+ visualizer=_visualizer(),
311
+ callbacks=[guard.seam_b])
312
+
313
+ # `Conversation(...)` is a strict factory and rejects this, but the
314
+ # LocalConversation it returns honours the attribute (`docs/0015` §4).
315
+ if max_budget_usd is not None:
316
+ try:
317
+ conv.max_budget_per_run = max_budget_usd
318
+ except Exception: # noqa: BLE001
319
+ print(" [agentctl] budget cap unsupported by this SDK build",
320
+ file=sys.stderr)
321
+
322
+ guard.attach(conv)
323
+
324
+ if verbose:
325
+ print(f" workspace {ws}")
326
+ print(f" model {model}"
327
+ + (f" (key from {key_env})" if key_env else ""))
328
+ print(f" conversation {cid}")
329
+ print(f" ledger {ledger_path}")
330
+ print(f" destructive {'CONFIRM' if confirm_destructive else 'ALLOWED'}")
331
+ print()
332
+
333
+ from agentctl.runtime import runs
334
+ run_id = runs.start(str(cid), ws, ledger_path, model, base_url,
335
+ task or None)
336
+
337
+ used_before = rp.Usage.of(conv)
338
+ stop = _PauseOnInterrupt(conv, verbose)
339
+ try:
340
+ if not resume:
341
+ conv.send_message(task)
342
+ elif (told := _tell_decisions(guard.store, str(cid))):
343
+ # What the user decided while the run was paused (`docs/0049`).
344
+ conv.send_message(told)
345
+ with stop:
346
+ conv.run()
347
+ except KeyboardInterrupt:
348
+ # The second Ctrl-C: stop now, not after the step. The ledger is
349
+ # fenced and fsynced, so whatever was in flight is reconciled on resume.
350
+ guard.close()
351
+ runs.end(run_id, "interrupted", detail="stopped with a second Ctrl-C")
352
+ raise SystemExit(f"\n stopped. Nothing is lost: agentctl resume "
353
+ f"{str(cid)[:8]}") from None
354
+ except Exception as e: # noqa: BLE001
355
+ # Give the lease back first: this run is over either way, and a held
356
+ # lease would make the very next `--resume` wait or ask.
357
+ guard.close()
358
+ # A provider refusing to serve is not a bug in the agent framework, and
359
+ # the SDK's own error ends with "please file a bug report at
360
+ # github.com/OpenHands" -- which sends you to the wrong place. Say what
361
+ # actually happened (`docs/0031`).
362
+ detail = _explain_provider_error(e)
363
+ if detail and _is_rate_limit(e):
364
+ runs.end(run_id, "rate_limited", detail=detail)
365
+ raise RateLimited(detail, str(cid), _reset_at(e)) from None
366
+ runs.end(run_id, "error", detail=detail or f"{type(e).__name__}: {e}")
367
+ if detail:
368
+ raise SystemExit(detail) from None
369
+ raise
370
+ finally:
371
+ # A cassette is only useful if it survives the run that produced it,
372
+ # including a run that died -- which is the interesting case here.
373
+ if recorder is not None:
374
+ from agentctl.adapters.litellm.recorder import detach
375
+ detach(recorder)
376
+
377
+ blocked = guard.blocked()
378
+ guard.close()
379
+
380
+ checked = None
381
+ if accept:
382
+ if verbose:
383
+ print(f"\n checking {accept}")
384
+ checked = rp.accept(accept, ws)
385
+
386
+ from agentctl.runtime.subagent import _final_text
387
+ said = _final_text(conv)
388
+ report = rp.Report(
389
+ outcome=("PASS" if checked["passed"] else "FAIL") if checked else "not checked",
390
+ accept=checked, changes=rp.changes(ws, start),
391
+ agent_said=None if said.startswith("[subagent") else said,
392
+ usage=rp.Usage.of(conv) - used_before, model=model,
393
+ seconds=time.time() - start.at, decisions=decisions,
394
+ blocked=[b.tool_call_id for b in blocked], ledger=str(ledger_path),
395
+ conversation_id=str(cid), workspace=str(ws),
396
+ awaiting=[(b.tool_call_id, (b.error or "").split("): ", 1)[-1])
397
+ for b in blocked if (b.error or "").startswith(AWAITING)],
398
+ paused=stop.paused)
399
+ runs.end(run_id,
400
+ "interrupted" if stop.paused else
401
+ ("done" if report.ok else
402
+ ("failed" if report.outcome == "FAIL" else "needs_you")),
403
+ outcome=report.outcome, requests=report.usage.requests)
404
+
405
+ out = {"conversation_id": str(cid), "workspace": str(ws),
406
+ "ledger": str(ledger_path), "decisions": decisions,
407
+ "blocked": [b.tool_call_id for b in blocked], "report": report}
408
+
409
+ if recorder is not None:
410
+ out["recorded"] = {"cassette": str(record),
411
+ "turns": len(recorder.cassette),
412
+ "errors": recorder.errors}
413
+ if replay_server is not None:
414
+ from agentctl.control.replay import summarise
415
+ out["replay"] = summarise(replay_server.cassette)
416
+ replay_server.stop()
417
+
418
+ return out
419
+
420
+
421
+ #: Marks a BLOCKED record as waiting on a human's approval, rather than on a
422
+ #: human's judgement of whether an ambiguous effect landed. Different questions,
423
+ #: different commands: `approve`/`deny` versus `resolve --landed/--retry`.
424
+ AWAITING = "AWAITING APPROVAL"
425
+
426
+
427
+ def _command_summary(call) -> str:
428
+ args = call.args or {}
429
+ text = args.get("command") or args.get("path") or str(args)
430
+ return f"{call.tool_name}: {str(text)[:200]}"
431
+
432
+
433
+ def _install_confirmation(guard, verbose: bool, workspace: Path | None = None,
434
+ interactive: bool | None = None) -> None:
435
+ """Ask before an effect that is dangerous, or that lands outside the workspace.
436
+
437
+ The gate is about replay safety, so a first-time `rm -rf` passes it. This is
438
+ the separate authorization question, asked here rather than buried in the
439
+ gate so the two stay distinguishable (`docs/0025` §4).
440
+
441
+ Two independent reasons to ask, and they are not the same question:
442
+
443
+ DESTRUCTIVE the effect is dangerous wherever it happens
444
+ escapes the root the effect is ordinary, but not the agent's
445
+ business -- `echo x > ~/.bashrc` classifies
446
+ IDEMPOTENT_WRITE and is correct to (`docs/0027`)
447
+
448
+ Escalating the *class* for the second case would have been the easy fix and
449
+ the wrong one: the ledger would then treat a replayable write as
450
+ unrecoverable, and a safe resume would start failing closed.
451
+
452
+ **With nobody at a terminal** (`docs/0042` I-06, `docs/0044` F7) it used to
453
+ call `input()`: unattended it hung, detached it read EOF and refused,
454
+ silently. Now the action is QUEUED -- recorded BLOCKED awaiting approval,
455
+ and the agent is told so and not to repeat it -- and `agentctl approve` /
456
+ `deny` answers it later. A decision already recorded is honoured without
457
+ asking again, which is what lets an approved action run on resume.
458
+ """
459
+ from agentctl.kernel.ledger.models import EffectClass, GateDecision, Verdict
460
+ from agentctl.kernel.paths import environment_installs, escaping_writes
461
+
462
+ inner = guard.gate.guard
463
+ store, gate = guard.gate.store, guard.gate
464
+ ask = sys.stdin.isatty() if interactive is None else interactive
465
+
466
+ def _not_run(call, why: str) -> None:
467
+ # A refused action provably did not run. Leaving its INTENT record
468
+ # behind made a later identical call look like a crash to reconcile.
469
+ try:
470
+ if (rec := store.lookup(call.tool_call_id)) is not None and rec.state.value == "INTENT":
471
+ store.fail(call.tool_call_id, why)
472
+ except Exception: # noqa: BLE001
473
+ pass
474
+
475
+ def _queue(call, cls, why: str) -> GateDecision:
476
+ try:
477
+ if store.lookup(call.tool_call_id) is None:
478
+ store.write_intent(call, cls, gate.fence)
479
+ store.block(call.tool_call_id,
480
+ f"{AWAITING} ({why}): {_command_summary(call)}")
481
+ except Exception: # noqa: BLE001
482
+ pass
483
+ if verbose:
484
+ print(f" [agentctl] queued for your approval: {_command_summary(call)}",
485
+ file=sys.stderr)
486
+ return GateDecision(
487
+ Verdict.BLOCK, cls,
488
+ reason=(f"This action needs the user's approval ({why}) and has been "
489
+ f"queued for them; it did not run. Do NOT repeat it. Carry "
490
+ f"on with anything that does not depend on it, then finish "
491
+ f"and say it is waiting for approval."))
492
+
493
+ def guard_with_confirmation(call):
494
+ decision = inner(call)
495
+ if decision.verdict is not Verdict.EXECUTE:
496
+ return decision
497
+
498
+ reasons: list[str] = []
499
+ if decision.effect_class is EffectClass.DESTRUCTIVE:
500
+ reasons.append("DESTRUCTIVE")
501
+ if outside := escaping_writes(call, workspace):
502
+ reasons.append("writes outside the workspace: "
503
+ + ", ".join(outside[:4]))
504
+ # Software installed into YOUR environment (`docs/0052`). Not asked
505
+ # inside the image: there it lands in the container and dies with it,
506
+ # which is the containment the image exists to provide.
507
+ if not os.environ.get("HANDCODE_CONTAINER") and \
508
+ (installs := environment_installs(call, workspace)):
509
+ reasons.append("installs into your environment: " + installs[0][:80])
510
+ if not reasons:
511
+ return decision
512
+ why = " | ".join(reasons)
513
+
514
+ prior = store.approval(call.conversation_id, call.intent_hash())
515
+ if prior == "approve":
516
+ store.use_approval(call.conversation_id, call.intent_hash())
517
+ if verbose:
518
+ print(f" [agentctl] approved earlier by you: "
519
+ f"{_command_summary(call)}", file=sys.stderr)
520
+ return decision
521
+ if prior == "deny":
522
+ _not_run(call, "denied by the operator")
523
+ return GateDecision(Verdict.BLOCK, decision.effect_class,
524
+ reason="The user denied this action. Do not run "
525
+ "it, and do not try to achieve the same "
526
+ "effect another way.")
527
+ if not ask:
528
+ return _queue(call, decision.effect_class, why)
529
+
530
+ print(f"\n !! {why}", file=sys.stderr)
531
+ print(f" {call.tool_name} {str(call.args)[:200]}", file=sys.stderr)
532
+ try:
533
+ answer = input(" allow? [y/N] ").strip().lower()
534
+ except EOFError:
535
+ # A terminal that closed mid-question: queue it, as for no terminal.
536
+ return _queue(call, decision.effect_class, why)
537
+ if answer != "y":
538
+ _not_run(call, "refused by the operator")
539
+ return GateDecision(Verdict.BLOCK, decision.effect_class,
540
+ reason="refused by the operator")
541
+ return decision
542
+
543
+ guard.gate.guard = guard_with_confirmation
544
+
545
+
546
+ class RateLimited(SystemExit):
547
+ """The provider refused for quota. Carries what a wait-and-resume needs:
548
+ the conversation, and the reset time when the provider said one."""
549
+
550
+ def __init__(self, text: str, conversation_id: str, reset_at: float | None):
551
+ super().__init__(text)
552
+ self.text, self.conversation_id, self.reset_at = text, conversation_id, reset_at
553
+
554
+
555
+ def _is_rate_limit(e: Exception) -> bool:
556
+ low = str(e).lower()
557
+ return ("rate limit" in low or "429" in str(e) or "ratelimiterror" in low
558
+ or "quota" in low)
559
+
560
+
561
+ def _reset_at(e: Exception) -> float | None:
562
+ """The provider's reset time, epoch seconds, when its error carried one."""
563
+ import re
564
+ if m := re.search(r'"X-RateLimit-Reset":"(\d{10,13})"', str(e)):
565
+ ts = int(m.group(1))
566
+ return ts / 1000 if ts > 10_000_000_000 else float(ts)
567
+ return None
568
+
569
+
570
+ class _PauseOnInterrupt:
571
+ """Ctrl-C pauses the conversation after its current step; a second one
572
+ stops at once (`docs/0043` Phase 4).
573
+
574
+ `pause()` takes effect between steps, and is callable from a signal
575
+ handler: the conversation's lock is re-entrant (`docs/0042` §8.2). The
576
+ paused conversation is persisted, so `agentctl resume` continues it.
577
+ Only installed on the main thread -- signals cannot be anywhere else.
578
+ """
579
+
580
+ def __init__(self, conv, verbose: bool):
581
+ self.conv, self.verbose, self.paused, self._old = conv, verbose, False, {}
582
+
583
+ def _handle(self, signum, frame):
584
+ if self.paused:
585
+ raise KeyboardInterrupt
586
+ self.paused = True
587
+ if self.verbose:
588
+ print("\n [agentctl] pausing after the current step "
589
+ "(Ctrl-C again to stop now)", file=sys.stderr)
590
+ try:
591
+ self.conv.pause()
592
+ except Exception: # noqa: BLE001
593
+ raise KeyboardInterrupt from None
594
+
595
+ def __enter__(self):
596
+ import signal
597
+ import threading
598
+ if threading.current_thread() is not threading.main_thread():
599
+ return self
600
+ for name in ("SIGINT", "SIGBREAK"):
601
+ if (sig := getattr(signal, name, None)) is not None:
602
+ self._old[sig] = signal.signal(sig, self._handle)
603
+ return self
604
+
605
+ def __exit__(self, *exc):
606
+ import signal
607
+ for sig, old in self._old.items():
608
+ signal.signal(sig, old)
609
+ return False
610
+
611
+
612
+ def _tell_decisions(store, conversation_id: str) -> str | None:
613
+ """The message a resumed agent gets about approvals decided meanwhile."""
614
+ try:
615
+ rows = store.untold(conversation_id)
616
+ except Exception: # noqa: BLE001
617
+ return None
618
+ if not rows:
619
+ return None
620
+ lines = ["While this run was paused, the user decided on actions that "
621
+ "needed approval:"]
622
+ for r in rows:
623
+ what = r.get("summary") or "an action"
624
+ if r["decision"] == "approve":
625
+ lines.append(f"- APPROVED: {what}. You may run it now if it is "
626
+ f"still needed.")
627
+ else:
628
+ lines.append(f"- DENIED: {what}. Do not run it, and do not try to "
629
+ f"achieve the same effect another way.")
630
+ lines.append("Continue the task.")
631
+ store.told(conversation_id)
632
+ return "\n".join(lines)
633
+
634
+
635
+ # ── policy enforcement (M7, docs/0030) ─────────────────────────────────
636
+ def _load_policy(policy):
637
+ """Load a compiled policy, or compile a .yaml on the spot.
638
+
639
+ Accepting the source form is a convenience with a sharp edge: compiling
640
+ here means a malformed policy is discovered at run time, which is exactly
641
+ what `docs/0012` §5.2 wants to avoid. So it is compiled BEFORE anything
642
+ else happens, and a failure stops the run before a single effect.
643
+ """
644
+ from pathlib import Path as _P
645
+
646
+ from agentctl.kernel.policy import Policy
647
+
648
+ if policy is None:
649
+ return Policy.empty()
650
+ p = _P(policy)
651
+ if p.suffix in (".yaml", ".yml"):
652
+ from agentctl.control.policy import PolicyError, compile_policy
653
+ try:
654
+ return Policy(compile_policy(p))
655
+ except PolicyError as e:
656
+ raise SystemExit(f"policy {p} does not compile:\n{e}")
657
+ return Policy.load(p)
658
+
659
+
660
+ def _daily_spend(cost_ledger) -> tuple[float, float] | None:
661
+ """(spent, coverage) for the last 24h, or None if nothing is recorded.
662
+
663
+ Read HERE, in the runtime, and handed to the kernel as a number. The kernel
664
+ may not import the cost ledger (`docs/0008` R2) and should not: a budget
665
+ guard that queried a database in-band would fail closed whenever the
666
+ control plane was down, turning a cost feature into an outage.
667
+ """
668
+ from pathlib import Path as _P
669
+
670
+ p = _P(cost_ledger) if cost_ledger else _P("cost.db")
671
+ if not p.exists():
672
+ return None
673
+ try:
674
+ from agentctl.control.cost import CostLedger
675
+ with CostLedger(p) as c:
676
+ t = c.totals(since=time.time() - 86400)
677
+ return (t.cost_usd, t.coverage) if t.calls else None
678
+ except Exception: # noqa: BLE001
679
+ return None
680
+
681
+
682
+ def _enforce_before_spending(pol, model, cost_ledger, max_budget_usd,
683
+ replaying: bool, verbose: bool) -> None:
684
+ """Refuse the run outright if policy already says no.
685
+
686
+ Before the ledger, before the agent, before a single token: a budget check
687
+ that happens after the work is an audit, not a cap.
688
+ """
689
+ if verbose:
690
+ print(f" policy {len(pol.pool(pol.default_pool or ''))} "
691
+ f"deployment(s) in {pol.default_pool!r}"
692
+ f" per-task ${pol.limit('per_task') or 0:.2f}")
693
+
694
+ # Replay spends nothing, so a budget cannot bind and a stale ledger must
695
+ # not stop an offline run.
696
+ if replaying:
697
+ return
698
+
699
+ spend = _daily_spend(cost_ledger)
700
+ if spend is not None:
701
+ spent, coverage = spend
702
+ v = pol.check_budget(spent, "daily", coverage)
703
+ if not v.allowed:
704
+ raise SystemExit(
705
+ f"refusing to start: {v.reason}\n"
706
+ f" raise budget.daily_usd, or wait for the window to roll.")
707
+ if not v.trustworthy and verbose:
708
+ print(f" ! budget {v.describe()}", file=sys.stderr)
709
+
710
+ # Escalation: using something outside the default pool is spending money
711
+ # the policy did not pre-authorise (`docs/0002` §5).
712
+ default = pol.default_pool
713
+ if default and model not in pol.pool(default):
714
+ esc = pol.escalation or {}
715
+ target = esc.get("pool")
716
+ in_escalation_pool = target and model in pol.pool(target)
717
+ if esc.get("require_confirmation", True):
718
+ where = f"the {target!r} pool" if in_escalation_pool else "no pool"
719
+ print(f"\n !! {model} is not in the default pool {default!r} "
720
+ f"({where})", file=sys.stderr)
721
+ try:
722
+ answer = input(" spend on it? [y/N] ").strip().lower()
723
+ except EOFError:
724
+ answer = "n"
725
+ if answer != "y":
726
+ raise SystemExit("refused: policy requires confirmation "
727
+ "before leaving the default pool")
728
+
729
+
730
+ def _explain_provider_error(exc: Exception) -> str | None:
731
+ """Turn a provider refusal into something actionable, or None.
732
+
733
+ Returns None for anything not recognised -- guessing at an unfamiliar
734
+ error would hide it, and an unhandled traceback is better than a confident
735
+ wrong explanation.
736
+ """
737
+ import re
738
+
739
+ text = str(exc)
740
+ low = text.lower()
741
+
742
+ if "rate limit" in low or "429" in text or "ratelimiterror" in low:
743
+ when = ""
744
+ if m := re.search(r'"X-RateLimit-Reset":"(\d{10,13})"', text):
745
+ from datetime import datetime
746
+ ts = int(m.group(1))
747
+ ts = ts / 1000 if ts > 10_000_000_000 else ts
748
+ r = datetime.fromtimestamp(ts).astimezone()
749
+ when = f"\n resets {r:%Y-%m-%d %H:%M} local"
750
+ remaining = ""
751
+ if m := re.search(r'"X-RateLimit-Remaining":"(\d+)"', text):
752
+ remaining = f"\n remaining {m.group(1)}"
753
+ per_day = ("\n note the free-model cap is account-wide across "
754
+ "every `:free`\n model, so switching model "
755
+ "does not help"
756
+ if "free-models-per-day" in low else "")
757
+ return (f"the provider is rate limiting you. This is not an agent "
758
+ f"error.{remaining}{when}{per_day}\n"
759
+ f" options wait for the reset (run with --wait 30m to "
760
+ f"wait automatically),\n or route through the "
761
+ f"pool so another provider serves: --pool")
762
+
763
+ if "insufficient" in low and "credit" in low:
764
+ return ("the provider says the account is out of credit. Nothing was "
765
+ "spent on this run.")
766
+
767
+ if "no auth credentials" in low or "invalid api key" in low or "401" in text:
768
+ return ("the provider rejected the API key. Check the environment "
769
+ "variable for your model's provider; the value is read from "
770
+ "the environment and never logged.")
771
+
772
+ if ("400 bad request" in low or "badrequesterror" in low
773
+ or "not a valid model" in low):
774
+ return ("the provider rejected the request — usually an unknown or "
775
+ "unavailable model id.\n"
776
+ " check the id at openrouter.ai/models; it needs the "
777
+ "provider prefix,\n"
778
+ " e.g. openrouter/vendor/model:free\n"
779
+ " known-good agentctl proxy (lists the ids it builds a "
780
+ "pool from)")
781
+
782
+ if "overloaded" in low or "503" in text or "502" in text:
783
+ return ("the provider is overloaded and refused the request. Free-tier "
784
+ "endpoints do this under load.\n"
785
+ " options agentctl resume, or route through the pool so a "
786
+ "failure fails over: --pool")
787
+ return None