handcode 0.3.0rc1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (67) hide show
  1. agentctl/__init__.py +0 -0
  2. agentctl/adapters/__init__.py +0 -0
  3. agentctl/adapters/litellm/__init__.py +9 -0
  4. agentctl/adapters/litellm/hook.py +49 -0
  5. agentctl/adapters/litellm/recorder.py +187 -0
  6. agentctl/adapters/openhands/__init__.py +169 -0
  7. agentctl/adapters/openhands/handoff.py +155 -0
  8. agentctl/adapters/openhands/seam_b.py +259 -0
  9. agentctl/adapters/openhands/seam_c.py +209 -0
  10. agentctl/cli.py +1450 -0
  11. agentctl/control/__init__.py +0 -0
  12. agentctl/control/cost/__init__.py +4 -0
  13. agentctl/control/cost/ledger.py +210 -0
  14. agentctl/control/dash.py +697 -0
  15. agentctl/control/keys.py +440 -0
  16. agentctl/control/matrix/__init__.py +0 -0
  17. agentctl/control/matrix/data/tools.yaml +149 -0
  18. agentctl/control/policy/__init__.py +10 -0
  19. agentctl/control/policy/compile.py +258 -0
  20. agentctl/control/policy/data/policy.compiled.json +38 -0
  21. agentctl/control/policy/data/policy.yaml +46 -0
  22. agentctl/control/probe.py +399 -0
  23. agentctl/control/providers.py +293 -0
  24. agentctl/control/proxy.py +536 -0
  25. agentctl/control/proxyenv.py +309 -0
  26. agentctl/control/replay/__init__.py +14 -0
  27. agentctl/control/replay/cassette.py +281 -0
  28. agentctl/control/replay/server.py +109 -0
  29. agentctl/demo/__init__.py +214 -0
  30. agentctl/demo/child.py +84 -0
  31. agentctl/demo/mock.py +79 -0
  32. agentctl/demo/tool.py +62 -0
  33. agentctl/gha.py +488 -0
  34. agentctl/kernel/__init__.py +0 -0
  35. agentctl/kernel/classify.py +170 -0
  36. agentctl/kernel/gate.py +391 -0
  37. agentctl/kernel/hook.py +229 -0
  38. agentctl/kernel/ledger/__init__.py +0 -0
  39. agentctl/kernel/ledger/models.py +160 -0
  40. agentctl/kernel/ledger/schema.sql +62 -0
  41. agentctl/kernel/ledger/store.py +596 -0
  42. agentctl/kernel/paths.py +203 -0
  43. agentctl/kernel/policy.py +160 -0
  44. agentctl/kernel/reconcile/__init__.py +31 -0
  45. agentctl/kernel/reconcile/base.py +106 -0
  46. agentctl/kernel/reconcile/external.py +137 -0
  47. agentctl/kernel/reconcile/filesystem.py +162 -0
  48. agentctl/kernel/reconcile/git.py +162 -0
  49. agentctl/runtime/__init__.py +20 -0
  50. agentctl/runtime/citations.py +179 -0
  51. agentctl/runtime/config.py +97 -0
  52. agentctl/runtime/doctor.py +335 -0
  53. agentctl/runtime/init.py +148 -0
  54. agentctl/runtime/lease.py +143 -0
  55. agentctl/runtime/orchestrate.py +187 -0
  56. agentctl/runtime/plugins.py +130 -0
  57. agentctl/runtime/report.py +361 -0
  58. agentctl/runtime/runner.py +787 -0
  59. agentctl/runtime/runs.py +191 -0
  60. agentctl/runtime/subagent.py +274 -0
  61. agentctl/runtime/tools.py +350 -0
  62. handcode-0.3.0rc1.dist-info/METADATA +659 -0
  63. handcode-0.3.0rc1.dist-info/RECORD +67 -0
  64. handcode-0.3.0rc1.dist-info/WHEEL +5 -0
  65. handcode-0.3.0rc1.dist-info/entry_points.txt +3 -0
  66. handcode-0.3.0rc1.dist-info/licenses/LICENSE +21 -0
  67. handcode-0.3.0rc1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,335 @@
1
+ r"""Preflight: what is ready, what is missing, what will stop you.
2
+
3
+ Built because a run burned fifteen tool calls and produced nothing, and the
4
+ only signal was a `RateLimitError` traceback thirty lines deep ending in
5
+ *"please file a bug report at github.com/OpenHands"* — which is not where the
6
+ problem was.
7
+
8
+ Everything here is free and offline except the provider probe, which reads your
9
+ own account status and costs no model request.
10
+
11
+ ## What this check fetches, and what it doesn't
12
+
13
+ This module's own OpenRouter check calls `/api/v1/auth/key`, which reports
14
+ dollar usage and free-tier status but not the remaining daily count. That was
15
+ recorded here as unknowable by any free endpoint -- checked again on
16
+ 2026-09-21 and found false. The sibling endpoint `/api/v1/key` exposes it
17
+ directly, at the same cost (zero) and with the same auth
18
+ (`probe.py::openrouter_quota`, `free_model_daily_requests:
19
+ {used, limit, remaining}`, confirmed live across six accounts). This check
20
+ still does not call it -- that number belongs in `agentctl dash`, not in a
21
+ preflight -- but it no longer claims the number cannot be known.
22
+ """
23
+ from __future__ import annotations
24
+
25
+ import json
26
+ import os
27
+ import shutil
28
+ import subprocess
29
+ import sys
30
+ from pathlib import Path
31
+
32
+ # Derived, never duplicated. `doctor` and `proxy` each kept their own provider
33
+ # list until one registry replaced both -- two lists drift, and drift here
34
+ # means a key you added is checked by one command and ignored by another. The
35
+ # same defect as `docs/0029` §4.
36
+ def _provider_rows() -> list[tuple[str, str, str]]:
37
+ from agentctl.control.providers import PROVIDERS as _P
38
+ return [(p.key, p.name,
39
+ "free tier available" if p.free_tier else "paid") for p in _P]
40
+
41
+
42
+ PROVIDERS = _provider_rows()
43
+
44
+ OK, WARN, BAD = "ok", "!!", "XX"
45
+
46
+
47
+ # ── Groq's turn-2 ceiling (`docs/0038` §4.1, `research/phase-10-2` V3-V5, I1) ──
48
+ # Three separately-dated facts. The conclusion below is COMPUTED from them,
49
+ # never written down as a bare number, so updating one fact updates the
50
+ # conclusion instead of leaving a stale headline behind.
51
+ #
52
+ # VERIFIED, in this repo, reproducible offline with no network and no key --
53
+ # `tests/test_doctor_groq_prefix.py` re-measures both live and fails if
54
+ # either has drifted from the constant pinned here:
55
+ _FIXED_PREFIX_TOKENS = 3_593 # 3,208 system prompt + 385 tool schemas
56
+ _PREFIX_MEASURED = "2026-09-20, openhands-sdk 1.45.0"
57
+ #
58
+ # _FIXED_PREFIX_TOKENS is environment-dependent by about two tokens: 3,593
59
+ # measured on Windows, 3,591 on Linux CI, same commit and same encoding. The
60
+ # cause is not identified -- neither the system prompt nor the tool schemas
61
+ # contain an OS string or a path. It does not move the conclusion (311 tokens
62
+ # of headroom versus 313, both far below one turn), and
63
+ # tests/test_doctor_groq_prefix.py asserts the conclusion rather than the
64
+ # constant for exactly that reason.
65
+ #
66
+ _RUNNER_MAX_OUTPUT_TOKENS = 4_096 # mirrors runtime/runner.py:221 -- not
67
+ # importable, no constant exists there;
68
+ # the same test pins this one too.
69
+ #
70
+ # INFERRED, external to this repo, and NOT re-checked by the test suite --
71
+ # dated so it is visibly different in kind from the two facts above. Source
72
+ # tier T3 (issue trackers + one article): Groq's own rate-limits page frames
73
+ # TPM as a budget but does not state the prompt+max_tokens mechanism in so
74
+ # many words. This can go stale or turn out wrong without this repo noticing.
75
+ _GROQ_TPM_CEILING = 8_000
76
+ _GROQ_TPM_SOURCE = "console.groq.com/docs/rate-limits, checked 2026-09-20"
77
+
78
+
79
+ def _groq_headroom(accts: list) -> tuple[str, str, str] | None:
80
+ """How much of Groq's TPM ceiling is left for conversation, and whether
81
+ Groq is all the user has.
82
+
83
+ Returns None when no Groq account is configured — nothing to warn about.
84
+ Status is WARN, never BAD: turn 1 does work, and the failure mode is
85
+ INFERRED, not observed from this repo (`docs/0038` §4.1, confirmed by an
86
+ independent re-check at §9.3, not yet confirmed by an actual `agentctl
87
+ run` against Groq). BAD means "this will not work"; this means "it likely
88
+ won't work past turn 1," a different claim.
89
+ """
90
+ groq = [a for a in accts if a.provider.name == "groq"]
91
+ if not groq:
92
+ return None
93
+
94
+ n_groq = len(groq) * len(groq[0].provider.models)
95
+ # Deployments in the POOL, not credentials held. Counting every
96
+ # account folds in providers that cannot serve -- six Cerebras
97
+ # keys that 402 on every completion, one paid Anthropic key that
98
+ # is never called -- and hides them inside the reassuring
99
+ # remainder ("the other N carry no such limit"). That is the
100
+ # accounts-held-vs-deployments-served conflation `docs/0038` 9.4
101
+ # was written to diagnose, reappearing one file over (`docs/0039`).
102
+ n_total = sum(len(a.provider.models) for a in accts
103
+ if a.provider.free_tier)
104
+ headroom = _GROQ_TPM_CEILING - _RUNNER_MAX_OUTPUT_TOKENS - _FIXED_PREFIX_TOKENS
105
+
106
+ only = (" Groq is the only provider you have configured — there is no "
107
+ "other deployment in your pool that carries a real conversation."
108
+ if len(groq) == len(accts) else "")
109
+
110
+ detail = (
111
+ f"{n_groq} of {n_total} configured deployment(s) are Groq.{only} "
112
+ f"Groq's published free-tier limit ({_GROQ_TPM_CEILING:,} tok/min, "
113
+ f"metered on prompt+max_tokens — {_GROQ_TPM_SOURCE}, INFERRED "
114
+ f"mechanism, not this repo's own test) leaves ~{headroom} tokens of "
115
+ f"conversation after this repo's measured {_FIXED_PREFIX_TOKENS:,}-"
116
+ f"tok system prompt + tool schemas ({_PREFIX_MEASURED}) and "
117
+ f"runner.py's max_output_tokens={_RUNNER_MAX_OUTPUT_TOKENS} — turn 2 "
118
+ f"of most real tasks will 413. The other {n_total - n_groq} "
119
+ f"deployment(s) carry no documented version of this limit, which is "
120
+ f"not the same as a clean bill of health — providers.py deliberately "
121
+ f"records no rate limits for anyone. Unconfirmed by a live call from "
122
+ f"this repo — falsify: agentctl run --model "
123
+ f"groq/openai/gpt-oss-120b <a trivial task>, watch turn 2."
124
+ )
125
+ return (WARN, "groq headroom", detail)
126
+
127
+
128
+ def _model() -> list[tuple[str, str, str]]:
129
+ """Which model `agentctl run` will use with no flags, and why that one."""
130
+ try:
131
+ from agentctl.runtime.config import resolve
132
+ s = resolve("model", None)
133
+ except SystemExit as e: # a broken config.toml
134
+ return [(BAD, "model", str(e).splitlines()[0])]
135
+ except Exception as e: # noqa: BLE001
136
+ return [(WARN, "model", f"could not resolve: {e}")]
137
+ if s.value is None:
138
+ return [(WARN, "model", "none -- run `agentctl init`")]
139
+ return [(OK, "model", f"{s.value} ({s.source})")]
140
+
141
+
142
+ def check_all(workspace: str | Path | None = None,
143
+ probe_network: bool = True) -> list[tuple[str, str, str]]:
144
+ """[(status, subject, detail)] — never raises, always reports."""
145
+ out: list[tuple[str, str, str]] = []
146
+ out += _packages()
147
+ out += _providers(probe_network)
148
+ out += _model()
149
+ out += _tooling()
150
+ out += _policy()
151
+ if workspace:
152
+ out += _workspace(Path(workspace))
153
+ return out
154
+
155
+
156
+ # ── checks ─────────────────────────────────────────────────────────────
157
+ def _packages() -> list[tuple[str, str, str]]:
158
+ from importlib.metadata import version
159
+
160
+ # Versions are reported, never judged. `mcp >= 2` used to be BAD here
161
+ # (docs/0021 §7: litellm[proxy] once downgraded it and broke the SDK), and
162
+ # it outlived its premise: SDK 1.50.1 resolves to mcp 1.30 and imports
163
+ # cleanly, so a fresh install from the README was told to stop by a rule
164
+ # the next two rows contradicted (docs/0044 N5). The import rows below
165
+ # test what actually breaks, on whatever versions pip chose.
166
+ rows = []
167
+ for pkg in ("openhands-sdk", "litellm", "mcp", "fastmcp", "pyyaml"):
168
+ try:
169
+ rows.append((OK, pkg, version(pkg)))
170
+ except Exception: # noqa: BLE001
171
+ rows.append((BAD, pkg, "not installed"))
172
+
173
+ try:
174
+ from fastmcp import Client # noqa: F401
175
+ rows.append((OK, "fastmcp client", "importable"))
176
+ except Exception as e: # noqa: BLE001
177
+ # `import fastmcp` alone succeeds even when this is broken (docs/0028).
178
+ rows.append((BAD, "fastmcp client", f"{type(e).__name__}: {e}"))
179
+
180
+ try:
181
+ from openhands.sdk.event.base import Event # noqa: F401
182
+ rows.append((OK, "openhands sdk", "imports cleanly"))
183
+ except Exception as e: # noqa: BLE001
184
+ rows.append((BAD, "openhands sdk", f"{type(e).__name__}: {e}"))
185
+ return rows
186
+
187
+
188
+ def _keys_file() -> list[tuple[str, str, str]]:
189
+ """Is the keys file exposed to git? The guard nobody thinks to run.
190
+
191
+ `keys.py` documents this as one of three guards, and it was not actually
192
+ wired anywhere until the docstring was checked against the code.
193
+ """
194
+ from agentctl.control.keys import check_not_tracked, resolve
195
+
196
+ p = resolve()
197
+ if p is None:
198
+ return [(WARN, "keys file",
199
+ "none found — run: agentctl init")]
200
+ if (warn := check_not_tracked(p)):
201
+ return [(BAD, "keys file", warn)]
202
+ return [(OK, "keys file", f"{p} (not exposed to git)")]
203
+
204
+
205
+ def _providers(probe: bool) -> list[tuple[str, str, str]]:
206
+ from agentctl.control.providers import all_accounts
207
+
208
+ rows = _keys_file()
209
+ accts = all_accounts()
210
+ for a in accts:
211
+ rows.append((OK, a.label,
212
+ f"{a.env} set "
213
+ f"({'free tier' if a.provider.free_tier else 'paid'})"))
214
+
215
+ if not accts:
216
+ rows.append((BAD, "providers", "no API key in the environment"))
217
+ return rows
218
+
219
+ providers = {a.provider.name for a in accts}
220
+ if len(accts) == 1:
221
+ # One account cannot fail over. The advice is a second PROVIDER: a
222
+ # second key at the same one is unverified as a second quota and
223
+ # unchecked against terms of service (docs/0042 §4.C, 0044 N9).
224
+ rows.append((WARN, "failover",
225
+ f"only {accts[0].label} — fine to start with; a daily "
226
+ f"cap on it stops work until it resets. A key at a "
227
+ f"second provider survives that (`agentctl keys`)."))
228
+ elif len(providers) == 1:
229
+ rows.append((WARN, "failover",
230
+ f"{len(accts)} accounts, all at "
231
+ f"{next(iter(providers))} — survives a cap, not the "
232
+ f"provider going down."))
233
+
234
+ # Offline, no network — runs regardless of `probe`.
235
+ if (row := _groq_headroom(accts)) is not None:
236
+ rows.append(row)
237
+
238
+ if probe and os.environ.get("OPENROUTER_API_KEY"):
239
+ rows.append(_openrouter())
240
+ return rows
241
+
242
+
243
+ def _openrouter() -> tuple[str, str, str]:
244
+ """Read our own account. No model request, so it costs nothing."""
245
+ import urllib.request
246
+
247
+ key = os.environ["OPENROUTER_API_KEY"]
248
+ try:
249
+ req = urllib.request.Request(
250
+ "https://openrouter.ai/api/v1/auth/key",
251
+ headers={"Authorization": f"Bearer {key}"})
252
+ with urllib.request.urlopen(req, timeout=15) as fh:
253
+ d = json.loads(fh.read())["data"]
254
+ except Exception as e: # noqa: BLE001
255
+ return (WARN, "openrouter", f"could not reach: {type(e).__name__}")
256
+
257
+ if d.get("is_free_tier"):
258
+ return (WARN, "openrouter quota",
259
+ "FREE TIER: 50 model requests/day, account-wide across every "
260
+ "`:free` model. This check reports free-tier status only; "
261
+ "the live remaining count is exposed separately by "
262
+ "`probe.py::openrouter_quota` (GET /api/v1/key) for callers "
263
+ "that want it.")
264
+ return (OK, "openrouter quota",
265
+ f"paid key, ${float(d.get('usage') or 0):.4f} used")
266
+
267
+
268
+ def _tooling() -> list[tuple[str, str, str]]:
269
+ rows = []
270
+ git = shutil.which("git")
271
+ rows.append((OK, "git", git) if git else
272
+ (BAD, "git", "not on PATH — the git probe and chaos suite need it"))
273
+ rows.append((OK, "python", sys.version.split()[0]))
274
+ if sys.version_info < (3, 12):
275
+ rows[-1] = (BAD, "python", f"{sys.version.split()[0]} — needs >= 3.12")
276
+ return rows
277
+
278
+
279
+ def _policy() -> list[tuple[str, str, str]]:
280
+ from agentctl.kernel.policy import DEFAULT_POLICY, Policy
281
+
282
+ if not DEFAULT_POLICY.exists():
283
+ return [(WARN, "policy", "not compiled — run: agentctl policy "
284
+ "agentctl/control/policy/data/policy.yaml")]
285
+ try:
286
+ p = Policy.load()
287
+ except Exception as e: # noqa: BLE001
288
+ return [(BAD, "policy", f"compiled artifact unreadable: {e}")]
289
+ caps = ", ".join(f"{s} ${p.limit(s):.2f}" for s in ("per_task", "daily")
290
+ if p.limit(s) is not None)
291
+ return [(OK, "policy", f"{caps or 'no caps'} | "
292
+ f"destructive: {p.effect_rule('DESTRUCTIVE') or '-'}")]
293
+
294
+
295
+ def _workspace(ws: Path) -> list[tuple[str, str, str]]:
296
+ rows = []
297
+ if not ws.exists():
298
+ return [(WARN, "workspace", f"{ws} does not exist yet")]
299
+ try:
300
+ probe = ws / ".agentctl_write_probe"
301
+ probe.write_text("x", encoding="utf-8")
302
+ probe.unlink()
303
+ rows.append((OK, "workspace", f"{ws} (writable)"))
304
+ except Exception as e: # noqa: BLE001
305
+ rows.append((BAD, "workspace", f"{ws} not writable: {e}"))
306
+
307
+ if (ws / ".git").exists():
308
+ dirty = subprocess.run(["git", "status", "--porcelain"], cwd=str(ws),
309
+ capture_output=True, text=True).stdout.strip()
310
+ rows.append((WARN, "uncommitted", f"{len(dirty.splitlines())} changed "
311
+ f"file(s) — the agent edits real files; commit first")
312
+ if dirty else (OK, "git tree", "clean"))
313
+ else:
314
+ rows.append((WARN, "git", f"{ws} is not a git repo — no way to undo "
315
+ f"what the agent writes"))
316
+
317
+ led = ws / ".agentctl" / "ledger.db"
318
+ if led.exists():
319
+ rows.append((OK, "ledger", f"{led} (agentctl --ledger <path> status)"))
320
+ return rows
321
+
322
+
323
+ def report(rows: list[tuple[str, str, str]]) -> int:
324
+ """Print, and return an exit code. Any BAD is a failure."""
325
+ width = max((len(s) for _, s, _ in rows), default=10)
326
+ for status, subject, detail in rows:
327
+ print(f" {status:<4} {subject:<{width}} {detail}")
328
+ bad = [r for r in rows if r[0] == BAD]
329
+ warn = [r for r in rows if r[0] == WARN]
330
+ print()
331
+ if bad:
332
+ print(f" {len(bad)} blocking problem(s). Fix these before running.")
333
+ return 1
334
+ print(f" ready{f' — {len(warn)} warning(s) worth reading' if warn else ''}")
335
+ return 0
@@ -0,0 +1,148 @@
1
+ """`agentctl init`: from one key to a working default, in one command.
2
+
3
+ `docs/0043` Phase 2. Phase 0 (`docs/0044`) took a new user nine commands and
4
+ two false "it will not work" verdicts to reach a first task. This is the
5
+ middle of the three commands the plan promises:
6
+
7
+ pip install ... (once)
8
+ agentctl init this
9
+ agentctl run "<task>" in any git repository
10
+
11
+ What it does, in order, and nothing else:
12
+
13
+ 1. **Find a key.** One already in the environment or a keys file is used. With
14
+ none, it asks which provider and reads the key without echoing it, into
15
+ `~/.agentctl/keys.env` -- outside every repository, ignored first.
16
+ 2. **Prove it.** One completion, through the same check as `keys --check`,
17
+ which tries each registry model past a dead one (`docs/0044` N6). The
18
+ model that ANSWERED is the one recorded: a default chosen by a completion,
19
+ not by a list.
20
+ 3. **Record it** in `~/.agentctl/config.toml`, which `run` reads.
21
+
22
+ A paid provider is not called unless asked (`docs/0002` §5: never silently
23
+ spend), and then it costs one four-token completion.
24
+ """
25
+ from __future__ import annotations
26
+
27
+ import getpass
28
+ import os
29
+ import sys
30
+
31
+ from agentctl.runtime import config
32
+
33
+
34
+ def _ask(prompt: str) -> str:
35
+ try:
36
+ return input(prompt).strip()
37
+ except EOFError:
38
+ return ""
39
+
40
+
41
+ def _choose(interactive: bool):
42
+ """A provider the user picks, free tiers first. None if they do not."""
43
+ from agentctl.control.providers import PROVIDERS
44
+
45
+ order = sorted(PROVIDERS, key=lambda p: (not p.free_tier, PROVIDERS.index(p)))
46
+ print(" No provider key found. Pick one to start with -- one is enough:\n")
47
+ for i, p in enumerate(order, 1):
48
+ print(f" {i}. {p.name:<11} {'free tier' if p.free_tier else 'PAID'}"
49
+ f" {p.console}")
50
+ print()
51
+ if not interactive:
52
+ return None
53
+ answer = _ask(f" provider [1-{len(order)}, default 1]: ") or "1"
54
+ try:
55
+ return order[int(answer) - 1]
56
+ except (ValueError, IndexError):
57
+ from agentctl.control.providers import BY_NAME
58
+ return BY_NAME.get(answer.lower())
59
+
60
+
61
+ def _store_key(p, value: str) -> None:
62
+ from agentctl.control.keys import HOME_PATH, set_value
63
+
64
+ path = set_value(HOME_PATH, p.key, value)
65
+ os.environ[p.key] = value
66
+ print(f" saved {p.key} -> {path} (never printed, outside any repo)")
67
+
68
+
69
+ def run_init(provider: str | None = None, model: str | None = None,
70
+ verify: bool = True, key_stdin: bool = False,
71
+ check_paid: bool = False, interactive: bool | None = None) -> int:
72
+ from agentctl.control.providers import BY_NAME, configured
73
+
74
+ if interactive is None:
75
+ interactive = sys.stdin.isatty()
76
+ print("agentctl init\n")
77
+
78
+ # ── 1. a key ────────────────────────────────────────────────────────
79
+ if provider:
80
+ p = BY_NAME.get(provider.lower())
81
+ if p is None:
82
+ print(f" no provider called {provider!r}. Known: "
83
+ f"{', '.join(BY_NAME)}")
84
+ return 2
85
+ else:
86
+ have = configured()
87
+ p = have[0] if have else _choose(interactive and not key_stdin)
88
+
89
+ if p is None:
90
+ print(" Non-interactive, and no key in the environment. Either:")
91
+ print(" export OPENROUTER_API_KEY=... then agentctl init")
92
+ print(" agentctl init --provider openrouter --key-stdin < keyfile")
93
+ return 2
94
+
95
+ if not p.configured:
96
+ print(f" {p.name}: create a key at {p.console}")
97
+ for step in p.steps:
98
+ print(f" - {step}")
99
+ if key_stdin:
100
+ value = sys.stdin.readline().strip()
101
+ elif interactive:
102
+ value = getpass.getpass(f" paste your {p.key} (not shown): ").strip()
103
+ else:
104
+ value = ""
105
+ if not value:
106
+ print(f"\n no key given. Set {p.key} and run `agentctl init` again.")
107
+ return 2
108
+ _store_key(p, value)
109
+ else:
110
+ print(f" key {p.key} (already set)")
111
+
112
+ # ── 2. prove it ────────────────────────────────────────────────────
113
+ chosen, how = model or p.default_model, "registry default, not checked"
114
+ if model:
115
+ how = "given with --model"
116
+ elif not verify:
117
+ how = "registry default; --no-verify, so not checked"
118
+ elif not p.free_tier and not check_paid:
119
+ how = ("registry default; not checked -- a paid provider is called "
120
+ "only with --check-paid (one 4-token completion)")
121
+ else:
122
+ from agentctl.control.probe import LIMITED, LIVE, check_inference
123
+
124
+ print(f" checking one completion against {p.name} ...")
125
+ r = check_inference(p.name, allow_paid=check_paid)
126
+ if r is not None and r.status == LIVE and r.model:
127
+ chosen, how = f"{p.prefix}{r.model}", "answered a completion just now"
128
+ elif r is not None and r.status == LIMITED:
129
+ how = ("the key works but is rate limited right now; registry "
130
+ "default recorded, unchecked")
131
+ else:
132
+ print(f" !! {p.name} cannot serve a request: "
133
+ f"{r.detail if r else 'no model to check'}")
134
+ print(f" the key is saved; nothing else was changed. Check it at "
135
+ f"{p.console}, then run `agentctl init` again.")
136
+ return 1
137
+
138
+ # ── 3. record it ───────────────────────────────────────────────────
139
+ cfg = config.load()
140
+ cfg["model"] = chosen
141
+ path = config.write(cfg, note=f"model: {how}")
142
+ print(f" model {chosen}")
143
+ print(f" ({how})")
144
+ print(f" config {path}")
145
+ print()
146
+ print(" Ready. In any git repository:")
147
+ print(' agentctl run "describe what this repository does"')
148
+ return 0
@@ -0,0 +1,143 @@
1
+ """Who holds a conversation's lease, and whether they are still alive.
2
+
3
+ `docs/0042` I-02. `--resume` used to pass `takeover=True` unconditionally, so a
4
+ second terminal could steal the lease from a run that was still working, and
5
+ two processes drove one conversation: one event history, two writers.
6
+
7
+ The fix keeps "crash, then resume" frictionless and refuses the dangerous
8
+ case. Holders are named `run@<host>:<pid>`, so a resume can ask whether the
9
+ holder is alive:
10
+
11
+ holder dead (same host) take over -- a crashed process cannot
12
+ release its own lease, and waiting out the
13
+ TTL helps nobody
14
+ holder alive, or unknowable refuse and name it; `--takeover` overrides
15
+ lease expired or released nothing to steal
16
+
17
+ "Unknowable" counts as alive: a holder on another host, or a name not in this
18
+ format. Taking over from a live process is the failure; waiting is only slow.
19
+ """
20
+ from __future__ import annotations
21
+
22
+ import os
23
+ import socket
24
+ import sys
25
+ import time
26
+ from dataclasses import dataclass
27
+ from pathlib import Path
28
+
29
+ PREFIX = "run@"
30
+
31
+
32
+ def holder_id(pid: int | None = None) -> str:
33
+ """This process's lease holder name."""
34
+ return f"{PREFIX}{socket.gethostname()}:{pid or os.getpid()}"
35
+
36
+
37
+ def _parse(holder: str) -> tuple[str, int] | None:
38
+ if not holder.startswith(PREFIX) or ":" not in holder:
39
+ return None
40
+ host, _, pid = holder[len(PREFIX):].rpartition(":")
41
+ try:
42
+ return host, int(pid)
43
+ except ValueError:
44
+ return None
45
+
46
+
47
+ def pid_alive(pid: int) -> bool:
48
+ """Is a process with this pid running on this machine?
49
+
50
+ Never `os.kill(pid, 0)` on Windows: there, `os.kill` with any signal other
51
+ than the two console events calls TerminateProcess. A liveness check that
52
+ kills what it checks would be a remarkable way to fail.
53
+ """
54
+ if pid <= 0:
55
+ return False
56
+ if sys.platform == "win32":
57
+ import ctypes
58
+ from ctypes import wintypes
59
+
60
+ k32 = ctypes.WinDLL("kernel32", use_last_error=True)
61
+ k32.OpenProcess.restype = wintypes.HANDLE
62
+ k32.OpenProcess.argtypes = (wintypes.DWORD, wintypes.BOOL, wintypes.DWORD)
63
+ k32.GetExitCodeProcess.argtypes = (wintypes.HANDLE,
64
+ ctypes.POINTER(wintypes.DWORD))
65
+ k32.CloseHandle.argtypes = (wintypes.HANDLE,)
66
+ PROCESS_QUERY_LIMITED_INFORMATION, STILL_ACTIVE = 0x1000, 259
67
+ handle = k32.OpenProcess(PROCESS_QUERY_LIMITED_INFORMATION, False, pid)
68
+ if not handle:
69
+ # ERROR_ACCESS_DENIED means it exists and is someone else's.
70
+ return ctypes.get_last_error() == 5
71
+ try:
72
+ code = wintypes.DWORD()
73
+ if not k32.GetExitCodeProcess(handle, ctypes.byref(code)):
74
+ return True # cannot tell: assume alive
75
+ return code.value == STILL_ACTIVE
76
+ finally:
77
+ k32.CloseHandle(handle)
78
+ try:
79
+ os.kill(pid, 0)
80
+ except ProcessLookupError:
81
+ return False
82
+ except PermissionError:
83
+ return True # exists, not ours
84
+ return True
85
+
86
+
87
+ def holder_alive(holder: str) -> bool | None:
88
+ """True / False when this machine can tell; None when it cannot."""
89
+ parsed = _parse(holder)
90
+ if parsed is None:
91
+ return None
92
+ host, pid = parsed
93
+ if host != socket.gethostname():
94
+ return None
95
+ return pid_alive(pid)
96
+
97
+
98
+ @dataclass(frozen=True)
99
+ class Claim:
100
+ """What a resume should do about the lease."""
101
+ takeover: bool
102
+ note: str = ""
103
+
104
+
105
+ class StillRunning(SystemExit):
106
+ """The conversation's holder is alive. Refuse, and say who and how."""
107
+
108
+
109
+ def claim(ledger: str | Path, conversation_id: str, *,
110
+ force: bool = False) -> Claim:
111
+ """Decide whether resuming `conversation_id` may take its lease.
112
+
113
+ Reads the lease and nothing else; `protect()` then acquires with the
114
+ answer. Between the two, another process could still take it -- in which
115
+ case `acquire` raises `LeaseHeld`, which is the same refusal.
116
+ """
117
+ from agentctl.kernel.ledger.store import LedgerStore
118
+
119
+ p = Path(ledger)
120
+ if not p.exists():
121
+ return Claim(False)
122
+ with LedgerStore(p, holder="claim") as s:
123
+ row = s.lease(conversation_id)
124
+ if row is None or row["expires_at"] <= time.time():
125
+ return Claim(False)
126
+
127
+ holder, left = row["holder"], row["expires_at"] - time.time()
128
+ alive = holder_alive(holder)
129
+ if alive is False:
130
+ return Claim(True, f"the run that held it ({holder}) is gone; "
131
+ f"taking over")
132
+ if force:
133
+ return Claim(True, f"--takeover: stealing the lease from {holder}, "
134
+ f"which {'is ALIVE' if alive else 'may be alive'}. "
135
+ f"Its next write will be refused.")
136
+ raise StillRunning(
137
+ f"conversation {conversation_id} is being driven by {holder}"
138
+ f"{'' if alive else ' (cannot tell from here whether it is alive)'}; "
139
+ f"its lease has {left:.0f}s left and renews while it runs.\n"
140
+ f" Two processes driving one conversation is the thing the ledger's\n"
141
+ f" single-writer guarantee forbids (docs/0042 I-02).\n"
142
+ f" If that run is still working: let it finish, or stop it first.\n"
143
+ f" If it is gone: add --takeover")