handcode 0.3.0rc1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (67) hide show
  1. agentctl/__init__.py +0 -0
  2. agentctl/adapters/__init__.py +0 -0
  3. agentctl/adapters/litellm/__init__.py +9 -0
  4. agentctl/adapters/litellm/hook.py +49 -0
  5. agentctl/adapters/litellm/recorder.py +187 -0
  6. agentctl/adapters/openhands/__init__.py +169 -0
  7. agentctl/adapters/openhands/handoff.py +155 -0
  8. agentctl/adapters/openhands/seam_b.py +259 -0
  9. agentctl/adapters/openhands/seam_c.py +209 -0
  10. agentctl/cli.py +1450 -0
  11. agentctl/control/__init__.py +0 -0
  12. agentctl/control/cost/__init__.py +4 -0
  13. agentctl/control/cost/ledger.py +210 -0
  14. agentctl/control/dash.py +697 -0
  15. agentctl/control/keys.py +440 -0
  16. agentctl/control/matrix/__init__.py +0 -0
  17. agentctl/control/matrix/data/tools.yaml +149 -0
  18. agentctl/control/policy/__init__.py +10 -0
  19. agentctl/control/policy/compile.py +258 -0
  20. agentctl/control/policy/data/policy.compiled.json +38 -0
  21. agentctl/control/policy/data/policy.yaml +46 -0
  22. agentctl/control/probe.py +399 -0
  23. agentctl/control/providers.py +293 -0
  24. agentctl/control/proxy.py +536 -0
  25. agentctl/control/proxyenv.py +309 -0
  26. agentctl/control/replay/__init__.py +14 -0
  27. agentctl/control/replay/cassette.py +281 -0
  28. agentctl/control/replay/server.py +109 -0
  29. agentctl/demo/__init__.py +214 -0
  30. agentctl/demo/child.py +84 -0
  31. agentctl/demo/mock.py +79 -0
  32. agentctl/demo/tool.py +62 -0
  33. agentctl/gha.py +488 -0
  34. agentctl/kernel/__init__.py +0 -0
  35. agentctl/kernel/classify.py +170 -0
  36. agentctl/kernel/gate.py +391 -0
  37. agentctl/kernel/hook.py +229 -0
  38. agentctl/kernel/ledger/__init__.py +0 -0
  39. agentctl/kernel/ledger/models.py +160 -0
  40. agentctl/kernel/ledger/schema.sql +62 -0
  41. agentctl/kernel/ledger/store.py +596 -0
  42. agentctl/kernel/paths.py +203 -0
  43. agentctl/kernel/policy.py +160 -0
  44. agentctl/kernel/reconcile/__init__.py +31 -0
  45. agentctl/kernel/reconcile/base.py +106 -0
  46. agentctl/kernel/reconcile/external.py +137 -0
  47. agentctl/kernel/reconcile/filesystem.py +162 -0
  48. agentctl/kernel/reconcile/git.py +162 -0
  49. agentctl/runtime/__init__.py +20 -0
  50. agentctl/runtime/citations.py +179 -0
  51. agentctl/runtime/config.py +97 -0
  52. agentctl/runtime/doctor.py +335 -0
  53. agentctl/runtime/init.py +148 -0
  54. agentctl/runtime/lease.py +143 -0
  55. agentctl/runtime/orchestrate.py +187 -0
  56. agentctl/runtime/plugins.py +130 -0
  57. agentctl/runtime/report.py +361 -0
  58. agentctl/runtime/runner.py +787 -0
  59. agentctl/runtime/runs.py +191 -0
  60. agentctl/runtime/subagent.py +274 -0
  61. agentctl/runtime/tools.py +350 -0
  62. handcode-0.3.0rc1.dist-info/METADATA +659 -0
  63. handcode-0.3.0rc1.dist-info/RECORD +67 -0
  64. handcode-0.3.0rc1.dist-info/WHEEL +5 -0
  65. handcode-0.3.0rc1.dist-info/entry_points.txt +3 -0
  66. handcode-0.3.0rc1.dist-info/licenses/LICENSE +21 -0
  67. handcode-0.3.0rc1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,259 @@
1
+ """Seam B binding for OpenHands. Spec: `docs/0008` §3.5, `docs/0012` §3.2.1.
2
+
3
+ Binds BLOCK and ESCALATE via `ConversationState.block_action`, which fires from
4
+ the event callback when an `ActionEvent` is emitted — before the tool runs
5
+ (`docs/0015` §2).
6
+
7
+ This is the whole of M2a. No executor wrapping, no fork, and it already gives
8
+ correct fail-closed behaviour: a dangerous effect whose outcome is unknown is
9
+ refused rather than repeated. What it cannot do is SUBSTITUTE a stored
10
+ observation — that needs Seam C (M2b).
11
+
12
+ Usage — note the two-step, forced by a chicken-and-egg: callbacks are passed at
13
+ construction, but the callback needs the conversation it is attached to.
14
+
15
+ seam = SeamB(gate)
16
+ conv = Conversation(agent=..., callbacks=[seam])
17
+ seam.attach(conv)
18
+ """
19
+ from __future__ import annotations
20
+
21
+ import json
22
+ import logging
23
+ from typing import Any, Callable
24
+
25
+ from agentctl.kernel.gate import EffectGate
26
+ from agentctl.kernel.ledger.models import GateDecision, ToolCall, Verdict
27
+
28
+ log = logging.getLogger("agentctl.seam_b")
29
+
30
+
31
+ class OpenHandsContext:
32
+ """Everything harness-specific lives here.
33
+
34
+ Keeping the translation in one small class is what makes porting to another
35
+ harness cheap (`docs/0008` §12).
36
+ """
37
+
38
+ def __init__(self, conversation_id: str):
39
+ self.conversation_id = conversation_id
40
+
41
+ def to_tool_call(self, event: Any) -> ToolCall | None:
42
+ """Normalise an ActionEvent into a harness-neutral ToolCall."""
43
+ tool_call_id = getattr(event, "tool_call_id", None)
44
+ if not tool_call_id:
45
+ tc = getattr(event, "tool_call", None)
46
+ tool_call_id = getattr(tc, "id", None) if tc else None
47
+ if not tool_call_id:
48
+ return None
49
+
50
+ action = getattr(event, "action", None)
51
+ args: dict = {}
52
+ if action is not None:
53
+ try:
54
+ args = action.model_dump()
55
+ except Exception: # noqa: BLE001
56
+ args = {k: v for k, v in vars(action).items()
57
+ if not k.startswith("_")}
58
+ args.pop("kind", None)
59
+
60
+ return ToolCall(
61
+ tool_call_id=str(tool_call_id),
62
+ conversation_id=self.conversation_id,
63
+ turn_id=str(getattr(event, "id", "") or ""),
64
+ tool_name=str(getattr(event, "tool_name", "") or ""),
65
+ args=args,
66
+ )
67
+
68
+ @staticmethod
69
+ def serialize(event_or_observation: Any) -> bytes:
70
+ """Serialize the Observation, not the wrapping ObservationEvent.
71
+
72
+ Seam C revives this with `observation_type.model_validate_json`, so it
73
+ must be the observation's own shape.
74
+ """
75
+ obj = getattr(event_or_observation, "observation", None) or event_or_observation
76
+ try:
77
+ return obj.model_dump_json().encode()
78
+ except Exception: # noqa: BLE001
79
+ return json.dumps(str(obj)).encode()
80
+
81
+
82
+ def is_action_event(event: Any) -> bool:
83
+ """Duck-typed so nothing here needs the SDK imported."""
84
+ return (
85
+ type(event).__name__ == "ActionEvent"
86
+ and getattr(event, "action", None) is not None
87
+ )
88
+
89
+
90
+ def is_observation_event(event: Any) -> bool:
91
+ return type(event).__name__ in ("ObservationEvent", "AgentErrorEvent")
92
+
93
+
94
+ class SeamB:
95
+ """Event callback that enforces the gate's BLOCK/ESCALATE verdicts.
96
+
97
+ Relies on the ordering contract in `docs/0006:V1`: the ActionEvent is
98
+ persisted, and callbacks fire, *before* the tool executes. Blocking here
99
+ prevents execution rather than merely recording disapproval.
100
+ """
101
+
102
+ def __init__(
103
+ self,
104
+ gate: EffectGate,
105
+ ctx: OpenHandsContext | None = None,
106
+ on_decision: Callable[[ToolCall, GateDecision], None] | None = None,
107
+ handoff: Any = None,
108
+ ):
109
+ self.gate = gate
110
+ self.ctx = ctx
111
+ self.on_decision = on_decision
112
+ # When Seam C is installed, SUBSTITUTE is handed to it instead of being
113
+ # downgraded to a block. Without it, behaviour degrades to M2a:
114
+ # still correct, just a rejection where a result was possible.
115
+ self.handoff = handoff
116
+ self._state: Any = None
117
+ self._pending: dict[str, str] = {} # action_event_id -> tool_call_id
118
+ # action_event_id -> the record whose result Seam C is handing back.
119
+ # Its ObservationEvent is when that result reaches the model.
120
+ self._substituted: dict[str, str] = {}
121
+
122
+ def attach(self, conversation: Any) -> "SeamB":
123
+ """Bind to a live conversation. Call immediately after construction."""
124
+ self._state = conversation.state
125
+ if self.ctx is None:
126
+ self.ctx = OpenHandsContext(str(conversation.state.id))
127
+ return self
128
+
129
+ # ── the callback ───────────────────────────────────────────────────
130
+ def __call__(self, event: Any) -> None:
131
+ try:
132
+ if is_observation_event(event):
133
+ self._close(event)
134
+ return
135
+ if not is_action_event(event):
136
+ return
137
+ if self._state is None:
138
+ log.error("SeamB not attached; the gate is INERT. "
139
+ "Call seam.attach(conversation) after construction.")
140
+ return
141
+ self._decide(event)
142
+ except Exception: # noqa: BLE001
143
+ # A callback bug must not take down the agent loop, but it must
144
+ # never silently disable the guard either. Log loudly.
145
+ log.exception("seam B callback failed for %r", event)
146
+
147
+ # ── internals ──────────────────────────────────────────────────────
148
+ def _decide(self, event: Any) -> None:
149
+ assert self.ctx is not None
150
+ call = self.ctx.to_tool_call(event)
151
+ if call is None:
152
+ return
153
+
154
+ decision = self.gate.guard(call)
155
+
156
+ if decision.verdict is Verdict.EXECUTE:
157
+ self._pending[str(event.id)] = call.tool_call_id
158
+ if self.handoff is not None:
159
+ # The fingerprint just taken describes the world before this
160
+ # whole batch. Let Seam C refresh it at the moment this call
161
+ # actually runs (`docs/0038` §3). Without Seam C the batch
162
+ # hazard remains, and the gate's `_sole_writer` check is what
163
+ # keeps that failing closed rather than silently wrong.
164
+ self.handoff.arm(event.action, call,
165
+ lambda c=call: self.gate.recapture(c))
166
+
167
+ elif decision.verdict in (Verdict.BLOCK, Verdict.ESCALATE):
168
+ reason = decision.reason or "blocked by agentctl"
169
+ self._state.block_action(event.id, reason)
170
+ log.warning("BLOCKED %s (%s): %s",
171
+ call.tool_name, call.tool_call_id, reason)
172
+
173
+ elif decision.verdict is Verdict.SUBSTITUTE:
174
+ if self.handoff is not None:
175
+ # Seam C will return the recorded observation. Let the harness
176
+ # proceed to the executor -- it never reaches the real tool.
177
+ self.handoff.offer(event.action, call, decision.observation)
178
+ if decision.record_id:
179
+ self._substituted[str(event.id)] = decision.record_id
180
+ log.info("handing %s to Seam C for substitution", call.tool_call_id)
181
+ else:
182
+ # No Seam C: block instead. Correctness preserved, resume
183
+ # quality is not. docs/0008 §3.5 -- degrade, never fail open.
184
+ reason = (
185
+ f"{call.tool_name} already ran and its result was recorded; "
186
+ f"agentctl blocked a repeat. Install the Seam C executor "
187
+ f"wrap to resume cleanly instead of blocking."
188
+ )
189
+ self._state.block_action(event.id, reason)
190
+ log.warning("SUBSTITUTE needed, Seam B can only block: %s",
191
+ call.tool_call_id)
192
+
193
+ if self.on_decision:
194
+ self.on_decision(call, decision)
195
+
196
+ def _close(self, event: Any) -> None:
197
+ """Close the ledger record. Pure observation, so Seam B can do it.
198
+
199
+ Everything here runs AFTER the SDK has persisted `event`: its default
200
+ callback appends and fsyncs before any caller callback, and if that
201
+ raises none of them run (`docs/0042` §8.2, checked in SDK 1.45.0 and
202
+ 1.50.1). That ordering is what lets this record the result as
203
+ OBSERVED -- in the history the model reads -- rather than merely
204
+ returned (`docs/0045`).
205
+ """
206
+ action_id = getattr(event, "action_id", None)
207
+ if (rid := self._substituted.pop(action_id, None)) is not None:
208
+ self.gate.mark_observed(rid)
209
+ return
210
+ tcid = self._pending.pop(action_id, None)
211
+ if tcid is None:
212
+ return
213
+ assert self.ctx is not None
214
+ if type(event).__name__ == "AgentErrorEvent":
215
+ self.gate.record_failure_by_id(
216
+ tcid, str(getattr(event, "error", "tool error"))[:500])
217
+ return
218
+
219
+ # The event type is not the whole story. A tool that RAN and failed
220
+ # still arrives as an ordinary ObservationEvent, and committing it
221
+ # tells a resume the work is done. A real run recorded five `exit=1`
222
+ # shell commands as COMMITTED (`docs/0031` §10). So the error travels
223
+ # with the observation, and the record says the tool reported failure.
224
+ self.gate.record_observation(tcid, self.ctx.serialize(event),
225
+ error=_tool_error(event))
226
+
227
+
228
+ def make_seam_b_callback(
229
+ conversation: Any,
230
+ gate: EffectGate,
231
+ ctx: OpenHandsContext | None = None,
232
+ on_decision: Callable[[ToolCall, GateDecision], None] | None = None,
233
+ ) -> SeamB:
234
+ """Convenience for when the conversation already exists."""
235
+ return SeamB(gate, ctx, on_decision).attach(conversation)
236
+
237
+
238
+ def _tool_error(event: Any) -> str | None:
239
+ """Did the tool report failure? Returns why, or None if it succeeded.
240
+
241
+ `Observation.is_error` is the SDK's own flag and the only honest source:
242
+ the event type is `ObservationEvent` either way, so reading the type alone
243
+ cannot tell a success from a failure (`docs/0031` §10).
244
+
245
+ Returns None when there is no observation or no flag, because "cannot tell"
246
+ must not become "failed" -- treating every unreadable event as an error
247
+ would block work for no reason.
248
+ """
249
+ obs = getattr(event, "observation", None)
250
+ if obs is None or not getattr(obs, "is_error", False):
251
+ return None
252
+
253
+ # Prefer the tool's own text; it names the actual shell error.
254
+ for attr in ("output", "status", "file_text"):
255
+ if (v := getattr(obs, attr, None)):
256
+ return str(v)[:300]
257
+ if (code := getattr(obs, "exit_code", None)) is not None:
258
+ return f"exit={code}"
259
+ return "the tool reported an error"
@@ -0,0 +1,209 @@
1
+ """Seam C binding for OpenHands — the substitution channel.
2
+
3
+ Spec: `docs/0008` §3.5, `docs/0012` §4, corrected by `docs/0014` C1.
4
+
5
+ Binds by replacing the `executor` field on a resolved `ToolDefinition`. No
6
+ subclassing of `ToolExecutor` to re-register, no fork:
7
+
8
+ gated = tool.model_copy(update={"executor": GatedExecutor(...)})
9
+
10
+ What it adds over Seam B alone:
11
+
12
+ 1. **Substitution** — when an effect already landed, the agent receives the
13
+ recorded observation instead of a rejection, so the loop continues as though
14
+ the crash never happened.
15
+ 2. **Idempotency-key injection** — Seam C is the only place that can modify a
16
+ call before it executes, and injecting a stable key is the only general way
17
+ to make a remote effect replay-safe (`docs/0020`).
18
+
19
+ The second is a deliberate widening of Seam C's role, and it stays narrow: the
20
+ key is *computed by the kernel* and merely written onto the call here. Seam C
21
+ still decides nothing.
22
+
23
+ It remains thin — a lookup, a branch, an optional field write, and the inner
24
+ call. All policy lives in the gate; all decisions are made at Seam B
25
+ (`handoff.py`). Keeping this small is the whole portability argument in
26
+ `docs/0008` §12.
27
+ """
28
+ from __future__ import annotations
29
+
30
+ import logging
31
+ from typing import Any
32
+
33
+ from openhands.sdk.tool import ToolDefinition, ToolExecutor, register_tool
34
+
35
+ from .handoff import SubstitutionHandoff
36
+
37
+ log = logging.getLogger("agentctl.seam_c")
38
+
39
+
40
+ class GatedExecutor(ToolExecutor):
41
+ """Wraps a real executor. Substitutes when Seam B says the effect landed."""
42
+
43
+ def __init__(self, inner: ToolExecutor, handoff: SubstitutionHandoff,
44
+ observation_type: type | None = None, tool_name: str = "",
45
+ idempotency: Any = None):
46
+ self._inner = inner
47
+ self._handoff = handoff
48
+ self._observation_type = observation_type
49
+ self._tool_name = tool_name
50
+ #: An `IdempotencyProbe`, when this tool declares a key field.
51
+ self._idempotency = idempotency
52
+
53
+ def __call__(self, action, conversation=None):
54
+ claim = self._handoff.claim(action, self._tool_name, _args(action))
55
+ if claim is not None:
56
+ _call, raw = claim
57
+ if (obs := self._revive(raw)) is not None:
58
+ log.info("substituted recorded observation for %s", self._tool_name)
59
+ return obs
60
+ if (obs := self._synthesize()) is not None:
61
+ log.info("synthesized a recovery observation for %s", self._tool_name)
62
+ return obs
63
+ # It landed, we cannot describe it, and re-running would duplicate
64
+ # the effect. Raising is the safe end: the harness surfaces an
65
+ # error, and the ledger already says COMMITTED.
66
+ raise RuntimeError(
67
+ f"{self._tool_name} already executed, but its result could not "
68
+ f"be recovered or described. It was NOT re-run."
69
+ )
70
+
71
+ # No claim means Seam B said EXECUTE. Seam C never decides.
72
+ #
73
+ # It does, however, re-fingerprint: Seam B decided for the whole batch,
74
+ # and this is the only point at which a call is seen at its own
75
+ # execution moment (`docs/0038` §3). Still no decision -- the kernel
76
+ # captures, this only says when.
77
+ self._handoff.before_execute(action, self._tool_name, _args(action))
78
+ return self._inner(self._stamp_idempotency_key(action), conversation)
79
+
80
+ def _stamp_idempotency_key(self, action):
81
+ """Write the kernel's stable key onto the call before it goes out.
82
+
83
+ Without this the IdempotencyProbe declines and the effect fails closed,
84
+ so the injection is not an optimisation -- it IS the mechanism
85
+ (`docs/0020`).
86
+ """
87
+ if self._idempotency is None:
88
+ return action
89
+ try:
90
+ from agentctl.kernel.ledger.models import ToolCall
91
+ call = ToolCall("stamp", "", "", self._tool_name, _args(action))
92
+ field = self._idempotency.key_field(call)
93
+ if field is None:
94
+ return action
95
+ existing = getattr(action, field, None)
96
+ if existing:
97
+ # The caller already supplied a key. It is stable across a
98
+ # resume for the same reason ours is, so leave it alone.
99
+ return action
100
+ return action.model_copy(
101
+ update={field: self._idempotency.key_for(call)})
102
+ except Exception: # noqa: BLE001
103
+ # A failed stamp means the probe declines and the effect fails
104
+ # closed. Safe, but say so loudly: it silently costs protection.
105
+ log.exception("could not stamp an idempotency key on %s",
106
+ self._tool_name)
107
+ return action
108
+
109
+ def _revive(self, raw: bytes | None):
110
+ """Rebuild the observation actually recorded at commit time."""
111
+ if not raw or self._observation_type is None:
112
+ return None
113
+ try:
114
+ return self._observation_type.model_validate_json(raw)
115
+ except Exception: # noqa: BLE001
116
+ log.exception("could not revive observation for %s", self._tool_name)
117
+ return None
118
+
119
+ def _synthesize(self):
120
+ """Describe a recovered effect when no observation was ever recorded.
121
+
122
+ A probe-reconciled effect has this shape: the crash happened *before*
123
+ the observation was written, so the ledger knows the effect landed but
124
+ holds no result. There is nothing to revive.
125
+
126
+ The agent still needs something truthful to continue on, so say plainly
127
+ what is known — it ran, its output is unavailable — rather than
128
+ inventing a plausible result or re-running the tool.
129
+ """
130
+ if self._observation_type is None:
131
+ return None
132
+ note = (
133
+ f"[agentctl] {self._tool_name} already completed before an "
134
+ f"interruption. The effect is confirmed; its original output was "
135
+ f"not recorded and is unavailable. It was not run again."
136
+ )
137
+ try:
138
+ obs = self._observation_type() # defaults only
139
+ except Exception: # noqa: BLE001
140
+ return None # required fields: cannot describe
141
+ try:
142
+ # Observation.content is list[TextContent | ImageContent], not a
143
+ # string. Assigning a bare str validates here and then explodes
144
+ # later when the message is assembled.
145
+ from openhands.sdk.llm import TextContent
146
+ return obs.model_copy(update={"content": [TextContent(text=note)]})
147
+ except Exception: # noqa: BLE001
148
+ log.exception("could not attach a recovery note to %s", self._tool_name)
149
+ return obs
150
+
151
+ # Delegate lifecycle so the wrapper is transparent.
152
+ def close(self) -> None:
153
+ getattr(self._inner, "close", lambda: None)()
154
+
155
+ def interrupt(self) -> None:
156
+ getattr(self._inner, "interrupt", lambda: None)()
157
+
158
+
159
+ def gate_tools(tools, handoff: SubstitutionHandoff, idempotency: Any = None):
160
+ """Replace each tool's executor with a gated one. `docs/0014` C1."""
161
+ out = []
162
+ for t in tools:
163
+ if t.executor is None:
164
+ out.append(t)
165
+ continue
166
+ out.append(t.model_copy(update={"executor": GatedExecutor(
167
+ t.executor, handoff, t.observation_type, t.name, idempotency)}))
168
+ return out
169
+
170
+
171
+ def gate_tool_class(inner_cls: type[ToolDefinition], handoff: SubstitutionHandoff,
172
+ idempotency: Any = None):
173
+ """Build a subclass whose `create()` returns gated tools."""
174
+
175
+ class Gated(inner_cls): # type: ignore[misc,valid-type]
176
+ @classmethod
177
+ def create(cls, conv_state=None, **params):
178
+ return gate_tools(
179
+ inner_cls.create(conv_state=conv_state, **params),
180
+ handoff, idempotency)
181
+
182
+ Gated.__name__ = f"Gated{inner_cls.__name__}"
183
+ Gated.__qualname__ = Gated.__name__
184
+ return Gated
185
+
186
+
187
+ def install(handoff: SubstitutionHandoff, tools: dict[str, type],
188
+ idempotency: Any = None) -> list[str]:
189
+ """Re-register each named tool with a gated executor.
190
+
191
+ Call before constructing the Agent. Returns the names that were gated.
192
+ """
193
+ gated = []
194
+ for name, cls in tools.items():
195
+ try:
196
+ register_tool(name, gate_tool_class(cls, handoff, idempotency))
197
+ gated.append(name)
198
+ except Exception: # noqa: BLE001
199
+ log.exception("could not gate tool %s", name)
200
+ return gated
201
+
202
+
203
+ def _args(action: Any) -> dict:
204
+ try:
205
+ d = action.model_dump()
206
+ except Exception: # noqa: BLE001
207
+ d = {k: v for k, v in vars(action).items() if not k.startswith("_")}
208
+ d.pop("kind", None)
209
+ return d