quackd 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (55) hide show
  1. quackd/__init__.py +8 -0
  2. quackd/agent/__init__.py +6 -0
  3. quackd/agent/loop.py +332 -0
  4. quackd/agent/prompts.py +135 -0
  5. quackd/agent/providers/__init__.py +6 -0
  6. quackd/agent/providers/anthropic.py +209 -0
  7. quackd/agent/providers/base.py +103 -0
  8. quackd/agent/providers/factory.py +63 -0
  9. quackd/agent/providers/fake.py +141 -0
  10. quackd/agent/providers/gemini.py +162 -0
  11. quackd/agent/providers/grok.py +19 -0
  12. quackd/agent/providers/openai.py +159 -0
  13. quackd/agent/transcript.py +73 -0
  14. quackd/cli.py +392 -0
  15. quackd/doctor.py +135 -0
  16. quackd/duckfile/__init__.py +6 -0
  17. quackd/duckfile/export.py +25 -0
  18. quackd/duckfile/parser.py +134 -0
  19. quackd/duckfile/schema.json +198 -0
  20. quackd/duckfile/schema.py +172 -0
  21. quackd/ducks/fetch.duck +41 -0
  22. quackd/ducks/find-and-kick.duck +41 -0
  23. quackd/ducks/follow-me.duck +40 -0
  24. quackd/ducks/hello-world.duck +33 -0
  25. quackd/ducks/patrol-and-quack.duck +40 -0
  26. quackd/mcp_server.py +250 -0
  27. quackd/perception/__init__.py +6 -0
  28. quackd/perception/base.py +49 -0
  29. quackd/perception/color_blob.py +96 -0
  30. quackd/perception/yolo.py +59 -0
  31. quackd/safety.py +329 -0
  32. quackd/sim2d/__init__.py +5 -0
  33. quackd/sim2d/live.py +39 -0
  34. quackd/sim2d/recorder.py +80 -0
  35. quackd/sim2d/render.py +116 -0
  36. quackd/sim2d/world.py +272 -0
  37. quackd/transport/__init__.py +6 -0
  38. quackd/transport/base.py +144 -0
  39. quackd/transport/factory.py +46 -0
  40. quackd/transport/jsonrpc_unix.py +313 -0
  41. quackd/transport/mock.py +93 -0
  42. quackd/transport/sim2d.py +163 -0
  43. quackd/transport/upstream_api.py +201 -0
  44. quackd/transport/websocket_stub.py +70 -0
  45. quackd/verbs/__init__.py +6 -0
  46. quackd/verbs/builtin.py +311 -0
  47. quackd/verbs/composite.py +199 -0
  48. quackd/verbs/learned.py +65 -0
  49. quackd/verbs/registry.py +152 -0
  50. quackd-0.1.0.dist-info/METADATA +433 -0
  51. quackd-0.1.0.dist-info/RECORD +55 -0
  52. quackd-0.1.0.dist-info/WHEEL +4 -0
  53. quackd-0.1.0.dist-info/entry_points.txt +2 -0
  54. quackd-0.1.0.dist-info/licenses/LICENSE +201 -0
  55. quackd-0.1.0.dist-info/licenses/NOTICE +29 -0
quackd/__init__.py ADDED
@@ -0,0 +1,8 @@
1
+ """quackd — the brain daemon Microduck was missing.
2
+
3
+ This package exists so that any LLM can pilot a Microduck (real or simulated) through a
4
+ small, safety-enforced vocabulary of *verbs*, driven by a `.duck` skill file. Everything else
5
+ in the repo serves that sentence.
6
+ """
7
+
8
+ __version__ = "0.1.0"
@@ -0,0 +1,6 @@
1
+ """The deliberation loop: observe → think → enforce → act.
2
+
3
+ This package exists to make the LLM a *high-level* controller and nothing more: one verb
4
+ per turn, judged against the `.duck` success criteria, with every prompt and result written
5
+ to a transcript so a run can be replayed and argued about.
6
+ """
quackd/agent/loop.py ADDED
@@ -0,0 +1,332 @@
1
+ """observe → think → enforce → act, until success, failure, budget, or abort.
2
+
3
+ This is the deliberation loop. It owns nothing clever: perception is a detector, safety is
4
+ the executor, memory is the transcript. What it does own is the *shape* of a turn — one
5
+ observation in, exactly one tool call out — and the honest bookkeeping of why a run ended.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import asyncio
11
+ import contextlib
12
+ from dataclasses import dataclass, field
13
+ from pathlib import Path
14
+ from typing import Any, Literal
15
+
16
+ from PIL import Image
17
+
18
+ from quackd.agent.prompts import (
19
+ META_TOOL_NAMES,
20
+ META_TOOLS,
21
+ build_observation_text,
22
+ build_system_prompt,
23
+ observation_features,
24
+ )
25
+ from quackd.agent.providers.base import (
26
+ Decision,
27
+ Exchange,
28
+ LLMProvider,
29
+ Observation,
30
+ ToolCall,
31
+ Usage,
32
+ )
33
+ from quackd.agent.transcript import Transcript, new_run_dir, png_bytes
34
+ from quackd.duckfile.schema import DuckFile
35
+ from quackd.perception.base import Detection, Detector
36
+ from quackd.safety import (
37
+ Aborted,
38
+ Budget,
39
+ BudgetExceeded,
40
+ ConfirmDenied,
41
+ ConfirmFn,
42
+ Executor,
43
+ Heartbeat,
44
+ SafetyStop,
45
+ VerbNotAllowed,
46
+ deny_all,
47
+ )
48
+ from quackd.transport.base import DuckTransport
49
+ from quackd.verbs.registry import VerbRegistry, VerbResult, default_registry
50
+
51
+ Outcome = Literal["success", "failure", "budget", "aborted", "error"]
52
+
53
+
54
+ @dataclass
55
+ class RunConfig:
56
+ duck: DuckFile
57
+ provider: LLMProvider
58
+ transport: DuckTransport
59
+ registry: VerbRegistry = field(default_factory=default_registry)
60
+ detector: Detector | None = None
61
+ dry_run: bool = False
62
+ confirm: ConfirmFn = deny_all
63
+ runs_dir: str | Path = "runs"
64
+ run_dir: Path | None = None
65
+ max_steps: int | None = None
66
+ heartbeat_period_s: float = 0.5
67
+ log: Any = lambda _m: None
68
+ on_frame: Any = None
69
+ """Optional callback (img, caption) for a recorder (M2). Called on every captured frame."""
70
+ keep_images_for_last_n: int = 2
71
+
72
+
73
+ @dataclass
74
+ class RunResult:
75
+ outcome: Outcome
76
+ reason: str
77
+ steps: int
78
+ llm_calls: int
79
+ usage: Usage
80
+ run_dir: Path
81
+ final_state: dict[str, Any] = field(default_factory=dict)
82
+ gif_path: Path | None = None
83
+
84
+ @property
85
+ def ok(self) -> bool:
86
+ return self.outcome == "success"
87
+
88
+
89
+ class AgentLoop:
90
+ def __init__(self, cfg: RunConfig) -> None:
91
+ self.cfg = cfg
92
+ self.duck = cfg.duck
93
+ self.fm = cfg.duck.frontmatter
94
+ if cfg.max_steps is not None:
95
+ self.fm = self.fm.model_copy(
96
+ update={"budgets": self.fm.budgets.model_copy(update={"max_steps": cfg.max_steps})}
97
+ )
98
+ self.run_dir = cfg.run_dir or new_run_dir(cfg.runs_dir, self.fm.name)
99
+ self.transcript = Transcript(self.run_dir)
100
+ self.budget = Budget(self.fm.budgets, now=cfg.transport.now)
101
+ self.executor = Executor(
102
+ registry=cfg.registry,
103
+ transport=cfg.transport,
104
+ contract=self.fm,
105
+ budget=self.budget,
106
+ detector=cfg.detector,
107
+ dry_run=cfg.dry_run,
108
+ confirm=cfg.confirm,
109
+ log=cfg.log,
110
+ on_frame=self._on_frame,
111
+ )
112
+ self.heartbeat = Heartbeat(
113
+ cfg.transport, self.executor.abort, period_s=cfg.heartbeat_period_s, log=cfg.log
114
+ )
115
+ self.history: list[Exchange] = []
116
+ self.usage = Usage()
117
+
118
+ # ── frames ──────────────────────────────────────────────────────────────────────
119
+
120
+ def _on_frame(self, img: Image.Image, caption: str) -> None:
121
+ self.transcript.save_frame(img, caption)
122
+ if self.cfg.on_frame is not None:
123
+ self.cfg.on_frame(img, caption)
124
+
125
+ async def _observe(
126
+ self, last_verb: str | None, last_result: VerbResult | None
127
+ ) -> tuple[Observation, Image.Image | None]:
128
+ state = await self.cfg.transport.get_state()
129
+ img = await self.cfg.transport.get_frame()
130
+ detections: list[Detection] = []
131
+ if img is not None:
132
+ if self.cfg.detector is not None:
133
+ detections = self.cfg.detector.detect(img)
134
+ self._on_frame(img, f"step {self.budget.steps}: {last_verb or 'start'}")
135
+ text = build_observation_text(
136
+ step=self.budget.steps,
137
+ max_steps=self.fm.budgets.max_steps,
138
+ state=state,
139
+ detections=detections,
140
+ last_verb=last_verb,
141
+ last_result=last_result,
142
+ budget_status=self.budget.status(),
143
+ )
144
+ features = observation_features(
145
+ state=state,
146
+ detections=detections,
147
+ last_verb=last_verb,
148
+ last_result=last_result,
149
+ allowed=self.executor.allowed,
150
+ )
151
+ image = png_bytes(img) if (img is not None and self.cfg.provider.supports_vision) else None
152
+ return Observation(text=text, image_png=image, features=features), img
153
+
154
+ def _history_for_provider(self) -> list[Exchange]:
155
+ """Older images are dropped to keep context small; the last N keep theirs."""
156
+ n = self.cfg.keep_images_for_last_n
157
+ out: list[Exchange] = []
158
+ for i, ex in enumerate(self.history):
159
+ if ex.observation.image_png is not None and i < len(self.history) - n:
160
+ ex = ex.model_copy(
161
+ update={"observation": ex.observation.model_copy(update={"image_png": None})}
162
+ )
163
+ out.append(ex)
164
+ return out
165
+
166
+ # ── the loop ────────────────────────────────────────────────────────────────────
167
+
168
+ async def run(self) -> RunResult:
169
+ cfg = self.cfg
170
+ tools = cfg.registry.tool_schemas(self.fm.verbs.allow) + META_TOOLS
171
+ system = build_system_prompt(
172
+ self.duck, [cfg.registry.get(n) for n in self.fm.verbs.allow], cfg.transport.name
173
+ )
174
+ self.transcript.write(
175
+ "run_start",
176
+ duck=self.fm.name,
177
+ duck_path=self.duck.path,
178
+ provider=cfg.provider.name,
179
+ model=cfg.provider.model,
180
+ transport=cfg.transport.name,
181
+ dry_run=cfg.dry_run,
182
+ contract=self.fm.model_dump(),
183
+ system_prompt=system,
184
+ tools=[t["name"] for t in tools],
185
+ )
186
+ outcome: Outcome = "error"
187
+ reason = "loop exited unexpectedly"
188
+ last_verb: str | None = None
189
+ last_result: VerbResult | None = None
190
+ retry_prompted = False
191
+
192
+ await cfg.transport.connect()
193
+ self.budget.start()
194
+ self.heartbeat.start()
195
+ try:
196
+ while True:
197
+ await asyncio.sleep(0) # let the heartbeat and kill switch run
198
+ if self.executor.abort.is_set():
199
+ raise Aborted(
200
+ str(self.heartbeat.failure) if self.heartbeat.failure else "kill switch"
201
+ )
202
+ obs, _ = await self._observe(last_verb, last_result)
203
+ if self.history and self.history[-1].decision is not None:
204
+ obs = obs.model_copy(
205
+ update={"tool_call_id": self.history[-1].decision.tool_call.id}
206
+ )
207
+ self.history.append(Exchange(observation=obs))
208
+ self.transcript.write(
209
+ "observation",
210
+ step=self.budget.steps,
211
+ text=obs.text,
212
+ has_image=obs.image_png is not None,
213
+ features=obs.features,
214
+ )
215
+
216
+ self.budget.note_llm_call()
217
+ turn = await cfg.provider.step(system, self._history_for_provider(), tools)
218
+ self.usage = self.usage + turn.usage
219
+ self.transcript.write(
220
+ "llm",
221
+ step=self.budget.steps,
222
+ provider=cfg.provider.name,
223
+ model=cfg.provider.model,
224
+ text=turn.text,
225
+ tool_calls=[tc.model_dump() for tc in turn.tool_calls],
226
+ usage=turn.usage.model_dump(),
227
+ stop_reason=turn.stop_reason,
228
+ )
229
+
230
+ if not turn.tool_calls:
231
+ if not retry_prompted:
232
+ retry_prompted = True
233
+ self.history[-1].decision = None
234
+ self.history.append(
235
+ Exchange(
236
+ observation=Observation(
237
+ text="You must call exactly one tool. Choose now.",
238
+ features=obs.features,
239
+ )
240
+ )
241
+ )
242
+ self.transcript.write(
243
+ "enforce",
244
+ step=self.budget.steps,
245
+ issue="no_tool_call",
246
+ action="re-prompt",
247
+ )
248
+ continue
249
+ outcome, reason = "failure", "the model produced no tool call twice in a row"
250
+ break
251
+ retry_prompted = False
252
+ if len(turn.tool_calls) > 1:
253
+ self.transcript.write(
254
+ "enforce",
255
+ step=self.budget.steps,
256
+ issue="multiple_tool_calls",
257
+ action="first_only",
258
+ )
259
+ call: ToolCall = turn.tool_calls[0]
260
+ self.history[-1].decision = Decision(tool_call=call, text=turn.text, raw=turn.raw)
261
+
262
+ if call.name in META_TOOL_NAMES:
263
+ outcome = "success" if call.name == "declare_success" else "failure"
264
+ reason = str(call.arguments.get("reason", ""))
265
+ self.transcript.write(
266
+ "declare", step=self.budget.steps, outcome=outcome, reason=reason
267
+ )
268
+ break
269
+
270
+ last_verb = call.name
271
+ try:
272
+ last_result = await self.executor.run_verb(
273
+ call.name, call.arguments, source="agent"
274
+ )
275
+ except VerbNotAllowed as e:
276
+ last_result = VerbResult.fail(str(e))
277
+ except ConfirmDenied as e:
278
+ last_result = VerbResult.fail(f"{e}; choose something else or declare_failure")
279
+ self.transcript.write(
280
+ "verb",
281
+ step=self.budget.steps,
282
+ name=call.name,
283
+ params=call.arguments,
284
+ ok=last_result.ok,
285
+ summary=last_result.summary,
286
+ data=last_result.data,
287
+ )
288
+ except BudgetExceeded as e:
289
+ outcome, reason = "budget", str(e)
290
+ except Aborted as e:
291
+ outcome, reason = "aborted", str(e)
292
+ except SafetyStop as e:
293
+ outcome, reason = "aborted", str(e)
294
+ finally:
295
+ await self.heartbeat.stop()
296
+ with contextlib.suppress(Exception):
297
+ await cfg.transport.stop()
298
+ final_state: dict[str, Any] = {}
299
+ with contextlib.suppress(Exception):
300
+ final_state = (await cfg.transport.get_state()).model_dump()
301
+ with contextlib.suppress(Exception):
302
+ await cfg.transport.close()
303
+ summary = {
304
+ "duck": self.fm.name,
305
+ "outcome": outcome,
306
+ "reason": reason,
307
+ "steps": self.budget.steps,
308
+ "llm_calls": self.budget.llm_calls,
309
+ "elapsed_s": round(self.budget.elapsed_s, 2),
310
+ "usage": self.usage.model_dump(),
311
+ "provider": cfg.provider.name,
312
+ "model": cfg.provider.model,
313
+ "transport": cfg.transport.name,
314
+ "dry_run": cfg.dry_run,
315
+ "final_state": final_state,
316
+ }
317
+ self.transcript.write("run_end", **summary)
318
+ self.transcript.write_summary(summary)
319
+ self.transcript.close()
320
+ return RunResult(
321
+ outcome=outcome,
322
+ reason=reason,
323
+ steps=self.budget.steps,
324
+ llm_calls=self.budget.llm_calls,
325
+ usage=self.usage,
326
+ run_dir=self.run_dir,
327
+ final_state=final_state,
328
+ )
329
+
330
+
331
+ async def run_duck(cfg: RunConfig) -> RunResult:
332
+ return await AgentLoop(cfg).run()
@@ -0,0 +1,135 @@
1
+ """Every word the LLM reads, in one place.
2
+
3
+ The system prompt carries the contract (in prose the model can act on) and the `.duck`
4
+ body verbatim. Each turn's observation is compact and structured — features, not frames —
5
+ with the image attached separately for providers that can see. One tool call per turn is
6
+ stated here *and* enforced by the loop; saying it is not the same as trusting it.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ from typing import Any
12
+
13
+ from quackd.duckfile.schema import DuckFile
14
+ from quackd.perception.base import Detection, summarize_detections
15
+ from quackd.transport.base import DuckState
16
+ from quackd.verbs.registry import Verb, VerbResult
17
+
18
+ DECLARE_SUCCESS = {
19
+ "name": "declare_success",
20
+ "description": "Call when the success criteria are met. Say which criterion and what evidence you have.",
21
+ "input_schema": {
22
+ "type": "object",
23
+ "properties": {
24
+ "reason": {
25
+ "type": "string",
26
+ "description": "Which criterion was met, and the evidence.",
27
+ }
28
+ },
29
+ "required": ["reason"],
30
+ "additionalProperties": False,
31
+ },
32
+ }
33
+
34
+ DECLARE_FAILURE = {
35
+ "name": "declare_failure",
36
+ "description": "Call when the task cannot be completed (target not found, repeated failures, an abort condition).",
37
+ "input_schema": {
38
+ "type": "object",
39
+ "properties": {"reason": {"type": "string"}},
40
+ "required": ["reason"],
41
+ "additionalProperties": False,
42
+ },
43
+ }
44
+
45
+ META_TOOLS = [DECLARE_SUCCESS, DECLARE_FAILURE]
46
+ META_TOOL_NAMES = {t["name"] for t in META_TOOLS}
47
+
48
+
49
+ def build_system_prompt(duck: DuckFile, verbs: list[Verb], transport_name: str) -> str:
50
+ fm = duck.frontmatter
51
+ verb_lines = "\n".join(f"- `{v.name}`: {v.description}" for v in verbs)
52
+ success = "\n".join(f"- {s}" for s in fm.success)
53
+ advisory = fm.advisory_abort_conditions
54
+ abort_lines = (
55
+ "\n".join(f"- {a}" for a in advisory) if advisory else "- (none beyond the enforced ones)"
56
+ )
57
+ persona = f"\n## Persona\n{fm.persona}\n" if fm.persona else ""
58
+ sim_note = (
59
+ "\nYou are in the built-in 2D simulator: a cartoon top-down world. Distances are metres, "
60
+ "the arena is about 2 m across, and the ball is orange.\n"
61
+ if transport_name == "sim2d"
62
+ else ""
63
+ )
64
+ return f"""You are the brain of a small biped duck robot (25 cm, 800 g). You are a high-level pilot:
65
+ you choose ONE verb per turn; the robot's own controllers handle balance and gait, and composite
66
+ verbs like `walk_to` close their own loops on the camera. Do not micro-manage.
67
+
68
+ ## Rules (enforced by the executor — not optional)
69
+ - Call exactly one tool per turn. Never zero, never two.
70
+ - Only these verbs are allowed: {", ".join(fm.verbs.allow)}. Anything else is refused.
71
+ - Budgets: {fm.budgets.max_steps} steps, {fm.budgets.max_minutes:g} minutes, {fm.budgets.max_llm_calls} LLM calls. The run stops when any is hit.
72
+ - Verbs marked confirm ({", ".join(fm.verbs.confirm) or "none"}) ask a human before running.
73
+ - When a success criterion is met, call `declare_success`. If the task is impossible, call `declare_failure`.
74
+
75
+ ## Success criteria
76
+ {success}
77
+
78
+ ## Abort conditions you must respect yourself
79
+ {abort_lines}
80
+
81
+ ## Verbs
82
+ {verb_lines}
83
+ {persona}{sim_note}
84
+ ## Task file: {fm.name} — {fm.description}
85
+
86
+ {duck.body}
87
+ """
88
+
89
+
90
+ def build_observation_text(
91
+ *,
92
+ step: int,
93
+ max_steps: int,
94
+ state: DuckState,
95
+ detections: list[Detection],
96
+ last_verb: str | None,
97
+ last_result: VerbResult | None,
98
+ budget_status: str,
99
+ ) -> str:
100
+ lines = [
101
+ f"[step {step}/{max_steps} · {budget_status}]",
102
+ f"state: {state.summary()}",
103
+ f"camera: {summarize_detections(detections)}",
104
+ ]
105
+ if last_verb is not None and last_result is not None:
106
+ lines.append(
107
+ f"last verb `{last_verb}`: {'ok' if last_result.ok else 'FAILED'} — {last_result.summary}"
108
+ )
109
+ lines.append("Choose exactly one tool.")
110
+ return "\n".join(lines)
111
+
112
+
113
+ def observation_features(
114
+ *,
115
+ state: DuckState,
116
+ detections: list[Detection],
117
+ last_verb: str | None,
118
+ last_result: VerbResult | None,
119
+ allowed: list[str],
120
+ ) -> dict[str, Any]:
121
+ return {
122
+ "state": state.model_dump(),
123
+ "detections": [d.model_dump() for d in detections],
124
+ "last_result": (
125
+ {
126
+ "verb": last_verb,
127
+ "ok": last_result.ok,
128
+ "summary": last_result.summary,
129
+ "data": last_result.data,
130
+ }
131
+ if last_result is not None
132
+ else None
133
+ ),
134
+ "allowed": allowed,
135
+ }
@@ -0,0 +1,6 @@
1
+ """One protocol, many vendors.
2
+
3
+ Providers exist so the rest of quackd never imports a vendor SDK. Each one is an optional
4
+ extra, lazily imported; `fake` is always available and is what tests and the zero-API-key
5
+ demo run on.
6
+ """