quackd 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- quackd/__init__.py +8 -0
- quackd/agent/__init__.py +6 -0
- quackd/agent/loop.py +332 -0
- quackd/agent/prompts.py +135 -0
- quackd/agent/providers/__init__.py +6 -0
- quackd/agent/providers/anthropic.py +209 -0
- quackd/agent/providers/base.py +103 -0
- quackd/agent/providers/factory.py +63 -0
- quackd/agent/providers/fake.py +141 -0
- quackd/agent/providers/gemini.py +162 -0
- quackd/agent/providers/grok.py +19 -0
- quackd/agent/providers/openai.py +159 -0
- quackd/agent/transcript.py +73 -0
- quackd/cli.py +392 -0
- quackd/doctor.py +135 -0
- quackd/duckfile/__init__.py +6 -0
- quackd/duckfile/export.py +25 -0
- quackd/duckfile/parser.py +134 -0
- quackd/duckfile/schema.json +198 -0
- quackd/duckfile/schema.py +172 -0
- quackd/ducks/fetch.duck +41 -0
- quackd/ducks/find-and-kick.duck +41 -0
- quackd/ducks/follow-me.duck +40 -0
- quackd/ducks/hello-world.duck +33 -0
- quackd/ducks/patrol-and-quack.duck +40 -0
- quackd/mcp_server.py +250 -0
- quackd/perception/__init__.py +6 -0
- quackd/perception/base.py +49 -0
- quackd/perception/color_blob.py +96 -0
- quackd/perception/yolo.py +59 -0
- quackd/safety.py +329 -0
- quackd/sim2d/__init__.py +5 -0
- quackd/sim2d/live.py +39 -0
- quackd/sim2d/recorder.py +80 -0
- quackd/sim2d/render.py +116 -0
- quackd/sim2d/world.py +272 -0
- quackd/transport/__init__.py +6 -0
- quackd/transport/base.py +144 -0
- quackd/transport/factory.py +46 -0
- quackd/transport/jsonrpc_unix.py +313 -0
- quackd/transport/mock.py +93 -0
- quackd/transport/sim2d.py +163 -0
- quackd/transport/upstream_api.py +201 -0
- quackd/transport/websocket_stub.py +70 -0
- quackd/verbs/__init__.py +6 -0
- quackd/verbs/builtin.py +311 -0
- quackd/verbs/composite.py +199 -0
- quackd/verbs/learned.py +65 -0
- quackd/verbs/registry.py +152 -0
- quackd-0.1.0.dist-info/METADATA +433 -0
- quackd-0.1.0.dist-info/RECORD +55 -0
- quackd-0.1.0.dist-info/WHEEL +4 -0
- quackd-0.1.0.dist-info/entry_points.txt +2 -0
- quackd-0.1.0.dist-info/licenses/LICENSE +201 -0
- quackd-0.1.0.dist-info/licenses/NOTICE +29 -0
quackd/__init__.py
ADDED
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
"""quackd — the brain daemon Microduck was missing.
|
|
2
|
+
|
|
3
|
+
This package exists so that any LLM can pilot a Microduck (real or simulated) through a
|
|
4
|
+
small, safety-enforced vocabulary of *verbs*, driven by a `.duck` skill file. Everything else
|
|
5
|
+
in the repo serves that sentence.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
__version__ = "0.1.0"
|
quackd/agent/__init__.py
ADDED
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
"""The deliberation loop: observe → think → enforce → act.
|
|
2
|
+
|
|
3
|
+
This package exists to make the LLM a *high-level* controller and nothing more: one verb
|
|
4
|
+
per turn, judged against the `.duck` success criteria, with every prompt and result written
|
|
5
|
+
to a transcript so a run can be replayed and argued about.
|
|
6
|
+
"""
|
quackd/agent/loop.py
ADDED
|
@@ -0,0 +1,332 @@
|
|
|
1
|
+
"""observe → think → enforce → act, until success, failure, budget, or abort.
|
|
2
|
+
|
|
3
|
+
This is the deliberation loop. It owns nothing clever: perception is a detector, safety is
|
|
4
|
+
the executor, memory is the transcript. What it does own is the *shape* of a turn — one
|
|
5
|
+
observation in, exactly one tool call out — and the honest bookkeeping of why a run ended.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import asyncio
|
|
11
|
+
import contextlib
|
|
12
|
+
from dataclasses import dataclass, field
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
from typing import Any, Literal
|
|
15
|
+
|
|
16
|
+
from PIL import Image
|
|
17
|
+
|
|
18
|
+
from quackd.agent.prompts import (
|
|
19
|
+
META_TOOL_NAMES,
|
|
20
|
+
META_TOOLS,
|
|
21
|
+
build_observation_text,
|
|
22
|
+
build_system_prompt,
|
|
23
|
+
observation_features,
|
|
24
|
+
)
|
|
25
|
+
from quackd.agent.providers.base import (
|
|
26
|
+
Decision,
|
|
27
|
+
Exchange,
|
|
28
|
+
LLMProvider,
|
|
29
|
+
Observation,
|
|
30
|
+
ToolCall,
|
|
31
|
+
Usage,
|
|
32
|
+
)
|
|
33
|
+
from quackd.agent.transcript import Transcript, new_run_dir, png_bytes
|
|
34
|
+
from quackd.duckfile.schema import DuckFile
|
|
35
|
+
from quackd.perception.base import Detection, Detector
|
|
36
|
+
from quackd.safety import (
|
|
37
|
+
Aborted,
|
|
38
|
+
Budget,
|
|
39
|
+
BudgetExceeded,
|
|
40
|
+
ConfirmDenied,
|
|
41
|
+
ConfirmFn,
|
|
42
|
+
Executor,
|
|
43
|
+
Heartbeat,
|
|
44
|
+
SafetyStop,
|
|
45
|
+
VerbNotAllowed,
|
|
46
|
+
deny_all,
|
|
47
|
+
)
|
|
48
|
+
from quackd.transport.base import DuckTransport
|
|
49
|
+
from quackd.verbs.registry import VerbRegistry, VerbResult, default_registry
|
|
50
|
+
|
|
51
|
+
Outcome = Literal["success", "failure", "budget", "aborted", "error"]
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
@dataclass
|
|
55
|
+
class RunConfig:
|
|
56
|
+
duck: DuckFile
|
|
57
|
+
provider: LLMProvider
|
|
58
|
+
transport: DuckTransport
|
|
59
|
+
registry: VerbRegistry = field(default_factory=default_registry)
|
|
60
|
+
detector: Detector | None = None
|
|
61
|
+
dry_run: bool = False
|
|
62
|
+
confirm: ConfirmFn = deny_all
|
|
63
|
+
runs_dir: str | Path = "runs"
|
|
64
|
+
run_dir: Path | None = None
|
|
65
|
+
max_steps: int | None = None
|
|
66
|
+
heartbeat_period_s: float = 0.5
|
|
67
|
+
log: Any = lambda _m: None
|
|
68
|
+
on_frame: Any = None
|
|
69
|
+
"""Optional callback (img, caption) for a recorder (M2). Called on every captured frame."""
|
|
70
|
+
keep_images_for_last_n: int = 2
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
@dataclass
|
|
74
|
+
class RunResult:
|
|
75
|
+
outcome: Outcome
|
|
76
|
+
reason: str
|
|
77
|
+
steps: int
|
|
78
|
+
llm_calls: int
|
|
79
|
+
usage: Usage
|
|
80
|
+
run_dir: Path
|
|
81
|
+
final_state: dict[str, Any] = field(default_factory=dict)
|
|
82
|
+
gif_path: Path | None = None
|
|
83
|
+
|
|
84
|
+
@property
|
|
85
|
+
def ok(self) -> bool:
|
|
86
|
+
return self.outcome == "success"
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
class AgentLoop:
|
|
90
|
+
def __init__(self, cfg: RunConfig) -> None:
|
|
91
|
+
self.cfg = cfg
|
|
92
|
+
self.duck = cfg.duck
|
|
93
|
+
self.fm = cfg.duck.frontmatter
|
|
94
|
+
if cfg.max_steps is not None:
|
|
95
|
+
self.fm = self.fm.model_copy(
|
|
96
|
+
update={"budgets": self.fm.budgets.model_copy(update={"max_steps": cfg.max_steps})}
|
|
97
|
+
)
|
|
98
|
+
self.run_dir = cfg.run_dir or new_run_dir(cfg.runs_dir, self.fm.name)
|
|
99
|
+
self.transcript = Transcript(self.run_dir)
|
|
100
|
+
self.budget = Budget(self.fm.budgets, now=cfg.transport.now)
|
|
101
|
+
self.executor = Executor(
|
|
102
|
+
registry=cfg.registry,
|
|
103
|
+
transport=cfg.transport,
|
|
104
|
+
contract=self.fm,
|
|
105
|
+
budget=self.budget,
|
|
106
|
+
detector=cfg.detector,
|
|
107
|
+
dry_run=cfg.dry_run,
|
|
108
|
+
confirm=cfg.confirm,
|
|
109
|
+
log=cfg.log,
|
|
110
|
+
on_frame=self._on_frame,
|
|
111
|
+
)
|
|
112
|
+
self.heartbeat = Heartbeat(
|
|
113
|
+
cfg.transport, self.executor.abort, period_s=cfg.heartbeat_period_s, log=cfg.log
|
|
114
|
+
)
|
|
115
|
+
self.history: list[Exchange] = []
|
|
116
|
+
self.usage = Usage()
|
|
117
|
+
|
|
118
|
+
# ── frames ──────────────────────────────────────────────────────────────────────
|
|
119
|
+
|
|
120
|
+
def _on_frame(self, img: Image.Image, caption: str) -> None:
|
|
121
|
+
self.transcript.save_frame(img, caption)
|
|
122
|
+
if self.cfg.on_frame is not None:
|
|
123
|
+
self.cfg.on_frame(img, caption)
|
|
124
|
+
|
|
125
|
+
async def _observe(
|
|
126
|
+
self, last_verb: str | None, last_result: VerbResult | None
|
|
127
|
+
) -> tuple[Observation, Image.Image | None]:
|
|
128
|
+
state = await self.cfg.transport.get_state()
|
|
129
|
+
img = await self.cfg.transport.get_frame()
|
|
130
|
+
detections: list[Detection] = []
|
|
131
|
+
if img is not None:
|
|
132
|
+
if self.cfg.detector is not None:
|
|
133
|
+
detections = self.cfg.detector.detect(img)
|
|
134
|
+
self._on_frame(img, f"step {self.budget.steps}: {last_verb or 'start'}")
|
|
135
|
+
text = build_observation_text(
|
|
136
|
+
step=self.budget.steps,
|
|
137
|
+
max_steps=self.fm.budgets.max_steps,
|
|
138
|
+
state=state,
|
|
139
|
+
detections=detections,
|
|
140
|
+
last_verb=last_verb,
|
|
141
|
+
last_result=last_result,
|
|
142
|
+
budget_status=self.budget.status(),
|
|
143
|
+
)
|
|
144
|
+
features = observation_features(
|
|
145
|
+
state=state,
|
|
146
|
+
detections=detections,
|
|
147
|
+
last_verb=last_verb,
|
|
148
|
+
last_result=last_result,
|
|
149
|
+
allowed=self.executor.allowed,
|
|
150
|
+
)
|
|
151
|
+
image = png_bytes(img) if (img is not None and self.cfg.provider.supports_vision) else None
|
|
152
|
+
return Observation(text=text, image_png=image, features=features), img
|
|
153
|
+
|
|
154
|
+
def _history_for_provider(self) -> list[Exchange]:
|
|
155
|
+
"""Older images are dropped to keep context small; the last N keep theirs."""
|
|
156
|
+
n = self.cfg.keep_images_for_last_n
|
|
157
|
+
out: list[Exchange] = []
|
|
158
|
+
for i, ex in enumerate(self.history):
|
|
159
|
+
if ex.observation.image_png is not None and i < len(self.history) - n:
|
|
160
|
+
ex = ex.model_copy(
|
|
161
|
+
update={"observation": ex.observation.model_copy(update={"image_png": None})}
|
|
162
|
+
)
|
|
163
|
+
out.append(ex)
|
|
164
|
+
return out
|
|
165
|
+
|
|
166
|
+
# ── the loop ────────────────────────────────────────────────────────────────────
|
|
167
|
+
|
|
168
|
+
async def run(self) -> RunResult:
|
|
169
|
+
cfg = self.cfg
|
|
170
|
+
tools = cfg.registry.tool_schemas(self.fm.verbs.allow) + META_TOOLS
|
|
171
|
+
system = build_system_prompt(
|
|
172
|
+
self.duck, [cfg.registry.get(n) for n in self.fm.verbs.allow], cfg.transport.name
|
|
173
|
+
)
|
|
174
|
+
self.transcript.write(
|
|
175
|
+
"run_start",
|
|
176
|
+
duck=self.fm.name,
|
|
177
|
+
duck_path=self.duck.path,
|
|
178
|
+
provider=cfg.provider.name,
|
|
179
|
+
model=cfg.provider.model,
|
|
180
|
+
transport=cfg.transport.name,
|
|
181
|
+
dry_run=cfg.dry_run,
|
|
182
|
+
contract=self.fm.model_dump(),
|
|
183
|
+
system_prompt=system,
|
|
184
|
+
tools=[t["name"] for t in tools],
|
|
185
|
+
)
|
|
186
|
+
outcome: Outcome = "error"
|
|
187
|
+
reason = "loop exited unexpectedly"
|
|
188
|
+
last_verb: str | None = None
|
|
189
|
+
last_result: VerbResult | None = None
|
|
190
|
+
retry_prompted = False
|
|
191
|
+
|
|
192
|
+
await cfg.transport.connect()
|
|
193
|
+
self.budget.start()
|
|
194
|
+
self.heartbeat.start()
|
|
195
|
+
try:
|
|
196
|
+
while True:
|
|
197
|
+
await asyncio.sleep(0) # let the heartbeat and kill switch run
|
|
198
|
+
if self.executor.abort.is_set():
|
|
199
|
+
raise Aborted(
|
|
200
|
+
str(self.heartbeat.failure) if self.heartbeat.failure else "kill switch"
|
|
201
|
+
)
|
|
202
|
+
obs, _ = await self._observe(last_verb, last_result)
|
|
203
|
+
if self.history and self.history[-1].decision is not None:
|
|
204
|
+
obs = obs.model_copy(
|
|
205
|
+
update={"tool_call_id": self.history[-1].decision.tool_call.id}
|
|
206
|
+
)
|
|
207
|
+
self.history.append(Exchange(observation=obs))
|
|
208
|
+
self.transcript.write(
|
|
209
|
+
"observation",
|
|
210
|
+
step=self.budget.steps,
|
|
211
|
+
text=obs.text,
|
|
212
|
+
has_image=obs.image_png is not None,
|
|
213
|
+
features=obs.features,
|
|
214
|
+
)
|
|
215
|
+
|
|
216
|
+
self.budget.note_llm_call()
|
|
217
|
+
turn = await cfg.provider.step(system, self._history_for_provider(), tools)
|
|
218
|
+
self.usage = self.usage + turn.usage
|
|
219
|
+
self.transcript.write(
|
|
220
|
+
"llm",
|
|
221
|
+
step=self.budget.steps,
|
|
222
|
+
provider=cfg.provider.name,
|
|
223
|
+
model=cfg.provider.model,
|
|
224
|
+
text=turn.text,
|
|
225
|
+
tool_calls=[tc.model_dump() for tc in turn.tool_calls],
|
|
226
|
+
usage=turn.usage.model_dump(),
|
|
227
|
+
stop_reason=turn.stop_reason,
|
|
228
|
+
)
|
|
229
|
+
|
|
230
|
+
if not turn.tool_calls:
|
|
231
|
+
if not retry_prompted:
|
|
232
|
+
retry_prompted = True
|
|
233
|
+
self.history[-1].decision = None
|
|
234
|
+
self.history.append(
|
|
235
|
+
Exchange(
|
|
236
|
+
observation=Observation(
|
|
237
|
+
text="You must call exactly one tool. Choose now.",
|
|
238
|
+
features=obs.features,
|
|
239
|
+
)
|
|
240
|
+
)
|
|
241
|
+
)
|
|
242
|
+
self.transcript.write(
|
|
243
|
+
"enforce",
|
|
244
|
+
step=self.budget.steps,
|
|
245
|
+
issue="no_tool_call",
|
|
246
|
+
action="re-prompt",
|
|
247
|
+
)
|
|
248
|
+
continue
|
|
249
|
+
outcome, reason = "failure", "the model produced no tool call twice in a row"
|
|
250
|
+
break
|
|
251
|
+
retry_prompted = False
|
|
252
|
+
if len(turn.tool_calls) > 1:
|
|
253
|
+
self.transcript.write(
|
|
254
|
+
"enforce",
|
|
255
|
+
step=self.budget.steps,
|
|
256
|
+
issue="multiple_tool_calls",
|
|
257
|
+
action="first_only",
|
|
258
|
+
)
|
|
259
|
+
call: ToolCall = turn.tool_calls[0]
|
|
260
|
+
self.history[-1].decision = Decision(tool_call=call, text=turn.text, raw=turn.raw)
|
|
261
|
+
|
|
262
|
+
if call.name in META_TOOL_NAMES:
|
|
263
|
+
outcome = "success" if call.name == "declare_success" else "failure"
|
|
264
|
+
reason = str(call.arguments.get("reason", ""))
|
|
265
|
+
self.transcript.write(
|
|
266
|
+
"declare", step=self.budget.steps, outcome=outcome, reason=reason
|
|
267
|
+
)
|
|
268
|
+
break
|
|
269
|
+
|
|
270
|
+
last_verb = call.name
|
|
271
|
+
try:
|
|
272
|
+
last_result = await self.executor.run_verb(
|
|
273
|
+
call.name, call.arguments, source="agent"
|
|
274
|
+
)
|
|
275
|
+
except VerbNotAllowed as e:
|
|
276
|
+
last_result = VerbResult.fail(str(e))
|
|
277
|
+
except ConfirmDenied as e:
|
|
278
|
+
last_result = VerbResult.fail(f"{e}; choose something else or declare_failure")
|
|
279
|
+
self.transcript.write(
|
|
280
|
+
"verb",
|
|
281
|
+
step=self.budget.steps,
|
|
282
|
+
name=call.name,
|
|
283
|
+
params=call.arguments,
|
|
284
|
+
ok=last_result.ok,
|
|
285
|
+
summary=last_result.summary,
|
|
286
|
+
data=last_result.data,
|
|
287
|
+
)
|
|
288
|
+
except BudgetExceeded as e:
|
|
289
|
+
outcome, reason = "budget", str(e)
|
|
290
|
+
except Aborted as e:
|
|
291
|
+
outcome, reason = "aborted", str(e)
|
|
292
|
+
except SafetyStop as e:
|
|
293
|
+
outcome, reason = "aborted", str(e)
|
|
294
|
+
finally:
|
|
295
|
+
await self.heartbeat.stop()
|
|
296
|
+
with contextlib.suppress(Exception):
|
|
297
|
+
await cfg.transport.stop()
|
|
298
|
+
final_state: dict[str, Any] = {}
|
|
299
|
+
with contextlib.suppress(Exception):
|
|
300
|
+
final_state = (await cfg.transport.get_state()).model_dump()
|
|
301
|
+
with contextlib.suppress(Exception):
|
|
302
|
+
await cfg.transport.close()
|
|
303
|
+
summary = {
|
|
304
|
+
"duck": self.fm.name,
|
|
305
|
+
"outcome": outcome,
|
|
306
|
+
"reason": reason,
|
|
307
|
+
"steps": self.budget.steps,
|
|
308
|
+
"llm_calls": self.budget.llm_calls,
|
|
309
|
+
"elapsed_s": round(self.budget.elapsed_s, 2),
|
|
310
|
+
"usage": self.usage.model_dump(),
|
|
311
|
+
"provider": cfg.provider.name,
|
|
312
|
+
"model": cfg.provider.model,
|
|
313
|
+
"transport": cfg.transport.name,
|
|
314
|
+
"dry_run": cfg.dry_run,
|
|
315
|
+
"final_state": final_state,
|
|
316
|
+
}
|
|
317
|
+
self.transcript.write("run_end", **summary)
|
|
318
|
+
self.transcript.write_summary(summary)
|
|
319
|
+
self.transcript.close()
|
|
320
|
+
return RunResult(
|
|
321
|
+
outcome=outcome,
|
|
322
|
+
reason=reason,
|
|
323
|
+
steps=self.budget.steps,
|
|
324
|
+
llm_calls=self.budget.llm_calls,
|
|
325
|
+
usage=self.usage,
|
|
326
|
+
run_dir=self.run_dir,
|
|
327
|
+
final_state=final_state,
|
|
328
|
+
)
|
|
329
|
+
|
|
330
|
+
|
|
331
|
+
async def run_duck(cfg: RunConfig) -> RunResult:
|
|
332
|
+
return await AgentLoop(cfg).run()
|
quackd/agent/prompts.py
ADDED
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
"""Every word the LLM reads, in one place.
|
|
2
|
+
|
|
3
|
+
The system prompt carries the contract (in prose the model can act on) and the `.duck`
|
|
4
|
+
body verbatim. Each turn's observation is compact and structured — features, not frames —
|
|
5
|
+
with the image attached separately for providers that can see. One tool call per turn is
|
|
6
|
+
stated here *and* enforced by the loop; saying it is not the same as trusting it.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from typing import Any
|
|
12
|
+
|
|
13
|
+
from quackd.duckfile.schema import DuckFile
|
|
14
|
+
from quackd.perception.base import Detection, summarize_detections
|
|
15
|
+
from quackd.transport.base import DuckState
|
|
16
|
+
from quackd.verbs.registry import Verb, VerbResult
|
|
17
|
+
|
|
18
|
+
DECLARE_SUCCESS = {
|
|
19
|
+
"name": "declare_success",
|
|
20
|
+
"description": "Call when the success criteria are met. Say which criterion and what evidence you have.",
|
|
21
|
+
"input_schema": {
|
|
22
|
+
"type": "object",
|
|
23
|
+
"properties": {
|
|
24
|
+
"reason": {
|
|
25
|
+
"type": "string",
|
|
26
|
+
"description": "Which criterion was met, and the evidence.",
|
|
27
|
+
}
|
|
28
|
+
},
|
|
29
|
+
"required": ["reason"],
|
|
30
|
+
"additionalProperties": False,
|
|
31
|
+
},
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
DECLARE_FAILURE = {
|
|
35
|
+
"name": "declare_failure",
|
|
36
|
+
"description": "Call when the task cannot be completed (target not found, repeated failures, an abort condition).",
|
|
37
|
+
"input_schema": {
|
|
38
|
+
"type": "object",
|
|
39
|
+
"properties": {"reason": {"type": "string"}},
|
|
40
|
+
"required": ["reason"],
|
|
41
|
+
"additionalProperties": False,
|
|
42
|
+
},
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
META_TOOLS = [DECLARE_SUCCESS, DECLARE_FAILURE]
|
|
46
|
+
META_TOOL_NAMES = {t["name"] for t in META_TOOLS}
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def build_system_prompt(duck: DuckFile, verbs: list[Verb], transport_name: str) -> str:
|
|
50
|
+
fm = duck.frontmatter
|
|
51
|
+
verb_lines = "\n".join(f"- `{v.name}`: {v.description}" for v in verbs)
|
|
52
|
+
success = "\n".join(f"- {s}" for s in fm.success)
|
|
53
|
+
advisory = fm.advisory_abort_conditions
|
|
54
|
+
abort_lines = (
|
|
55
|
+
"\n".join(f"- {a}" for a in advisory) if advisory else "- (none beyond the enforced ones)"
|
|
56
|
+
)
|
|
57
|
+
persona = f"\n## Persona\n{fm.persona}\n" if fm.persona else ""
|
|
58
|
+
sim_note = (
|
|
59
|
+
"\nYou are in the built-in 2D simulator: a cartoon top-down world. Distances are metres, "
|
|
60
|
+
"the arena is about 2 m across, and the ball is orange.\n"
|
|
61
|
+
if transport_name == "sim2d"
|
|
62
|
+
else ""
|
|
63
|
+
)
|
|
64
|
+
return f"""You are the brain of a small biped duck robot (25 cm, 800 g). You are a high-level pilot:
|
|
65
|
+
you choose ONE verb per turn; the robot's own controllers handle balance and gait, and composite
|
|
66
|
+
verbs like `walk_to` close their own loops on the camera. Do not micro-manage.
|
|
67
|
+
|
|
68
|
+
## Rules (enforced by the executor — not optional)
|
|
69
|
+
- Call exactly one tool per turn. Never zero, never two.
|
|
70
|
+
- Only these verbs are allowed: {", ".join(fm.verbs.allow)}. Anything else is refused.
|
|
71
|
+
- Budgets: {fm.budgets.max_steps} steps, {fm.budgets.max_minutes:g} minutes, {fm.budgets.max_llm_calls} LLM calls. The run stops when any is hit.
|
|
72
|
+
- Verbs marked confirm ({", ".join(fm.verbs.confirm) or "none"}) ask a human before running.
|
|
73
|
+
- When a success criterion is met, call `declare_success`. If the task is impossible, call `declare_failure`.
|
|
74
|
+
|
|
75
|
+
## Success criteria
|
|
76
|
+
{success}
|
|
77
|
+
|
|
78
|
+
## Abort conditions you must respect yourself
|
|
79
|
+
{abort_lines}
|
|
80
|
+
|
|
81
|
+
## Verbs
|
|
82
|
+
{verb_lines}
|
|
83
|
+
{persona}{sim_note}
|
|
84
|
+
## Task file: {fm.name} — {fm.description}
|
|
85
|
+
|
|
86
|
+
{duck.body}
|
|
87
|
+
"""
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def build_observation_text(
|
|
91
|
+
*,
|
|
92
|
+
step: int,
|
|
93
|
+
max_steps: int,
|
|
94
|
+
state: DuckState,
|
|
95
|
+
detections: list[Detection],
|
|
96
|
+
last_verb: str | None,
|
|
97
|
+
last_result: VerbResult | None,
|
|
98
|
+
budget_status: str,
|
|
99
|
+
) -> str:
|
|
100
|
+
lines = [
|
|
101
|
+
f"[step {step}/{max_steps} · {budget_status}]",
|
|
102
|
+
f"state: {state.summary()}",
|
|
103
|
+
f"camera: {summarize_detections(detections)}",
|
|
104
|
+
]
|
|
105
|
+
if last_verb is not None and last_result is not None:
|
|
106
|
+
lines.append(
|
|
107
|
+
f"last verb `{last_verb}`: {'ok' if last_result.ok else 'FAILED'} — {last_result.summary}"
|
|
108
|
+
)
|
|
109
|
+
lines.append("Choose exactly one tool.")
|
|
110
|
+
return "\n".join(lines)
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def observation_features(
|
|
114
|
+
*,
|
|
115
|
+
state: DuckState,
|
|
116
|
+
detections: list[Detection],
|
|
117
|
+
last_verb: str | None,
|
|
118
|
+
last_result: VerbResult | None,
|
|
119
|
+
allowed: list[str],
|
|
120
|
+
) -> dict[str, Any]:
|
|
121
|
+
return {
|
|
122
|
+
"state": state.model_dump(),
|
|
123
|
+
"detections": [d.model_dump() for d in detections],
|
|
124
|
+
"last_result": (
|
|
125
|
+
{
|
|
126
|
+
"verb": last_verb,
|
|
127
|
+
"ok": last_result.ok,
|
|
128
|
+
"summary": last_result.summary,
|
|
129
|
+
"data": last_result.data,
|
|
130
|
+
}
|
|
131
|
+
if last_result is not None
|
|
132
|
+
else None
|
|
133
|
+
),
|
|
134
|
+
"allowed": allowed,
|
|
135
|
+
}
|