microduck-cli 0.7.0__tar.gz → 0.9.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/cicd/SKILL.md +41 -14
- microduck_cli-0.9.0/.claude/skills/operate-microduck/SKILL.md +232 -0
- microduck_cli-0.9.0/.claude/skills/validate-delivery/SKILL.md +227 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills.local.yaml.example +5 -0
- microduck_cli-0.9.0/.devague/current +1 -0
- microduck_cli-0.9.0/.devague/current_plan +1 -0
- microduck_cli-0.9.0/.devague/deliveries/microduck-cli-env-teach-operate-rules.json +628 -0
- microduck_cli-0.9.0/.devague/frames/microduck-cli-env-teach-operate-rules.json +1165 -0
- microduck_cli-0.9.0/.devague/plans/microduck-cli-env-teach-operate-rules.json +1210 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.github/workflows/tests.yml +39 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.gitignore +8 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.markdownlint-cli2.yaml +7 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/CHANGELOG.md +54 -0
- microduck_cli-0.9.0/CLAUDE.md +448 -0
- microduck_cli-0.9.0/PKG-INFO +126 -0
- microduck_cli-0.9.0/README.md +109 -0
- microduck_cli-0.9.0/docs/deliveries/2026-09-03-microduck-cli-env-teach-operate-rules.md +183 -0
- microduck_cli-0.9.0/docs/operating-the-duck.md +117 -0
- microduck_cli-0.9.0/docs/plans/2026-09-03-microduck-cli-env-teach-operate-rules.md +264 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/docs/skill-sources.md +81 -7
- microduck_cli-0.9.0/docs/specs/2026-09-03-microduck-cli-env-teach-operate-rules.md +198 -0
- microduck_cli-0.9.0/docs/upstream-pins.md +20 -0
- microduck_cli-0.9.0/docs/verification/2026-09-04-sim-bringup.md +211 -0
- microduck_cli-0.9.0/microduck_cli/behavior/__init__.py +15 -0
- microduck_cli-0.9.0/microduck_cli/behavior/compose.py +709 -0
- microduck_cli-0.9.0/microduck_cli/behavior/default_rules.toml +117 -0
- microduck_cli-0.9.0/microduck_cli/behavior/defaults.py +53 -0
- microduck_cli-0.9.0/microduck_cli/behavior/engine.py +534 -0
- microduck_cli-0.9.0/microduck_cli/behavior/human_gate.py +296 -0
- microduck_cli-0.9.0/microduck_cli/behavior/idle.py +308 -0
- microduck_cli-0.9.0/microduck_cli/behavior/intents.py +774 -0
- microduck_cli-0.9.0/microduck_cli/behavior/liveness.py +300 -0
- microduck_cli-0.9.0/microduck_cli/behavior/model.py +297 -0
- microduck_cli-0.9.0/microduck_cli/behavior/release.py +372 -0
- microduck_cli-0.9.0/microduck_cli/behavior/replay.py +337 -0
- microduck_cli-0.9.0/microduck_cli/behavior/rule_engine.py +379 -0
- microduck_cli-0.9.0/microduck_cli/behavior/rules.py +733 -0
- microduck_cli-0.9.0/microduck_cli/behavior/sense.py +382 -0
- microduck_cli-0.9.0/microduck_cli/behavior/senselog.py +130 -0
- microduck_cli-0.9.0/microduck_cli/behavior/sink.py +522 -0
- microduck_cli-0.9.0/microduck_cli/behavior/skills.py +277 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/microduck_cli/cli/__init__.py +16 -6
- microduck_cli-0.9.0/microduck_cli/cli/_commands/duck.py +1107 -0
- microduck_cli-0.9.0/microduck_cli/cli/_commands/env.py +637 -0
- microduck_cli-0.9.0/microduck_cli/cli/_commands/learn.py +207 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/microduck_cli/cli/_commands/overview.py +11 -8
- microduck_cli-0.9.0/microduck_cli/cli/_commands/policy.py +1013 -0
- microduck_cli-0.9.0/microduck_cli/cli/_commands/rules.py +1345 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/microduck_cli/cli/_output.py +13 -0
- microduck_cli-0.9.0/microduck_cli/duck/__init__.py +1 -0
- microduck_cli-0.9.0/microduck_cli/duck/addressing.py +253 -0
- microduck_cli-0.9.0/microduck_cli/duck/gate.py +218 -0
- microduck_cli-0.9.0/microduck_cli/duck/record.py +382 -0
- microduck_cli-0.9.0/microduck_cli/env/__init__.py +1 -0
- microduck_cli-0.9.0/microduck_cli/env/doctor.py +635 -0
- microduck_cli-0.9.0/microduck_cli/env/hosts.py +252 -0
- microduck_cli-0.9.0/microduck_cli/env/params.py +267 -0
- microduck_cli-0.9.0/microduck_cli/env/stack.py +797 -0
- microduck_cli-0.9.0/microduck_cli/explain/catalog.py +214 -0
- microduck_cli-0.9.0/microduck_cli/explain/duck.py +411 -0
- microduck_cli-0.9.0/microduck_cli/explain/env.py +206 -0
- microduck_cli-0.9.0/microduck_cli/explain/policy.py +497 -0
- microduck_cli-0.9.0/microduck_cli/explain/rules.py +373 -0
- microduck_cli-0.9.0/microduck_cli/ipc/__init__.py +14 -0
- microduck_cli-0.9.0/microduck_cli/ipc/client.py +1006 -0
- microduck_cli-0.9.0/microduck_cli/ipc/proto.py +304 -0
- microduck_cli-0.9.0/microduck_cli/train/__init__.py +8 -0
- microduck_cli-0.9.0/microduck_cli/train/artifacts.py +99 -0
- microduck_cli-0.9.0/microduck_cli/train/lane.py +485 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/pyproject.toml +8 -2
- microduck_cli-0.9.0/tests/fake_robotd.py +1037 -0
- microduck_cli-0.9.0/tests/fixtures/duck_ipc_proto.json +123 -0
- microduck_cli-0.9.0/tests/live/__init__.py +0 -0
- microduck_cli-0.9.0/tests/live/test_live_cli.py +317 -0
- microduck_cli-0.9.0/tests/test_addressing.py +189 -0
- microduck_cli-0.9.0/tests/test_client.py +743 -0
- microduck_cli-0.9.0/tests/test_default_rules.py +128 -0
- microduck_cli-0.9.0/tests/test_duck.py +659 -0
- microduck_cli-0.9.0/tests/test_engine.py +499 -0
- microduck_cli-0.9.0/tests/test_env.py +669 -0
- microduck_cli-0.9.0/tests/test_env_doctor.py +337 -0
- microduck_cli-0.9.0/tests/test_fake_robotd.py +629 -0
- microduck_cli-0.9.0/tests/test_gate.py +189 -0
- microduck_cli-0.9.0/tests/test_hosts.py +155 -0
- microduck_cli-0.9.0/tests/test_human_gate.py +331 -0
- microduck_cli-0.9.0/tests/test_idle.py +267 -0
- microduck_cli-0.9.0/tests/test_intents.py +395 -0
- microduck_cli-0.9.0/tests/test_lane.py +634 -0
- microduck_cli-0.9.0/tests/test_learn_skill.py +106 -0
- microduck_cli-0.9.0/tests/test_links.py +207 -0
- microduck_cli-0.9.0/tests/test_liveness.py +195 -0
- microduck_cli-0.9.0/tests/test_lockstep.py +104 -0
- microduck_cli-0.9.0/tests/test_model.py +288 -0
- microduck_cli-0.9.0/tests/test_no_config_writes.py +199 -0
- microduck_cli-0.9.0/tests/test_no_hardware_paths.py +127 -0
- microduck_cli-0.9.0/tests/test_no_secrets_in_output.py +73 -0
- microduck_cli-0.9.0/tests/test_params.py +141 -0
- microduck_cli-0.9.0/tests/test_policy.py +741 -0
- microduck_cli-0.9.0/tests/test_proto.py +196 -0
- microduck_cli-0.9.0/tests/test_record.py +388 -0
- microduck_cli-0.9.0/tests/test_release.py +302 -0
- microduck_cli-0.9.0/tests/test_replay.py +227 -0
- microduck_cli-0.9.0/tests/test_rule_engine.py +552 -0
- microduck_cli-0.9.0/tests/test_rules.py +500 -0
- microduck_cli-0.9.0/tests/test_rules_cli.py +1147 -0
- microduck_cli-0.9.0/tests/test_sense.py +274 -0
- microduck_cli-0.9.0/tests/test_senselog.py +81 -0
- microduck_cli-0.9.0/tests/test_sink.py +473 -0
- microduck_cli-0.9.0/tests/test_skills.py +293 -0
- microduck_cli-0.9.0/tests/test_stack.py +535 -0
- microduck_cli-0.9.0/tests/test_zero_deps.py +98 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/uv.lock +31 -31
- microduck_cli-0.7.0/CLAUDE.md +0 -28
- microduck_cli-0.7.0/PKG-INFO +0 -76
- microduck_cli-0.7.0/README.md +0 -59
- microduck_cli-0.7.0/microduck_cli/cli/_commands/learn.py +0 -88
- microduck_cli-0.7.0/microduck_cli/explain/catalog.py +0 -130
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/agent-config/SKILL.md +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/agent-config/data/backend-fingerprints.yaml +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/agent-config/scripts/show.sh +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/ask-colleague/SKILL.md +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/ask-colleague/prompts/explore.md +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/ask-colleague/prompts/review.md +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/ask-colleague/prompts/write.md +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/ask-colleague/scripts/ask-colleague.sh +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/assign-to-workforce/SKILL.md +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/assign-to-workforce/scripts/assign-to-workforce.sh +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/challenge/SKILL.md +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/cicd/scripts/_resolve-nick.sh +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/cicd/scripts/portability-lint.sh +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/cicd/scripts/pr-reply.sh +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/cicd/scripts/pr-status.sh +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/cicd/scripts/workflow.sh +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/communicate/SKILL.md +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/communicate/scripts/fetch-issues.sh +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/communicate/scripts/mesh-message.sh +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/communicate/scripts/post-comment.sh +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/communicate/scripts/post-issue.sh +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/communicate/scripts/templates/skill-new-brief.md +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/communicate/scripts/templates/skill-update-brief.md +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/deviate/SKILL.md +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/doc-test-alignment/SKILL.md +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/doc-test-alignment/scripts/check.sh +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/pypi-maintainer/SKILL.md +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/pypi-maintainer/scripts/switch-source.sh +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/recall/SKILL.md +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/recall/scripts/recall.sh +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/remember/SKILL.md +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/remember/scripts/remember.sh +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/run-tests/SKILL.md +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/run-tests/scripts/test.sh +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/scope/SKILL.md +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/sonarclaude/SKILL.md +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/sonarclaude/scripts/sonar.sh +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/spec-to-plan/SKILL.md +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/spec-to-plan/scripts/spec-to-plan.sh +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/summarize-delivery/SKILL.md +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/think/SKILL.md +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/think/scripts/think.sh +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/version-bump/SKILL.md +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.claude/skills/version-bump/scripts/bump.py +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.flake8 +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/.github/workflows/publish.yml +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/AGENTS.colleague.md +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/LICENSE +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/culture.yaml +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/microduck_cli/__init__.py +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/microduck_cli/__main__.py +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/microduck_cli/cli/_commands/__init__.py +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/microduck_cli/cli/_commands/cli.py +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/microduck_cli/cli/_commands/doctor.py +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/microduck_cli/cli/_commands/explain.py +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/microduck_cli/cli/_commands/whoami.py +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/microduck_cli/cli/_errors.py +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/microduck_cli/explain/__init__.py +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/sonar-project.properties +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/tests/__init__.py +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/tests/test_cli.py +0 -0
- {microduck_cli-0.7.0 → microduck_cli-0.9.0}/tests/test_cli_introspection.py +0 -0
|
@@ -163,14 +163,27 @@ only when the user explicitly asks for one of them.
|
|
|
163
163
|
|
|
164
164
|
For every comment, decide **FIX** or **PUSHBACK** with reasoning.
|
|
165
165
|
|
|
166
|
-
Default to **FIX** for: portability complaints (
|
|
167
|
-
|
|
166
|
+
Default to **FIX** for: portability complaints (a recurring bug
|
|
167
|
+
class across AgentCulture repos), test or doc requests, style nits
|
|
168
168
|
aligned with workspace conventions.
|
|
169
169
|
|
|
170
170
|
Default to **PUSHBACK** for: architecture opinions that conflict with
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
171
|
+
`CLAUDE.md` or the all-backends rule; scaffold false-positives — the
|
|
172
|
+
duck layer is still landing, so "this CLI doesn't do anything with the
|
|
173
|
+
robot yet" is a roadmap item, not a PR defect. "The tick engine belongs
|
|
174
|
+
upstream in `neurosymbolic-system`" is also a PUSHBACK: decision c20
|
|
175
|
+
(recorded in `CLAUDE.md`) puts the engine in `microduck_cli/behavior/`
|
|
176
|
+
now, built extraction-first, with `neurosymbolic-system` extracted later
|
|
177
|
+
from reachy-mini-cli plus this engine.
|
|
178
|
+
|
|
179
|
+
What a reviewer should check instead is that the **seams stay pure**, so
|
|
180
|
+
the later extraction is a move and not a rewrite: `TargetSink` and
|
|
181
|
+
`SenseProviders` remain the only ways a pose leaves and a reading enters;
|
|
182
|
+
`tick_seam` remains the ONE per-tick integration point (a second loop or
|
|
183
|
+
a second process IS a defect); rules stay data, not code; there is one
|
|
184
|
+
admission registry; liveness stays a heartbeat, not a flag file; and
|
|
185
|
+
nothing under `behavior/` imports a transport, an SDK, or a CLI module
|
|
186
|
+
beyond `cli/_errors`. A comment naming one of those is a **FIX**.
|
|
174
187
|
|
|
175
188
|
### Alignment-delta rule
|
|
176
189
|
|
|
@@ -179,18 +192,29 @@ If the PR touches `CLAUDE.md`, `culture.yaml`, or anything under
|
|
|
179
192
|
PUSHBACK on each comment. Note any sibling that needs a follow-up PR
|
|
180
193
|
and mention it in your reply.
|
|
181
194
|
|
|
182
|
-
##
|
|
195
|
+
## Pre-PR steps (microduck-cli)
|
|
183
196
|
|
|
184
|
-
|
|
185
|
-
|
|
197
|
+
**Consumer adaptation** — upstream ships this section as conditional
|
|
198
|
+
"greenfield-aware" no-ops. microduck-cli's stack has landed, so the steps
|
|
199
|
+
below are unconditional here; run all of them before `workflow.sh open`.
|
|
200
|
+
Recorded as a tracked divergence in `docs/skill-sources.md`.
|
|
186
201
|
|
|
187
202
|
```bash
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
203
|
+
uv run pytest -n auto # full suite must be green
|
|
204
|
+
uv run black --check microduck_cli tests # the CI lint job, locally
|
|
205
|
+
uv run isort --check-only microduck_cli tests
|
|
206
|
+
uv run flake8 microduck_cli tests
|
|
207
|
+
uv run bandit -c pyproject.toml -r microduck_cli
|
|
208
|
+
uv run teken cli doctor . --strict # the agent-first rubric gate
|
|
209
|
+
markdownlint-cli2 "**/*.md" "#node_modules" "#.local" "#.claude/skills" "#.teken"
|
|
210
|
+
/version-bump patch|minor|major # REQUIRED on every PR
|
|
191
211
|
```
|
|
192
212
|
|
|
193
|
-
|
|
213
|
+
The version bump is not optional: the `version-check` job compares
|
|
214
|
+
`pyproject.toml` against `origin/main` and fails the PR when they match — even
|
|
215
|
+
for docs/config/CI-only changes. Use the `version-bump` skill so `CHANGELOG.md`
|
|
216
|
+
gets its Keep-a-Changelog entry in the same pass.
|
|
217
|
+
|
|
194
218
|
A `pr lint --extra=tests,version,markdown` ask is filed upstream
|
|
195
219
|
([devex#41](https://github.com/agentculture/devex/issues/41)).
|
|
196
220
|
|
|
@@ -203,6 +227,9 @@ in the fix-up commit message.
|
|
|
203
227
|
The `status` extension queries SonarCloud directly (it predates the
|
|
204
228
|
upstream Sonar integration in `devex pr read`). Both surfaces are
|
|
205
229
|
trustworthy — `devex pr read` for display in the briefing, `status` for
|
|
206
|
-
the gate.
|
|
230
|
+
the gate. microduck-cli isn't yet a registered mesh agent, so the
|
|
207
231
|
post-merge IRC ping that Culture's `pr-review` includes is still
|
|
208
|
-
skipped — that returns when
|
|
232
|
+
skipped — that returns when microduck-cli joins the mesh. (Until then,
|
|
233
|
+
a cross-repo follow-up — e.g. a runtime gap that belongs to
|
|
234
|
+
`neurosymbolic-system` — goes out as a tracked issue via the
|
|
235
|
+
`communicate` skill.)
|
|
@@ -0,0 +1,232 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: operate-microduck
|
|
3
|
+
description: >
|
|
4
|
+
Open the MicroDuck simulation with its MuJoCo window, operate the duck —
|
|
5
|
+
stand it up, enable its policy, run skills, turn its head, send motion,
|
|
6
|
+
drive the data-only rules engine, monitor and record its senses — and close
|
|
7
|
+
the simulation down again when the user is done. Drives the `microduck` CLI
|
|
8
|
+
(`env up` / `duck` / `rules` / `env down`) end to end, including the
|
|
9
|
+
screenshot recipe for watching the MuJoCo window from a headless session.
|
|
10
|
+
First-party to microduck-cli — authored here, not vendored. Use when the
|
|
11
|
+
user says "operate the duck", "open the simulation", "start the sim",
|
|
12
|
+
"close the sim", "make the duck stand / walk / look / quack", or asks to
|
|
13
|
+
see what the duck is doing.
|
|
14
|
+
type: command
|
|
15
|
+
---
|
|
16
|
+
|
|
17
|
+
# operate-microduck — open the sim, drive the duck, close it down
|
|
18
|
+
|
|
19
|
+
The skill is named **`operate-microduck`**. It is the operator front door to
|
|
20
|
+
the `microduck` CLI: one lane from a cold box to a standing duck in a MuJoCo
|
|
21
|
+
window, through the verbs that move it, and back down to nothing running.
|
|
22
|
+
|
|
23
|
+
## When to use
|
|
24
|
+
|
|
25
|
+
- The user asks to **open / start the simulation** (with or without a window).
|
|
26
|
+
- The user asks to **make the duck do something** — stand, enable, run a
|
|
27
|
+
skill, look somewhere, move, quack, or react to rules.
|
|
28
|
+
- The user wants to **watch** the duck (a screenshot of the MuJoCo window) or
|
|
29
|
+
**record** what it senses.
|
|
30
|
+
- The user asks to **close the sim** / tear the stack down.
|
|
31
|
+
|
|
32
|
+
Do *not* use it to change the CLI's code — that is ordinary repo work. This
|
|
33
|
+
skill operates the shipped CLI; it does not modify it.
|
|
34
|
+
|
|
35
|
+
## Prerequisites
|
|
36
|
+
|
|
37
|
+
1. **Two sibling clones at the pinned commits** in
|
|
38
|
+
[`docs/upstream-pins.md`](../../../docs/upstream-pins.md):
|
|
39
|
+
`pollen-robotics/microduck` (branch `sim-remote-io`) and
|
|
40
|
+
`pollen-robotics/microduck_rl` (branch `develop`). Re-pinning is a
|
|
41
|
+
deliberate, separate act — never bump a clone to drive the duck.
|
|
42
|
+
2. **`microduck env doctor` must be healthy.** Run it first, every time:
|
|
43
|
+
|
|
44
|
+
```bash
|
|
45
|
+
uv run microduck env doctor
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
It exits `2` and names each failing check's remediation when the box is not
|
|
49
|
+
ready (clones, cargo toolchain, built daemons, the `microduck_rl` venv with
|
|
50
|
+
`onnxruntime`, a short-enough state dir, a free `duck-body` port, host
|
|
51
|
+
class, optional credentials). Do not proceed on a `2`.
|
|
52
|
+
3. **Environment variables** — all optional, all with defaults:
|
|
53
|
+
|
|
54
|
+
| Variable | Default | What it points at |
|
|
55
|
+
|---|---|---|
|
|
56
|
+
| `MICRODUCK_CLONE` | `../microduck` | the `microduck` checkout |
|
|
57
|
+
| `DUCK_SIM_RL` | `../microduck_rl` | the `microduck_rl` checkout |
|
|
58
|
+
| `DUCK_SIM_STATE` | `~/.cache/duck-sim` | pidfiles, sockets, engine state |
|
|
59
|
+
|
|
60
|
+
The operator never types a socket path, a params file or an ORT path — the
|
|
61
|
+
CLI derives all three.
|
|
62
|
+
|
|
63
|
+
## Open the simulation
|
|
64
|
+
|
|
65
|
+
```bash
|
|
66
|
+
uv run microduck env up --sim # MuJoCo body, window on the box's display
|
|
67
|
+
uv run microduck env up --sim --headless # MuJoCo body, no window
|
|
68
|
+
uv run microduck env up --fake # single-process stand-in, for sanity/tests
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
- `--sim` runs the real `duck-body` MuJoCo simulator behind `robotd --sim`
|
|
72
|
+
(120 s health timeout). A **window** appears on the box's display — so when
|
|
73
|
+
you are driving from a headless session, the display owner's `DISPLAY` and
|
|
74
|
+
`XAUTHORITY` must be exported into the command's environment (see
|
|
75
|
+
*Watch it* below), otherwise the body cannot open one.
|
|
76
|
+
- `--headless` is the same body with no window. Use it when nobody is
|
|
77
|
+
watching; it is also the safe default over SSH.
|
|
78
|
+
- `--fake` is `robotd --fake`: no physics, no window, 60 s timeout. Use it for
|
|
79
|
+
a sanity check or when the user just wants the verbs exercised.
|
|
80
|
+
- Add `--skip-build` when the daemons are already built.
|
|
81
|
+
|
|
82
|
+
Then stand the duck up and hand it to its policy:
|
|
83
|
+
|
|
84
|
+
```bash
|
|
85
|
+
uv run microduck duck init --apply # power the joints, ramp to home (~8 s to stand)
|
|
86
|
+
uv run microduck duck enable --apply # hand the robot to its policy — it holds a stance
|
|
87
|
+
```
|
|
88
|
+
|
|
89
|
+
`init` takes roughly **eight seconds** to finish standing. Do not judge the
|
|
90
|
+
duck before it has.
|
|
91
|
+
|
|
92
|
+
## Operate
|
|
93
|
+
|
|
94
|
+
```bash
|
|
95
|
+
uv run microduck duck health # the robot's own verdict; exit 2 when unhealthy
|
|
96
|
+
uv run microduck duck version # API version, daemon version, build revision
|
|
97
|
+
uv run microduck duck monitor --frames 5 --json # N state frames, one JSON object each
|
|
98
|
+
```
|
|
99
|
+
|
|
100
|
+
**Look** — point the head at a trunk-frame point:
|
|
101
|
+
|
|
102
|
+
```bash
|
|
103
|
+
uv run microduck duck look --x 0.2 --y 0.4 --z -0.1 --apply
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
Pick a **visibly large** target. `--x 0.2 --y 0.4 --z -0.1` turns the head
|
|
107
|
+
about 60° — obvious in the window. Small offsets move the head a couple of
|
|
108
|
+
degrees and read as "nothing happened"; if the user asked to see the duck
|
|
109
|
+
look somewhere, use a big target.
|
|
110
|
+
|
|
111
|
+
**Skills** — the one-shot moves the daemon knows:
|
|
112
|
+
|
|
113
|
+
```bash
|
|
114
|
+
uv run microduck rules check --duck duck-a # lists the skills this daemon actually has
|
|
115
|
+
uv run microduck duck do roulade --apply
|
|
116
|
+
```
|
|
117
|
+
|
|
118
|
+
Never guess a skill name. `rules check --duck` reads the live snapshot (from
|
|
119
|
+
`robot.subscribe` on API 16) and tells you what exists.
|
|
120
|
+
|
|
121
|
+
**Move** — drive at intent rate for a duration, then stop:
|
|
122
|
+
|
|
123
|
+
```bash
|
|
124
|
+
uv run microduck duck move --vx 0.15 --duration 3 --apply
|
|
125
|
+
```
|
|
126
|
+
|
|
127
|
+
**Say this before you run it:** at the pinned commits the walk network is
|
|
128
|
+
selected and the twist reaches the daemon, but its joint targets are static —
|
|
129
|
+
**the duck leans, it does not take steps.** Locomotion is *not* achieved at
|
|
130
|
+
this pin; see
|
|
131
|
+
[`docs/verification/2026-09-04-sim-bringup.md`](../../../docs/verification/2026-09-04-sim-bringup.md)
|
|
132
|
+
("Walking in the MuJoCo body — not achieved at this pin"). Promising a walk
|
|
133
|
+
and delivering a lean is the failure mode this paragraph exists to prevent.
|
|
134
|
+
|
|
135
|
+
**Voice and recording:**
|
|
136
|
+
|
|
137
|
+
```bash
|
|
138
|
+
uv run microduck duck quack # this robot's own voice
|
|
139
|
+
uv run microduck duck record > senses.jsonl # pure JSONL on stdout
|
|
140
|
+
```
|
|
141
|
+
|
|
142
|
+
**Rules** — the data-only reaction layer and its 50 Hz tick engine:
|
|
143
|
+
|
|
144
|
+
```bash
|
|
145
|
+
uv run microduck rules list # merged config (shipped + overlay) by origin
|
|
146
|
+
uv run microduck rules check --rules ./my.toml --duck duck-a
|
|
147
|
+
uv run microduck rules intent look # one intent through the ONE registry
|
|
148
|
+
uv run microduck rules engine run --duck duck-a --rules ./my.toml --apply --max-ticks 300
|
|
149
|
+
```
|
|
150
|
+
|
|
151
|
+
`rules engine run` is the **only** process that opens the duck's control
|
|
152
|
+
socket: connect → `hello` → `health` → `init` → `enable` → armed, each step
|
|
153
|
+
logged on stderr. Always bound an operator run with `--max-ticks N` (300 ticks
|
|
154
|
+
≈ 6 s at 50 Hz) unless the user asked for a long-running engine. On any
|
|
155
|
+
abnormal exit it releases what it energized; it never sends `robot.relax`.
|
|
156
|
+
|
|
157
|
+
**The human-driving gate.** While a human is driving — a recent `pad.report`,
|
|
158
|
+
`pad_active`, or `robot.remoteSessionActive` — the engine withholds every
|
|
159
|
+
motion channel it composed. robotd arbitrates nothing between clients, so if
|
|
160
|
+
someone picks up the pad the engine gets out of the way rather than fighting
|
|
161
|
+
for the socket. If the duck "ignores" the engine, check whether a pad is live
|
|
162
|
+
before debugging anything else.
|
|
163
|
+
|
|
164
|
+
**What `--apply` means.** Every verb that moves hardware goes through the same
|
|
165
|
+
gate: on a TTY it asks for confirmation; on a pipe **without** `--apply` it
|
|
166
|
+
prints a zero-side-effect dry-run plan and sends nothing; on a pipe **with**
|
|
167
|
+
`--apply` it proceeds (agent mode). So an agent driving this CLI must pass
|
|
168
|
+
`--apply` deliberately — and a command that printed a plan did *not* move the
|
|
169
|
+
duck.
|
|
170
|
+
|
|
171
|
+
## Watch it
|
|
172
|
+
|
|
173
|
+
The MuJoCo window lives on the box's graphical session, not in your terminal.
|
|
174
|
+
To capture it from a headless session, export the **display owner's** session
|
|
175
|
+
environment into the screenshot command — `DISPLAY`, `XAUTHORITY`, and the
|
|
176
|
+
session `DBUS_SESSION_BUS_ADDRESS` — then take the shot:
|
|
177
|
+
|
|
178
|
+
```bash
|
|
179
|
+
# Discover the display owner's environment rather than hard-coding it:
|
|
180
|
+
# - the X display and cookie: the desktop session's DISPLAY / XAUTHORITY
|
|
181
|
+
# - the session bus: tr '\0' '\n' < /proc/<gnome-shell-pid>/environ \
|
|
182
|
+
# | grep '^DBUS_SESSION_BUS_ADDRESS='
|
|
183
|
+
export DISPLAY=<the session's display>
|
|
184
|
+
export XAUTHORITY=<the session's X authority file>
|
|
185
|
+
export DBUS_SESSION_BUS_ADDRESS=<the gnome-shell session bus>
|
|
186
|
+
gnome-screenshot -f /tmp/duck.png
|
|
187
|
+
```
|
|
188
|
+
|
|
189
|
+
Then read the PNG back. Two shots a few seconds apart are the honest way to
|
|
190
|
+
tell motion from a static pose (this is how the "no gait" finding above was
|
|
191
|
+
made). If no graphical session exists, say so and use `--headless` plus
|
|
192
|
+
`duck monitor --json` instead of pretending to have looked.
|
|
193
|
+
|
|
194
|
+
## Close the simulation
|
|
195
|
+
|
|
196
|
+
```bash
|
|
197
|
+
uv run microduck env down # stop every process env up started
|
|
198
|
+
uv run microduck env status # verify: nothing tracked is alive
|
|
199
|
+
```
|
|
200
|
+
|
|
201
|
+
`env down` reads each pidfile under the state directory, deletes it *before*
|
|
202
|
+
signalling, and signals a pid only when `/proc/<pid>/cmdline` still names the
|
|
203
|
+
binary that pidfile was written for. **Never kill by name** — no `pkill`, no
|
|
204
|
+
`killall`, no `kill $(pgrep robotd)`. Pids are recycled; a name-based kill can
|
|
205
|
+
take out something else entirely.
|
|
206
|
+
|
|
207
|
+
## Hard rules
|
|
208
|
+
|
|
209
|
+
- **Never `relax` a duck someone wants standing without saying what it does.**
|
|
210
|
+
`duck relax` cuts power and the robot *collapses*. It is gated and wants
|
|
211
|
+
`--yes`; if the user asks for it, state the consequence first.
|
|
212
|
+
- **Never send motion without `--apply` on a pipe** and then claim the duck
|
|
213
|
+
moved. A dry-run plan is a plan, not an action — report it as one.
|
|
214
|
+
- **Tear down what you started.** If you brought the stack up for a task, take
|
|
215
|
+
it down when the task ends: `env down`, then `env status` to prove it.
|
|
216
|
+
- **Leave the window up if the user is watching — and say so.** When the user
|
|
217
|
+
is looking at the MuJoCo window, do not run `env down` at the end of a step;
|
|
218
|
+
tell them the stack is still up and how to close it.
|
|
219
|
+
- **Do not smooth over what does not work.** Locomotion at this pin, the API-16
|
|
220
|
+
method gaps (`robot.policies`, `policy.*` answer `-32601`), a missing display
|
|
221
|
+
— report them as they are.
|
|
222
|
+
|
|
223
|
+
## Provenance
|
|
224
|
+
|
|
225
|
+
**First-party to `microduck-cli`.** Origin = this repo; it is *not* vendored
|
|
226
|
+
from guildmaster, and the cite-don't-import rule points the other way for it:
|
|
227
|
+
guildmaster may broadcast this skill to the AgentCulture mesh, and downstream
|
|
228
|
+
consumers cite this copy rather than editing their own. The canonical copy is
|
|
229
|
+
`.claude/skills/operate-microduck/SKILL.md` in `microduck-cli`; an agent in
|
|
230
|
+
another runtime copies it verbatim, or writes it from the recipe
|
|
231
|
+
`microduck learn` prints ("Authoring the operator skill"). Fixes belong here,
|
|
232
|
+
upstream — never patched locally in a consumer.
|
|
@@ -0,0 +1,227 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: validate-delivery
|
|
3
|
+
description: >
|
|
4
|
+
Run the confirmed plan's behavioral tests agent-side after
|
|
5
|
+
assign-to-workforce merges its waves and before summarize-delivery closes
|
|
6
|
+
the loop, then file what was found — evidence for what passed, behavioral
|
|
7
|
+
deltas for what the run added, amended, or removed — as first-class,
|
|
8
|
+
record-only entries via the devague CLI. Never runs the tests inside the
|
|
9
|
+
CLI (issue #20); never suppresses a failing or partial outcome. Use when
|
|
10
|
+
the user says "validate delivery", "run behavioral tests", "check what
|
|
11
|
+
actually behaves", "file evidence", "record a behavioral delta", or after
|
|
12
|
+
assign-to-workforce merges (or fails to merge) a plan's waves and before
|
|
13
|
+
summarize-delivery runs. Authored and maintained in agentculture/devague
|
|
14
|
+
(origin = devague); guildmaster pulls this skill from here and broadcasts
|
|
15
|
+
it to the AgentCulture mesh — it is NOT vendored from guildmaster like the
|
|
16
|
+
inbound skills here.
|
|
17
|
+
type: command
|
|
18
|
+
---
|
|
19
|
+
|
|
20
|
+
# validate-delivery — run behavioral tests, file evidence and deltas
|
|
21
|
+
|
|
22
|
+
The skill is named **`validate-delivery`**; it is the **execution-to-evidence
|
|
23
|
+
leg** of the devague method — the *seventh* leg in flow order (the *eighth*
|
|
24
|
+
origin skill, chronologically), sitting between the two closing execution
|
|
25
|
+
skills:
|
|
26
|
+
|
|
27
|
+
```text
|
|
28
|
+
scope -> think -> challenge -> spec-to-plan -> assign-to-workforce ->
|
|
29
|
+
deviate -> validate-delivery -> summarize-delivery
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
Where `/assign-to-workforce` fans out a converged plan's waves and
|
|
33
|
+
`/summarize-delivery` closes the loop afterward, `/validate-delivery` runs
|
|
34
|
+
**after waves merge and before the delivery summary is written**. It is the
|
|
35
|
+
gap that used to be filled by memory: the confirmed plan's claims are
|
|
36
|
+
obligations, and until this skill existed nothing forced the run to check
|
|
37
|
+
whether the merged code actually behaves as claimed before the summary
|
|
38
|
+
asserted it did.
|
|
39
|
+
|
|
40
|
+
## When to invoke
|
|
41
|
+
|
|
42
|
+
Run this skill once a wave (or the whole plan) has merged and there is
|
|
43
|
+
behavior to check against a claim or an approved deviation — always before
|
|
44
|
+
`/summarize-delivery`, never as a substitute for it. It is not gated on a
|
|
45
|
+
complete run: a partial or failed fan-out is still worth validating for
|
|
46
|
+
whatever did merge.
|
|
47
|
+
|
|
48
|
+
## The method
|
|
49
|
+
|
|
50
|
+
1. **Identify the obligations.** Read the plan's confirmed claims (via the
|
|
51
|
+
frame) and any approved `/deviate` records to see what the run promised —
|
|
52
|
+
an announcement, an after-state, a success signal, an acceptance
|
|
53
|
+
criterion. Each one that has a behavioral test backing it is an
|
|
54
|
+
obligation this leg checks.
|
|
55
|
+
2. **Locate the behavioral tests.** Consuming repos identify behavioral
|
|
56
|
+
tests one of two ways (either is valid; pick whichever the repo already
|
|
57
|
+
uses, and say which one in the filed evidence):
|
|
58
|
+
- a **pytest marker**, e.g. `@pytest.mark.behavioral` — run with
|
|
59
|
+
`pytest -m behavioral`;
|
|
60
|
+
- a **dedicated folder**, e.g. `tests/behavioral/` or
|
|
61
|
+
`behavioral-tests/` — run that path directly.
|
|
62
|
+
3. **Run the tests agent-side.** The agent (not the devague CLI) executes
|
|
63
|
+
the behavioral test suite, or the specific tests relevant to the
|
|
64
|
+
obligations in scope. This is read-only against the codebase — it does
|
|
65
|
+
not modify code to make a test pass.
|
|
66
|
+
4. **File evidence for every obligation checked.** For each obligation, file
|
|
67
|
+
an evidence record naming the obligation met, the test that asserted the
|
|
68
|
+
behavior, and the outcome — `pass` or `fail`. A failing outcome is filed
|
|
69
|
+
exactly like a passing one; it is never omitted or reworded into
|
|
70
|
+
something softer.
|
|
71
|
+
5. **File behavioral deltas for what changed.** When the run's actual
|
|
72
|
+
behavior added, amended, or removed a behavior relative to the plan, file
|
|
73
|
+
a delta record — `added` / `amended` / `removed` — with provenance back
|
|
74
|
+
to the claim or approved deviation that motivated it, and forward to the
|
|
75
|
+
evidence record(s) that back it.
|
|
76
|
+
6. **Report faithfully.** Summarize what was validated, what passed, what
|
|
77
|
+
failed, and what could not be checked at all (no behavioral test exists
|
|
78
|
+
for that obligation yet). An unmet obligation is unmet — it is reported
|
|
79
|
+
as such, not folded into a passing tally or left out of the report.
|
|
80
|
+
7. **Hand off to `/summarize-delivery`.** The filed evidence and deltas feed
|
|
81
|
+
directly into `devague summary`'s Delivery Claims table: evidence
|
|
82
|
+
strength (coverage / fidelity / execution / sensitivity) is the
|
|
83
|
+
confidence vocabulary there, and any approved lapse on a claim caps its
|
|
84
|
+
confidence the same way it always has.
|
|
85
|
+
|
|
86
|
+
## The CLI surface this skill drives
|
|
87
|
+
|
|
88
|
+
**Record-only.** The devague CLI never runs a test itself (issue
|
|
89
|
+
[#20](https://github.com/agentculture/devague/issues/20)) — it only records
|
|
90
|
+
what the agent already ran and found. The exact verb shapes below are
|
|
91
|
+
minimal placeholders while the underlying schema lands in a parallel task;
|
|
92
|
+
treat the verb names as stable and the flags as illustrative, and reconcile
|
|
93
|
+
against `devague explain <move>` once that task merges.
|
|
94
|
+
|
|
95
|
+
| Move | What it records |
|
|
96
|
+
|------|------------------|
|
|
97
|
+
| `devague oblige <cN> --seam "<seam>" --behavior "<behavior>"` | Files a behavioral obligation against a claim, naming the seam to test and the behavior to assert (snapshots the claim text at filing). |
|
|
98
|
+
| `devague evidence --obligation <oN> --test "<ref>" --behavior "<asserted>" --contract "<claim text>" --type <type> --strength <level> --basis "<basis>" --outcome pass\|fail [--run-commit <sha> --run-timestamp <ts>]` | Files an evidence record: obligation met by this test, asserting this behavior, outcome pass or fail (a run reference is required at execution strength and above). `llm`-origin filings land `proposed`; the human adjudicates. |
|
|
99
|
+
| `devague delta --kind added\|amended\|removed --behavior "<what changed>" --caused-by <cN\|dN> [--evidence <eN> ...]` | Files a behavioral delta: provenance back to the claim/deviation it diverges from (`--caused-by`), forward to the evidence that backs it. |
|
|
100
|
+
| `devague summary [--pr] [--json]` | Reads the filed evidence and deltas back into the Delivery Claims table (`/summarize-delivery`'s starting point). |
|
|
101
|
+
|
|
102
|
+
`--origin llm` on `oblige` / `evidence` / `delta` lands the record `proposed`
|
|
103
|
+
— exactly the same anti-fabrication contract as `deviate` and `lapse`: an
|
|
104
|
+
agent's own filing never self-confirms, and only the human's `--confirm` /
|
|
105
|
+
`--reject` moves a proposed record forward. A `user`-origin filing
|
|
106
|
+
auto-approves, mirroring `deviate` and `lapse`.
|
|
107
|
+
|
|
108
|
+
## Hard rules (do not violate)
|
|
109
|
+
|
|
110
|
+
- **The CLI never runs tests.** `devague oblige` / `evidence` / `delta` are
|
|
111
|
+
record-only moves — they take the agent's already-obtained result and
|
|
112
|
+
file it. Running the suite is the agent's job, agent-side, exactly like
|
|
113
|
+
`/summarize-delivery`'s read-only verification step (issue #20).
|
|
114
|
+
- **Unmet is unmet.** A failing or unchecked obligation is filed and
|
|
115
|
+
reported as failing or unchecked — never smoothed into "mostly passing" or
|
|
116
|
+
silently dropped from the report. This is the direct fix for the
|
|
117
|
+
motivating failure below: findings discovered only by reading data after
|
|
118
|
+
the fact, never by a test failing loudly in the record.
|
|
119
|
+
- **A partial or failed run is still a valid input.** There is no
|
|
120
|
+
completion precondition — validate whatever merged, report the rest as
|
|
121
|
+
not yet checkable.
|
|
122
|
+
- **`llm`-origin filings stay proposed until the user confirms.** Same
|
|
123
|
+
anti-fabrication contract as every other origin vocabulary in this
|
|
124
|
+
method — an agent's own proposal never self-confirms.
|
|
125
|
+
- **Provenance both ways.** Every evidence record ties back to an obligation
|
|
126
|
+
(a claim or an approved deviation); every delta ties back to what it
|
|
127
|
+
diverges from and forward to the evidence that backs it. An untraceable
|
|
128
|
+
evidence or delta record is not filed.
|
|
129
|
+
- **This is not a new gate.** Like `/deviate`, `/validate-delivery` does not
|
|
130
|
+
add a fourth standing human gate — it produces the record `/summarize-
|
|
131
|
+
delivery` and the final PR review consume; the three gates (spec,
|
|
132
|
+
implementation split plan, final PR) are unchanged.
|
|
133
|
+
|
|
134
|
+
## Worked example
|
|
135
|
+
|
|
136
|
+
Wave 2 of a plan merged the `export --format widget-md` verb. The plan's
|
|
137
|
+
confirmed `success_signal` claim `c9` said "round-tripping a widget through
|
|
138
|
+
`export` and back loses no fields." A behavioral test exists for it,
|
|
139
|
+
marked `@pytest.mark.behavioral`, plus two more behavioral tests for
|
|
140
|
+
adjacent claims — one of which fails.
|
|
141
|
+
|
|
142
|
+
```bash
|
|
143
|
+
# 1. Identify the obligation (echoes its id, e.g. o1)
|
|
144
|
+
devague oblige c9 --seam "widget export round-trip" \
|
|
145
|
+
--behavior "round-tripping a widget loses no fields"
|
|
146
|
+
|
|
147
|
+
# 2. Locate and run the behavioral tests agent-side (read-only)
|
|
148
|
+
pytest -m behavioral -q
|
|
149
|
+
# -> tests/behavioral/test_widget_export.py::test_round_trip PASSED
|
|
150
|
+
# -> tests/behavioral/test_widget_export.py::test_empty_field_rendering FAILED
|
|
151
|
+
|
|
152
|
+
# 3. File evidence for each outcome — the failure included, not smoothed over
|
|
153
|
+
devague evidence --obligation o1 \
|
|
154
|
+
--test tests/behavioral/test_widget_export.py::test_round_trip \
|
|
155
|
+
--behavior "asserts an exported-then-reimported widget compares equal field by field" \
|
|
156
|
+
--contract "round-tripping a widget loses no fields" \
|
|
157
|
+
--type automated --strength execution \
|
|
158
|
+
--basis "behavioral test ran green at the named commit" \
|
|
159
|
+
--outcome pass --run-commit abc1234 --run-timestamp 2026-08-31T12:00:00
|
|
160
|
+
devague evidence --obligation o2 \
|
|
161
|
+
--test tests/behavioral/test_widget_export.py::test_empty_field_rendering \
|
|
162
|
+
--behavior "asserts an absent widget field renders as an empty line" \
|
|
163
|
+
--contract "an absent field renders honestly, never as filler" \
|
|
164
|
+
--type automated --strength execution \
|
|
165
|
+
--basis "behavioral test ran red at the named commit" \
|
|
166
|
+
--outcome fail --run-commit abc1234 --run-timestamp 2026-08-31T12:00:00
|
|
167
|
+
|
|
168
|
+
# 4. File a delta if the failure reveals a real behavioral divergence
|
|
169
|
+
devague delta --kind amended \
|
|
170
|
+
--behavior "empty widget fields render as garbled text, not an empty line" \
|
|
171
|
+
--caused-by c11 --evidence e2
|
|
172
|
+
|
|
173
|
+
# 5. Report faithfully: c9 is validated; c11's claimed behavior is unmet —
|
|
174
|
+
# say so plainly, hand it to /summarize-delivery as Remaining Work, not
|
|
175
|
+
# as a passing claim.
|
|
176
|
+
```
|
|
177
|
+
|
|
178
|
+
`/summarize-delivery` then reads these back — `c9`'s Delivery Claims row
|
|
179
|
+
cites evidence `e1` at `high` confidence (a passing behavioral test); `c11`'s
|
|
180
|
+
row is `unverified` or explicitly failing, never rounded up.
|
|
181
|
+
|
|
182
|
+
## After validating — hand off to /summarize-delivery
|
|
183
|
+
|
|
184
|
+
Once every obligation in scope has an evidence record (or is reported as not
|
|
185
|
+
yet checkable) and any behavioral deltas are filed, this leg is done — there
|
|
186
|
+
is nothing separate to export, the filed records already live in devague
|
|
187
|
+
state. Continue with the sibling **`/summarize-delivery`** skill: its
|
|
188
|
+
Delivery Claims table reads the evidence and deltas filed here directly
|
|
189
|
+
(`devague summary`), so the confidence a claim carries in the final delivery
|
|
190
|
+
artifact traces back to a test that actually ran, not to memory. Don't stop
|
|
191
|
+
at "tests ran" — the standing flow is **file the evidence, then
|
|
192
|
+
`/summarize-delivery`**.
|
|
193
|
+
|
|
194
|
+
## The motivating record
|
|
195
|
+
|
|
196
|
+
The Reasoning Degradation Ledger (`devague lapse`, issue
|
|
197
|
+
[`agentculture/devague#97`](https://github.com/agentculture/devague/issues/97))
|
|
198
|
+
exists because of this, cited verbatim: "Four graders failed in that
|
|
199
|
+
cycle... Every one was found by reading data afterwards; none by a test
|
|
200
|
+
failing." That gap — a corrections record reconstructed only at the end,
|
|
201
|
+
from memory, because nothing forced a behavioral check to run and be filed
|
|
202
|
+
along the way — is exactly what `/validate-delivery` closes for the
|
|
203
|
+
*execution* side, the same way `/challenge` closes it for the *spec* side.
|
|
204
|
+
|
|
205
|
+
The design itself traces to issue
|
|
206
|
+
[`agentculture/devague#107`](https://github.com/agentculture/devague/issues/107),
|
|
207
|
+
"Suggestion: behavioral validation and a derived current spec," which
|
|
208
|
+
proposed behavior as the primary contract, four evidence types, a strength
|
|
209
|
+
ladder, and the current spec as a projection of a behavior ledger rather
|
|
210
|
+
than a hand-maintained document. This skill is the method-only front door to
|
|
211
|
+
that idea: it does not implement the full ledger or the derived-spec
|
|
212
|
+
projection — it establishes where in the flow behavioral checking happens,
|
|
213
|
+
what gets filed, and how the failure mode #97 documented gets closed instead
|
|
214
|
+
of rediscovered.
|
|
215
|
+
|
|
216
|
+
## Provenance
|
|
217
|
+
|
|
218
|
+
This is a **first-party** skill — its origin is `agentculture/devague`, the
|
|
219
|
+
*eighth* in the outbound family after `/scope`, `/think`, `/challenge`,
|
|
220
|
+
`/spec-to-plan`, `/assign-to-workforce`, `/deviate`, and
|
|
221
|
+
`/summarize-delivery`, covering the execution-to-evidence leg that runs
|
|
222
|
+
after a plan's waves merge and before the delivery summary is written.
|
|
223
|
+
guildmaster pulls it from here and broadcasts it to the AgentCulture mesh;
|
|
224
|
+
because devague is upstream, it is **never re-vendored back** from
|
|
225
|
+
guildmaster's re-broadcast copy. The `cite, don't import` policy still
|
|
226
|
+
holds: downstream repos copy it, they don't symlink or depend on it. See
|
|
227
|
+
`docs/skill-sources.md`.
|
|
@@ -14,3 +14,8 @@ sibling_projects:
|
|
|
14
14
|
- ../guildmaster
|
|
15
15
|
- ../steward
|
|
16
16
|
- ../teken
|
|
17
|
+
# Robot-family siblings — the architecture microduck-cli composes.
|
|
18
|
+
# See CLAUDE.md "The three sibling repos, and what to take from each".
|
|
19
|
+
- ../neurosymbolic-system
|
|
20
|
+
- ../reachy-mini-cli
|
|
21
|
+
- ../arm101-cli
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
microduck-cli-env-teach-operate-rules
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
microduck-cli-env-teach-operate-rules
|