dsh-embedded-workbench 0.8.12 → 0.9.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.en-US.md +32 -9
- package/README.md +32 -9
- package/cordis.patch.yml +4 -0
- package/lib/client.js +214 -0
- package/lib/index.js +242 -211
- package/lib/types/index.d.ts +59 -51
- package/package.json +52 -33
- package/skills/embedded-workbench/SKILL.md +63 -78
- package/skills/embedded-workbench/references/platform-tool-mapping.md +59 -22
- package/skills/embedded-workbench/references/proactive-suggestions.md +27 -0
- package/src/client.js +214 -0
- package/src/index.ts +79 -35
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "dsh-embedded-workbench",
|
|
3
|
-
"version": "0.
|
|
4
|
-
"description": "Embedded C/C++ firmware development toolbox — 4 agents, 8 skills covering FreeRTOS, ISR, NVM storage, Keil MDK, ARMCLANG, HardFault, state machines, architecture, LVGL patterns, and claim fact-checking. Ships a native DeepSeek Harness (dsh) bundle that
|
|
3
|
+
"version": "0.9.1",
|
|
4
|
+
"description": "Embedded C/C++ firmware development toolbox — 4 agents, 8 skills covering FreeRTOS, ISR, NVM storage, Keil MDK, ARMCLANG, HardFault, state machines, architecture, LVGL patterns, and claim fact-checking. Ships a native DeepSeek Harness (dsh) bundle that folds a short gate — the Plan Verification Gate plus a context-budget rule — into the first model step.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "lib/index.js",
|
|
7
7
|
"types": "lib/types/index.d.ts",
|
|
@@ -11,6 +11,7 @@
|
|
|
11
11
|
"default": "./lib/index.js"
|
|
12
12
|
},
|
|
13
13
|
"./cordis.patch.yml": "./cordis.patch.yml",
|
|
14
|
+
"./client": "./lib/client.js",
|
|
14
15
|
"./package.json": "./package.json"
|
|
15
16
|
},
|
|
16
17
|
"files": [
|
|
@@ -31,48 +32,63 @@
|
|
|
31
32
|
"bundle": {
|
|
32
33
|
"patch": "./cordis.patch.yml"
|
|
33
34
|
},
|
|
35
|
+
"client": {
|
|
36
|
+
"platform": "web",
|
|
37
|
+
"inject": [
|
|
38
|
+
"@deepseek-ai/dsh-client-locale",
|
|
39
|
+
"@deepseek-ai/dsh-client-ui-settings",
|
|
40
|
+
"@deepseek-ai/dsh-client-ui-plugin-manager"
|
|
41
|
+
]
|
|
42
|
+
},
|
|
34
43
|
"compatibility": {
|
|
35
|
-
"dsh": "^0.1.0-rc.7 || ^0.1.1-rc.1 || ^0.1.2-alpha.2 || ^0.1.2-alpha.3 || ^0.1.2-alpha.4 || ^0.1.2-alpha.5 || ^0.1.2-rc.1 || ^0.1.3-alpha.1 || ^0.1.3-alpha.2 || ^0.1.5-alpha.1 || ^0.1.5-rc.1 || ^0.1.5-alpha.2 || ^0.1.5-rc.2 || ^0.1.6-alpha.1 || ^0.1.6-alpha.2 || ^0.1.7-alpha.1",
|
|
44
|
+
"dsh": "^0.1.0-rc.7 || ^0.1.1-rc.1 || ^0.1.2-alpha.2 || ^0.1.2-alpha.3 || ^0.1.2-alpha.4 || ^0.1.2-alpha.5 || ^0.1.2-rc.1 || ^0.1.3-alpha.1 || ^0.1.3-alpha.2 || ^0.1.5-alpha.1 || ^0.1.5-rc.1 || ^0.1.5-alpha.2 || ^0.1.5-rc.2 || ^0.1.5-rc.3 || ^0.1.6-alpha.1 || ^0.1.6-alpha.2 || ^0.1.7-alpha.1 || ^0.1.7-alpha.2 || ^0.1.7-rc.1 || ^0.1.7-rc.2 || ^0.2.0-rc.1",
|
|
36
45
|
"dshReleases": {
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
46
|
+
"0.1.0-rc.7": "compatible",
|
|
47
|
+
"0.1.0-rc.8": "compatible",
|
|
48
|
+
"0.1.1-rc.1": "compatible",
|
|
49
|
+
"0.1.1-rc.2": "compatible",
|
|
50
|
+
"0.1.2-alpha.2": "compatible",
|
|
51
|
+
"0.1.2-alpha.3": "compatible",
|
|
52
|
+
"0.1.2-alpha.4": "compatible",
|
|
53
|
+
"0.1.2-alpha.5": "compatible",
|
|
54
|
+
"0.1.2-rc.1": "compatible",
|
|
55
|
+
"0.1.3-alpha.1": "compatible",
|
|
56
|
+
"0.1.3-alpha.2": "compatible",
|
|
57
|
+
"0.1.5-alpha.1": "compatible",
|
|
58
|
+
"0.1.5-alpha.2": "compatible",
|
|
59
|
+
"0.1.5-rc.1": "compatible",
|
|
60
|
+
"0.1.5-rc.2": "compatible",
|
|
61
|
+
"0.1.5-rc.3": "compatible",
|
|
62
|
+
"0.1.6-alpha.1": "compatible",
|
|
63
|
+
"0.1.6-alpha.2": "compatible",
|
|
64
|
+
"0.1.7-alpha.1": "compatible",
|
|
65
|
+
"0.1.7-alpha.2": "compatible",
|
|
66
|
+
"0.1.7-rc.1": "compatible",
|
|
67
|
+
"0.1.7-rc.2": "compatible",
|
|
68
|
+
"0.2.0-rc.1": "compatible",
|
|
69
|
+
"0.2.0-rc.2": "compatible"
|
|
55
70
|
},
|
|
56
71
|
"profiles": [
|
|
57
|
-
"headless"
|
|
72
|
+
"headless",
|
|
73
|
+
"web"
|
|
58
74
|
]
|
|
59
75
|
}
|
|
60
76
|
},
|
|
61
77
|
"scripts": {
|
|
62
|
-
"build": "tsc -p tsconfig.json",
|
|
78
|
+
"build": "tsc -p tsconfig.json && node scripts/build-client.mjs",
|
|
63
79
|
"typecheck": "tsc -p tsconfig.json --noEmit",
|
|
64
80
|
"bump": "node scripts/bump-version.mjs"
|
|
65
81
|
},
|
|
66
82
|
"peerDependencies": {
|
|
67
|
-
"@deepseek-ai/cordis": "^4.0.
|
|
68
|
-
"@deepseek-ai/dsh-agent": "^0.1.0-rc.6",
|
|
69
|
-
"@deepseek-ai/dsh-llm": "^0.1.0-rc.6",
|
|
70
|
-
"@deepseek-ai/dsh-session": "^0.1.0-rc.6",
|
|
71
|
-
"@deepseek-ai/dsh-skill-filesystem": "^0.1.0-rc.8",
|
|
72
|
-
"@deepseek-ai/schemastery": "^3.18.
|
|
83
|
+
"@deepseek-ai/cordis": "^4.0.4",
|
|
84
|
+
"@deepseek-ai/dsh-agent": "^0.1.0-rc.6 || ^0.2.0-rc.1",
|
|
85
|
+
"@deepseek-ai/dsh-llm": "^0.1.0-rc.6 || ^0.2.0-rc.1",
|
|
86
|
+
"@deepseek-ai/dsh-session": "^0.1.0-rc.6 || ^0.2.0-rc.1",
|
|
87
|
+
"@deepseek-ai/dsh-skill-filesystem": "^0.1.0-rc.8 || ^0.2.0-rc.1",
|
|
88
|
+
"@deepseek-ai/schemastery": "^3.18.4"
|
|
73
89
|
},
|
|
74
90
|
"devDependencies": {
|
|
75
|
-
"@deepseek-ai/cordis": "^4.0.
|
|
91
|
+
"@deepseek-ai/cordis": "^4.0.4",
|
|
76
92
|
"@deepseek-ai/dsh-agent": "^0.1.0-rc.6",
|
|
77
93
|
"@deepseek-ai/dsh-cordis-host-runner": "^0.1.0-rc.8",
|
|
78
94
|
"@deepseek-ai/dsh-home-paths": "^0.1.0-rc.8",
|
|
@@ -82,8 +98,8 @@
|
|
|
82
98
|
"@deepseek-ai/dsh-skill": "^0.1.0-rc.8",
|
|
83
99
|
"@deepseek-ai/dsh-skill-filesystem": "^0.1.0-rc.8",
|
|
84
100
|
"@deepseek-ai/dsh-timeout": "^0.0.1-rc.1",
|
|
85
|
-
"@
|
|
86
|
-
"@
|
|
101
|
+
"@deepseek-ai/schemastery": "^3.18.4",
|
|
102
|
+
"@types/node": "^26.6.2",
|
|
87
103
|
"typescript": "^7.0.2"
|
|
88
104
|
},
|
|
89
105
|
"author": {
|
|
@@ -91,7 +107,10 @@
|
|
|
91
107
|
"url": "https://github.com/AmethystLuna"
|
|
92
108
|
},
|
|
93
109
|
"license": "MIT",
|
|
94
|
-
"repository":
|
|
110
|
+
"repository": {
|
|
111
|
+
"type": "git",
|
|
112
|
+
"url": "git+https://github.com/AmethystLuna/embedded-workbench.git"
|
|
113
|
+
},
|
|
95
114
|
"keywords": [
|
|
96
115
|
"embedded",
|
|
97
116
|
"firmware",
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: embedded-workbench
|
|
3
|
-
description: "Use when starting any non-trivial coding task — loads
|
|
3
|
+
description: "Use when starting any non-trivial coding task — loads risk-proportional workflows, engineering policies, and principles for embedded C/C++ firmware development. NOT for trivial single-line fixes, formatting-only changes, or read-only queries."
|
|
4
4
|
---
|
|
5
5
|
|
|
6
6
|
<SUBAGENT-STOP>
|
|
@@ -23,7 +23,7 @@ If a user's CLAUDE.md says "skip design review for hotfixes" and the workflow re
|
|
|
23
23
|
|
|
24
24
|
## Platform Adaptation
|
|
25
25
|
|
|
26
|
-
This plugin's skills and agents use Claude Code tool names (`Read`, `Write`, `Edit`, `Bash`, `Skill()`). If you are NOT on Claude Code, load `references/platform-tool-mapping.md` for the tool name equivalents on your platform (Codex CLI, Cursor, Kimi CLI, OpenCode, ZCode, Copilot CLI).
|
|
26
|
+
This plugin's skills and agents use Claude Code tool names (`Read`, `Write`, `Edit`, `Bash`, `Skill()`). If you are NOT on Claude Code, load `references/platform-tool-mapping.md` for the tool name equivalents on your platform (Codex CLI, Cursor, Kimi CLI, OpenCode, ZCode, Copilot CLI, DeepSeek Harness). On DeepSeek Harness (dsh) the names are lowercase (`read`/`write`/`edit`/`glob`/`grep`/`pwsh`), skills load through the `skill` tool, the plan gate is `exit_plan_mode`, parallel work uses `subagent`/`subagent_fork`, and the 4 custom agents below are not ported.
|
|
27
27
|
|
|
28
28
|
## Red Flags
|
|
29
29
|
|
|
@@ -34,8 +34,8 @@ If you catch yourself thinking any of these, STOP — you are rationalizing:
|
|
|
34
34
|
| "This is just a quick fix, I don't need a plan" | Quick fixes are the most likely to break something else. A 3-line design check costs 30 seconds. |
|
|
35
35
|
| "I already understand the architecture" | You're looking at one file. The blast radius may span 5 modules you haven't read. |
|
|
36
36
|
| "The worker can figure out the details" | The worker has NO context from previous calls. A vague plan = the worker guessing. |
|
|
37
|
-
| "I'll review it myself, no need for quality-coordinator" | Self-review
|
|
38
|
-
| "This change is too small for a Detailed Change Plan" |
|
|
37
|
+
| "I'll review it myself, no need for quality-coordinator" | Self-review is enough for a bounded, reversible edit. A second pass earns its cost when the blast radius is unclear or the change is hard to undo. |
|
|
38
|
+
| "This change is too small for a Detailed Change Plan" | Bounded, reversible, single-site edits do not need the ceremony. Crossing a module boundary, a public interface, or a non-obvious failure mode does. Name the invariants either way. |
|
|
39
39
|
| "I've explored enough, time to exit plan mode" | ExitPlanMode is the verification gate. Have you loaded `Skill("logicprobe")` or, if it is not installed, the built-in fallback `Skill("fact-check")`? Every plan — simple or complex — must pass this gate before exit. |
|
|
40
40
|
| "This plan is too simple for logicprobe" | logicprobe auto-classifies depth (LIGHTWEIGHT/STANDARD/ESCALATED); the fallback fact-check verifies every claim regardless. You don't decide whether verification is needed. |
|
|
41
41
|
| "I already read the code, I know the file paths and API names are correct" | Organic verification leaves no audit trail. Load `Skill("logicprobe")` or the fallback `Skill("fact-check")`, verify each claim, append the `## Plan Verification` block. |
|
|
@@ -46,65 +46,82 @@ If you catch yourself thinking any of these, STOP — you are rationalizing:
|
|
|
46
46
|
|
|
47
47
|
- Build context before acting: identify domain → load relevant skills → read key sources → analyze → edit.
|
|
48
48
|
- **Facts first, code is truth**: verify every document claim (counts, API names, enum values) against the actual codebase with Grep. Design on verified facts, not assumptions.
|
|
49
|
-
-
|
|
49
|
+
- For a broad search that would otherwise take many Grep/Glob rounds, a read-only `Explore` sub-agent keeps the noise out of this context — but two direct searches are cheaper than dispatching one.
|
|
50
50
|
- Verify every change with `Bash` compilation or tests before reporting success. No verification = no claim of success.
|
|
51
51
|
- Reference file locations with line numbers in all reports: `[path/to/file.c#L100-L110]`.
|
|
52
52
|
- Write project memory to `<workspace>/.github/memory/`, update `MEMORY.md` index. Personal preferences only in `~/.claude/projects/.../memory/`.
|
|
53
53
|
|
|
54
54
|
---
|
|
55
55
|
|
|
56
|
-
##
|
|
56
|
+
## Proportionality
|
|
57
|
+
|
|
58
|
+
Match the process to the risk in front of you, not to habit. Start on the lightest path that covers the risk, and escalate only when you actually cross a boundary.
|
|
59
|
+
|
|
60
|
+
| Task shape | Path |
|
|
61
|
+
| --- | --- |
|
|
62
|
+
| Question, read-only investigation, or a bounded reversible edit (typo, constant, one call site) | Do it directly — no plan header, no sub-agent |
|
|
63
|
+
| One module, known repro, local refactor | Lite |
|
|
64
|
+
| Cross-module, public interface, shared-state ownership, new state machine, non-obvious failure/recovery mode | Full |
|
|
65
|
+
| Platform layer, contracts, staged migration | Full + audit matrix |
|
|
66
|
+
|
|
67
|
+
You have crossed into a heavier path when the change spans more than one module, alters a public interface or state ownership, hides a failure mode you cannot describe, or two attempts have not converged. Scaling a bounded edit up costs more than the edit.
|
|
57
68
|
|
|
58
|
-
|
|
69
|
+
## Context Budget
|
|
59
70
|
|
|
60
|
-
|
|
71
|
+
Complexity decides **how much process**; the budget decides **whether to split the window**. Split it for discovery whose detail you will not cite again — a suite, a log sweep, docs, several independent areas. Keep it for phases that share context (plan → implement → test), or when the change is quick and latency matters.
|
|
61
72
|
|
|
62
|
-
**
|
|
73
|
+
**You cannot see your budget**: most harnesses show a token count to the user's interface, not to you, so never guess one. Act on what you do see — a result truncated, pruned, or spilled to a file, or a compaction / checkpoint summary. When a large step shows no signal at all, **ask the user** what to spend context on rather than deciding silently; if nobody can answer (headless run), take the reversible option and say so.
|
|
63
74
|
|
|
64
|
-
|
|
75
|
+
Delegation is not free: the sub-agent spends its own tokens and its summary still lands here. When it needs context you already built, inherit it (dsh `subagent_fork`; Codex `fork_turns`, default `all`; Claude Code fork mode; Kimi `/btw`) instead of rebuilding it in a prompt. Per-harness markers and tool names: `references/platform-tool-mapping.md`.
|
|
76
|
+
|
|
77
|
+
Delegation is not free: the sub-agent spends its own tokens, its summary still lands in this context, and its window is sized by *its* model, not this one. If the returns would be verbose, ask before delegating. Prefer read-only delegation when the point is to discard detail. When the delegated work needs the context you have already built, inherit it (dsh `subagent_fork`; Codex `fork_turns`, which defaults to `all`; Claude Code forked subagent; Kimi `/btw`) rather than rebuilding it inside a prompt (dsh `subagent`, Claude Code fresh subagent).
|
|
78
|
+
|
|
79
|
+
## Workflows
|
|
65
80
|
|
|
66
|
-
|
|
81
|
+
Sub-agents are **optional equipment, not mandatory stages** — dispatch one when it buys something concrete (a frozen plan, context isolation, an independent review pass), never as ceremony. Each `Agent()` call is stateless: it sees only what its prompt contains. Choose isolation when the detail can be discarded, and inheritance when the sub-agent needs the context you already built (see Context Budget above).
|
|
67
82
|
|
|
68
|
-
###
|
|
83
|
+
### Lite — one module, known repro
|
|
69
84
|
|
|
70
|
-
|
|
85
|
+
Plan the change, make it, verify it. Self-review is sufficient for a bounded, reversible edit.
|
|
71
86
|
|
|
72
|
-
|
|
87
|
+
If the plan rests on claims about the codebase (API names, file paths, enum values, counts), verify those claims before implementing — inline for a small change, or with `logicprobe` / the built-in `fact-check` skill. Escalate to Full when the change turns out to cross a boundary, or when two revisions fail to converge.
|
|
73
88
|
|
|
74
|
-
###
|
|
89
|
+
### Full — cross-module, new interfaces, state-ownership changes
|
|
75
90
|
|
|
76
|
-
|
|
91
|
+
Design first (module boundaries, slice breakdown, recovery paths), then per slice: plan → approve → implement → verify → close. Add an independent review pass (`quality-coordinator`, or a `subagent` with a review prompt) when the change is hard to reverse or the blast radius is unclear; for a well-bounded slice, self-review plus verification evidence is enough.
|
|
77
92
|
|
|
78
|
-
|
|
93
|
+
### Framework — platform layer, contracts, staged migration
|
|
79
94
|
|
|
80
|
-
|
|
95
|
+
Full, plus an audit matrix and rollback triggers; each slice reports its audit delta.
|
|
81
96
|
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
|
85
|
-
|
|
|
86
|
-
| `
|
|
87
|
-
| `
|
|
97
|
+
### Sub-Agents
|
|
98
|
+
|
|
99
|
+
| Agent | Role | When it earns its cost |
|
|
100
|
+
| ------- | ------ | ------ |
|
|
101
|
+
| `architecture-steward` | Read-only planning: design packages, module boundaries, slice breakdown | A design spanning modules you have not read |
|
|
102
|
+
| `design-reviewer` | Design doc fact-check: verifies claims against codebase before implementation | The plan rests on claims you have not verified |
|
|
103
|
+
| `execution-worker` | Plan round → Detailed Change Plan. Implement round → edit + verify | You want the plan frozen before edits, or a noisy investigation kept out of this context |
|
|
104
|
+
| `quality-coordinator` | Implementation review: bugs, compliance, closure completeness | The change is hard to reverse or the risk surface is wide |
|
|
105
|
+
|
|
106
|
+
These are roles, not gates: the main model can play any of them directly when that is cheaper.
|
|
88
107
|
|
|
89
108
|
---
|
|
90
109
|
|
|
91
110
|
## Plan Mode Integration
|
|
92
111
|
|
|
93
|
-
Claude Code's built-in `EnterPlanMode` / `ExitPlanMode` maps to the **
|
|
112
|
+
Claude Code's built-in `EnterPlanMode` / `ExitPlanMode` maps to the **plan phase** of the Lite and Full paths. Plan mode is a read-only exploration + plan-writing phase — it does NOT exempt you from embedded-workbench verification gates.
|
|
94
113
|
|
|
95
114
|
### Plan Verification Gate
|
|
96
115
|
|
|
97
|
-
> **⚠️ logicprobe 已拆分为独立插件 / moved to a standalone plugin** (v0.6.0): the full verification skill (executable model checks, adversarial probing) now ships in its own plugin — <https://github.com/AmethystLuna/logicprobe
|
|
116
|
+
> **⚠️ logicprobe 已拆分为独立插件 / moved to a standalone plugin** (v0.6.0): the full verification skill (executable model checks, adversarial probing) now ships in its own plugin — <https://github.com/AmethystLuna/logicprobe> (install commands are in the README's "Other Plugins"). This plugin ships a built-in simplified fallback — `Skill("fact-check")` — for claim-by-claim verification when logicprobe is not installed; behavioral/model claims then degrade to manual confirmation.
|
|
98
117
|
|
|
99
118
|
**Before calling `ExitPlanMode`**, exactly one of the following must happen:
|
|
100
119
|
|
|
101
|
-
1. **Load `Skill("logicprobe")`** (standalone plugin
|
|
102
|
-
2. **Load `Skill("fact-check")`** (built-in fallback, only when logicprobe is not installed) — verifies every verifiable claim against the codebase with evidence, appends a `## Plan Verification` block marked `fact-check (fallback)`, and tells the user that state-machine/behavioral claims degrade to manual confirmation
|
|
103
|
-
3. **Inform the user** — if you
|
|
104
|
-
|
|
105
|
-
Silent skip is not an option. Either verify, or tell the user you didn't.
|
|
120
|
+
1. **Load `Skill("logicprobe")`** (standalone plugin) — it classifies depth (LIGHTWEIGHT / STANDARD / ESCALATED), runs the verification including executable model checks, and appends a `## Plan Verification` summary block to the plan.
|
|
121
|
+
2. **Load `Skill("fact-check")`** (built-in fallback, only when logicprobe is not installed) — verifies every verifiable claim against the codebase with evidence, appends a `## Plan Verification` block marked `fact-check (fallback)`, and tells the user that state-machine/behavioral claims degrade to manual confirmation.
|
|
122
|
+
3. **Inform the user** — if you load neither, say: *"此计划未经核查。是否需要我在审批前运行事实核查?(This plan has not been fact-verified. Would you like me to run verification before approving?)"*
|
|
106
123
|
|
|
107
|
-
Plan mode permits `Read`, `Glob`, `Grep`, and `Skill` calls
|
|
124
|
+
Silent skip is not an option. Plan mode permits `Read`, `Glob`, `Grep`, and `Skill` calls, so all of this executes before exit.
|
|
108
125
|
|
|
109
126
|
---
|
|
110
127
|
|
|
@@ -113,20 +130,18 @@ Plan mode permits `Read`, `Glob`, `Grep`, and `Skill` calls — all verification
|
|
|
113
130
|
<HARD-GATE>
|
|
114
131
|
### Approval Gate
|
|
115
132
|
|
|
116
|
-
-
|
|
117
|
-
-
|
|
118
|
-
- If execution reveals facts that change scope
|
|
119
|
-
- Do NOT skip the plan phase because "the change is obvious" or "I've done this before."
|
|
133
|
+
- A slice that crosses a module boundary, changes a public interface or state ownership, or carries a non-obvious failure/recovery mode MUST have its plan written and approved before editing: objective, entry point, intended files, change shape, invariants, risks, validation, stop conditions.
|
|
134
|
+
- Bounded, reversible, single-site edits proceed without the ceremony — state the intent, make the change, verify the result.
|
|
135
|
+
- If execution reveals facts that change scope, boundaries, acceptance, or the verification surface, pause and re-approve.
|
|
120
136
|
</HARD-GATE>
|
|
121
137
|
|
|
122
138
|
<HARD-GATE>
|
|
123
139
|
### Closure Gate
|
|
124
140
|
|
|
125
|
-
-
|
|
141
|
+
- Work is not done until implementation intent, verification evidence, and residual risks are all explicit.
|
|
126
142
|
- Skipped checks MUST record a concrete reason. "Looks good" is not a reason.
|
|
127
|
-
- For fault/recovery scenarios,
|
|
128
|
-
- Documentation and memory updates
|
|
129
|
-
- Do NOT call a slice closed if verification, documentation impact, or audit deltas are unclear.
|
|
143
|
+
- For fault/recovery scenarios, cover the normal, failure, and recovery paths.
|
|
144
|
+
- Documentation and memory updates are completed, or explicitly skipped with a reason.
|
|
130
145
|
</HARD-GATE>
|
|
131
146
|
|
|
132
147
|
### Escalation Triggers
|
|
@@ -140,7 +155,7 @@ Plan mode permits `Read`, `Glob`, `Grep`, and `Skill` calls — all verification
|
|
|
140
155
|
Sub-agents are **stateless with no implicit context inheritance** — each spawn only gets what's in its prompt:
|
|
141
156
|
|
|
142
157
|
- **Explicit prompt construction**: put design conclusions, approved Plans, review findings directly in the prompt. Do NOT assume the agent "remembers" previous conversations.
|
|
143
|
-
- **Plan is the key handoff artifact**:
|
|
158
|
+
- **Plan is the key handoff artifact**: when you do dispatch sub-agents, the Detailed Change Plan and review verdicts are the only bridge between Design → Plan round → Implement round. Vague Plans = the next agent guessing.
|
|
144
159
|
- **Pass only what's needed**: Design phase doesn't need full source code. Implement phase doesn't need the full Audit Matrix.
|
|
145
160
|
- **Memory for cross-session persistence**: rules, pitfalls, constraints that need to survive across sessions go in `<workspace>/.github/memory/`. In-session coordination stays in chat.
|
|
146
161
|
- **Long content via path references**: if context is too large, write long content to workspace docs and put only the path in the prompt. Let the agent Read it.
|
|
@@ -179,7 +194,7 @@ When multiple skills could apply, use this order:
|
|
|
179
194
|
"Add retry logic" → state-machine-design first, then c-cpp-dev for implementation.
|
|
180
195
|
"Review this design" → logicprobe first (or the built-in fact-check fallback if logicprobe is not installed), then escalate findings to design-reviewer agent.
|
|
181
196
|
|
|
182
|
-
**Cross-domain links**: load secondary
|
|
197
|
+
**Cross-domain links**: load a secondary skill only when the primary skill's findings call for it — don't pre-load. Each skill's own `Use when` and NOT clauses already tell you when it applies.
|
|
183
198
|
|
|
184
199
|
## Domain Skills
|
|
185
200
|
|
|
@@ -198,45 +213,15 @@ Design doc review, claim verification, logic primitive + adversarial probing →
|
|
|
198
213
|
|
|
199
214
|
## Templates & References
|
|
200
215
|
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
### Platform
|
|
204
|
-
|
|
205
|
-
- `platform-tool-mapping.md` — Claude Code → Codex/Cursor/Kimi/OpenCode/ZCode/Copilot tool name equivalents. **Load this immediately if you are NOT on Claude Code.**
|
|
206
|
-
|
|
207
|
-
### Workflow Templates
|
|
216
|
+
The `references/` directory holds the workflow document templates plus three notes. Read the file you need by name:
|
|
208
217
|
|
|
209
|
-
- `
|
|
210
|
-
- `
|
|
211
|
-
- `
|
|
212
|
-
- `steward-memo.md`
|
|
213
|
-
- `result-note.md` — Post-edit closure evidence
|
|
214
|
-
- `final-qc.md` — Formal review verdict
|
|
215
|
-
- `decision-log.md` — Approved decisions with rationale
|
|
216
|
-
- `audit-ledger.md` — Recurring audit tracking (Framework Workflow)
|
|
217
|
-
- `contract-matrix.md` — Contract-to-sentinel mapping (Framework Workflow)
|
|
218
|
-
- `durable-requirement-notes.md` — Long-lived business invariants
|
|
218
|
+
- `platform-tool-mapping.md` — tool-name equivalents for Codex/Cursor/Kimi/OpenCode/ZCode/Copilot/dsh, and the per-harness context-budget table. **Read this immediately if you are not on Claude Code.**
|
|
219
|
+
- `proactive-suggestions.md` — ready-to-use phrasings for the suggestions below.
|
|
220
|
+
- `INDEX.md` — the working-memory index template.
|
|
221
|
+
- Workflow templates: `detailed-change-plan.md`, `task-charter.md`, `iteration-notes.md`, `steward-memo.md`, `result-note.md`, `final-qc.md`, `decision-log.md`, `audit-ledger.md`, `contract-matrix.md`, `durable-requirement-notes.md`.
|
|
219
222
|
|
|
220
223
|
---
|
|
221
224
|
|
|
222
225
|
## Proactive Suggestions
|
|
223
226
|
|
|
224
|
-
|
|
225
|
-
|
|
226
|
-
| Pattern You Observe | Suggest |
|
|
227
|
-
|---------------------|--------|
|
|
228
|
-
| User describes refactoring a state machine (splitting/merging states, changing transitions) | "Before you start, would you like me to run logic-primitive verification on the refactoring? I can extract the current state machine from code, compare it against your plan, and flag any regressions, deadlocks, or behavioral deltas before you change a single line." |
|
|
229
|
-
| User describes a new state machine or protocol with ≥3 states | "I can run an adversarial verification on this design — 14 automated checks for deadlocks, unreachable states, race conditions, guard completeness, and invariant violations. Want me to do that before we implement?" |
|
|
230
|
-
| User pastes or writes a state enum + switch-case dispatcher | "I notice a state machine here. Would you like me to model it and run completeness checks? I can find missing transitions, detect absorbing error loops, and verify that every state is reachable." |
|
|
231
|
-
| User says "always" / "never" / "guaranteed" about behavior | "That's a behavioral invariant. I can model this and try to find a counter-example — the shortest event sequence that would violate 'X always happens before Y'. Want me to check?" |
|
|
232
|
-
| User reviews a PR or diff that touches a state machine file | "This PR changes state machine logic. Would you like me to extract the before/after models and verify no regressions were introduced?" |
|
|
233
|
-
| User debugs a crash or lockup in a stateful module | "This might be a state machine completeness issue. I can model the state machine from the code and check for deadlocks, unreachable states, or event ordering problems that could cause the lockup." |
|
|
234
|
-
| Task would benefit from parallel execution (multiple independent modules, files, or dimensions) | "These are independent. I can dispatch parallel subagents to handle each module concurrently and synthesize the results. Want me to do that?" |
|
|
235
|
-
| User writes a Detailed Change Plan without design review | "Before implementing, would you like the design-reviewer agent to fact-check this plan against the codebase? It catches API mismatches, missing modules, and mechanism feasibility issues before you write code." |
|
|
236
|
-
|
|
237
|
-
### Suggestion Rules
|
|
238
|
-
|
|
239
|
-
- **Suggest once per task**, not repeatedly. If the user declines, don't push.
|
|
240
|
-
- **Be specific about what the feature does** — don't just name-drop. Say "I can find deadlocks and missing transitions" not "I can run logicprobe." If logicprobe is not installed, offer the built-in fact-check skill: "I can check every claim in the plan against the codebase."
|
|
241
|
-
- **Estimate cost**: for lightweight checks, say "this takes ~30 seconds." For Python harness runs, say "this will generate and run a verification script."
|
|
242
|
-
- **Respect the user's decision**: if they decline, move on. The features are tools, not requirements.
|
|
227
|
+
Suggest a capability when it clearly applies and the user is unlikely to know it exists — state-machine verification for a state machine, a protocol, or an "always"/"never" claim; parallel sub-agents for genuinely independent modules; a design fact-check for a plan that had no review. Suggest **once per task**, say what the check finds rather than which tool runs it, and drop it if the user declines. Ready-to-use phrasings: `references/proactive-suggestions.md`.
|
|
@@ -1,31 +1,31 @@
|
|
|
1
1
|
# Platform Tool Mapping
|
|
2
2
|
|
|
3
|
-
This plugin's skills and agents are written with Claude Code tool names. This reference maps each tool to equivalents on other platforms. When an agent prompt says "use `Read`" but you are on Codex CLI, use the mapped tool instead.
|
|
3
|
+
This plugin's skills and agents are written with Claude Code tool names. This reference maps each tool to equivalents on the other supported platforms — Codex CLI, Cursor, Kimi CLI, OpenCode, ZCode, Copilot CLI, and DeepSeek Harness (dsh). When an agent prompt says "use `Read`" but you are on Codex CLI, use the mapped tool instead.
|
|
4
4
|
|
|
5
5
|
## Core File Tools
|
|
6
6
|
|
|
7
|
-
| Claude Code | Codex CLI | Cursor | Kimi CLI | OpenCode | ZCode | Copilot CLI |
|
|
8
|
-
|
|
9
|
-
| `Read` | `read_file` | `read_file` | `read_file` | `read` | `read_file` | `
|
|
10
|
-
| `Write` | `write_file` | `write_to_file` | `write_file` | `write` | `write_file` | `
|
|
11
|
-
| `Edit` | `edit_file` | `replace_in_file` | `edit_file` | `edit` | `edit_file` | `
|
|
12
|
-
| `Glob` | `search_file` | `search_file` | `glob` | `glob` | `search_file` | `
|
|
13
|
-
| `Grep` | `search_content` | `search_content` | `grep` | `grep` | `search_content` | `
|
|
14
|
-
| `Bash` | `run_shell` | `execute_command` | `execute_command` | `terminal` | `run_shell` | `
|
|
7
|
+
| Claude Code | Codex CLI | Cursor | Kimi CLI | OpenCode | ZCode | Copilot CLI | DeepSeek Harness (dsh) |
|
|
8
|
+
|-------------|-----------|--------|----------|----------|-------|-------------|------------------------|
|
|
9
|
+
| `Read` | `read_file` | `read_file` | `read_file` | `read` | `read_file` | `view` | `read` |
|
|
10
|
+
| `Write` | `write_file` | `write_to_file` | `write_file` | `write` | `write_file` | `create` | `write` |
|
|
11
|
+
| `Edit` | `edit_file` | `replace_in_file` | `edit_file` | `edit` | `edit_file` | `edit` | `edit` |
|
|
12
|
+
| `Glob` | `search_file` | `search_file` | `glob` | `glob` | `search_file` | `glob` | `glob` |
|
|
13
|
+
| `Grep` | `search_content` | `search_content` | `grep` | `grep` | `search_content` | `grep` | `grep` |
|
|
14
|
+
| `Bash` | `run_shell` | `execute_command` | `execute_command` | `terminal` | `run_shell` | `bash` | `pwsh` / `bash` |
|
|
15
15
|
|
|
16
16
|
## Agent & Skill Tools
|
|
17
17
|
|
|
18
|
-
| Claude Code | Codex CLI | Cursor | Kimi CLI | OpenCode | ZCode | Copilot CLI |
|
|
19
|
-
|
|
20
|
-
| `Skill("name")` | `$name` (auto) | `use_skill` | `/skill:name` | `$name` | `$name` | `skill(
|
|
21
|
-
| `Agent` | `task` | `task` | `agent` | `task` | `agent` | `task` |
|
|
18
|
+
| Claude Code | Codex CLI | Cursor | Kimi CLI | OpenCode | ZCode | Copilot CLI | DeepSeek Harness (dsh) |
|
|
19
|
+
|-------------|-----------|--------|----------|----------|-------|-------------|------------------------|
|
|
20
|
+
| `Skill("name")` | `$name` (auto) | `use_skill` | `/skill:name` | `$name` | `$name` | `skill` tool, or `/skill-name` | `skill` tool (`name: "..."`) |
|
|
21
|
+
| `Agent` | `task` | `task` | `agent` | `task` | `agent` | `task` (`agent_type`) | `subagent` / `subagent_fork` |
|
|
22
22
|
|
|
23
23
|
## Web Tools
|
|
24
24
|
|
|
25
|
-
| Claude Code | Codex CLI | Cursor | Kimi CLI | OpenCode | ZCode | Copilot CLI |
|
|
26
|
-
|
|
27
|
-
| `WebFetch` | `web_fetch` | `web_fetch` | `web_search` | `fetch` | `web_fetch` | `web_fetch` |
|
|
28
|
-
| `WebSearch` | `web_search` | `web_search` | `web_search` | `search` | `web_search` | `web_search` |
|
|
25
|
+
| Claude Code | Codex CLI | Cursor | Kimi CLI | OpenCode | ZCode | Copilot CLI | DeepSeek Harness (dsh) |
|
|
26
|
+
|-------------|-----------|--------|----------|----------|-------|-------------|------------------------|
|
|
27
|
+
| `WebFetch` | `web_fetch` | `web_fetch` | `web_search` | `fetch` | `web_fetch` | `web_fetch` | `web_fetch` |
|
|
28
|
+
| `WebSearch` | `web_search` | `web_search` | `web_search` | `search` | `web_search` | — (no tool) | `web_search` |
|
|
29
29
|
|
|
30
30
|
## Platform-Specific Notes
|
|
31
31
|
|
|
@@ -62,13 +62,50 @@ This plugin's skills and agents are written with Claude Code tool names. This re
|
|
|
62
62
|
|
|
63
63
|
### Copilot CLI (GitHub Copilot)
|
|
64
64
|
|
|
65
|
-
- `
|
|
66
|
-
-
|
|
67
|
-
-
|
|
65
|
+
- Copilot CLI has a plugin system: a `plugin.json` manifest plus `agents/NAME.agent.md`, `skills/NAME/SKILL.md`, hooks, and MCP servers, distributed through marketplaces (`copilot-plugins`, `awesome-copilot`). Install with `copilot plugin install NAME@MARKETPLACE`, or register this repository with `copilot plugin marketplace add AmethystLuna/embedded-workbench`.
|
|
66
|
+
- **This repository already works as a Copilot plugin**: Copilot's legacy manifest lookup checks `.plugin/plugin.json`, `plugin.json`, `.github/plugin/plugin.json`, then `.claude-plugin/plugin.json` — and it also reads `marketplace.json` from `.claude-plugin/`. Both files already exist here.
|
|
67
|
+
- Skills follow the Agent Skills open standard, so `skills/NAME/SKILL.md` loads as-is. Invoke one with `/skill-name` in a prompt, e.g. `Use the /debug-methodology skill to ...`; inspect them with `/skills list` or `copilot skill list`.
|
|
68
|
+
- Tool names: `view` (read), `create` (write), `edit`, `glob`, `grep`, `bash`, `web_fetch`, `task`, `skill`. The edit tool is `str_replace_editor` under the hood, with `view`/`create`/`edit` as its aliases.
|
|
69
|
+
- There is **no** `web_search` tool — use `web_fetch` against a search URL.
|
|
70
|
+
- `task` requires an `agent_type` of `explore`, `task`, `general-purpose`, or `code-review`. `explore` is read-only and the cheapest; prefer it for searches.
|
|
71
|
+
- Sub-agent status comes from `read_agent` / `list_agents`; async shell sessions use `bash` with `mode: "async"` plus `read_bash` / `write_bash` / `stop_bash` / `list_bash`.
|
|
72
|
+
- Agents in this plugin are **not** loaded by Copilot: it requires `agents/NAME.agent.md`, and this repository ships `agents/NAME.md`.
|
|
73
|
+
|
|
74
|
+
### DeepSeek Harness (dsh)
|
|
75
|
+
|
|
76
|
+
- dsh is the one platform this plugin ships a **native bundle** for (root `package.json` declares `dsh.bundle`), so no skill-copy step is needed: `dsh plugin --profile <name> add dsh-embedded-workbench`.
|
|
77
|
+
- Tool names are lowercase: `read`, `write`, `edit`, `glob`, `grep`. Shell access is `pwsh` on Windows hosts and `bash` elsewhere.
|
|
78
|
+
- Skills load through the `skill` tool by name — there is no `Skill(...)` call syntax.
|
|
79
|
+
- Plan mode is the native `exit_plan_mode` tool, not `ExitPlanMode`.
|
|
80
|
+
- The 4 Claude Code sub-agents are intentionally **not** ported: use the native sub-agent tools for parallel work, and take the steward/reviewer roles directly.
|
|
81
|
+
- `subagent` starts a **fresh** context and returns only its result — reach for it when the detail can be discarded (read-only discovery, sweeping logs, running a suite). `subagent_fork` is **seeded with this conversation** — reach for it when the sub-task needs context you already built, instead of rebuilding that context inside a prompt.
|
|
82
|
+
- `Agent(subagent_type: "Explore")` has no separate equivalent — use `glob` / `grep` directly, or dispatch a read-only `subagent`.
|
|
83
|
+
- **Context-budget markers** you can actually see (no token meter is exposed to the model): a tool result rewritten with `[... tool result middle pruned ...]`, a spill notice naming the omitted bytes and the complete-result path, and the compaction checkpoint preamble (`This is an automatically generated checkpoint condensing an earlier span…`). Treat any of them as "stop pulling content in whole" — switch to file references, or delegate the reading.
|
|
84
|
+
- Install, verify, and config-override details live in `.dsh/INSTALL.md` at the repository root.
|
|
85
|
+
|
|
86
|
+
## Context Budget Interfaces
|
|
87
|
+
|
|
88
|
+
What each harness actually exposes to the **model** about its context budget, checked 2026-09-25 against vendor documentation (and, for Codex CLI, the `openai/codex` source). "User-only" means the signal exists but never reaches the model. `UNVERIFIED` means the vendor does not document it — do not invent a marker string for that cell.
|
|
89
|
+
|
|
90
|
+
| Harness | Model sees a token/percent readout? | Compaction — trigger and what the model receives | Oversized tool output | Sub-agent context |
|
|
91
|
+
| --- | --- | --- | --- | --- |
|
|
92
|
+
| Claude Code | No — `/context` and the status line are user-facing (the status line even exposes `used_percentage`, but to the terminal, not the model) | Automatic near the limit (`/autocompact <100K–1M>`). Older tool outputs are cleared first, then history is summarized; afterwards an oversized re-read returns as a `Referenced file` path instead of content, and invoked skill bodies are re-injected capped at 5,000 tokens each / 25,000 total | Bash success output past ~30,000 characters becomes a file path plus a preview of up to 2,000 characters (a failure is truncated in place with **no** path); hook `additionalContext` over 10,000 characters is saved to a file and the model gets a preview plus the path; Read adds a `PARTIAL view` notice and Glob flags a 100-file cap | Fresh isolated window; **fork mode** (on by default in interactive sessions, off under `-p`/SDK) inherits the parent conversation. Note a skill's `context: fork` is *not* a conversation fork |
|
|
93
|
+
| Codex CLI | Yes, but feature-gated: a `get_context_remaining` tool (feature `token_budget`, **off by default**) answers "You have {N} tokens left in this context window", and a reminder is injected once remaining drops below the threshold. `/status` percentages are user-only | Default 90% of the window, hard cap 95%. The model receives the summary behind a fixed handoff preamble ("Another language model started to solve this problem…") | `truncation_policy` defaults to bytes/10000: the model sees `Warning: truncated output (original token count: N)` and `…N tokens truncated…`. Spill-to-file exists only for hooks (`Full hook output saved to: <path>`) | `spawn_agent` `fork_turns`: `"none"` / `"all"` / a turn count — **the default is `"all"`** |
|
|
94
|
+
| Copilot CLI | No — `/context`, `/usage`, and `footer.showContextWindow` are user-only | ~80% of the window in the background (defers to ~90% when static context already uses ≥75%), pausing at ~95%. The model gets the summary plus preserved user instructions and plan/todo state; each compaction writes a numbered checkpoint (`/session checkpoints`) | **Over 20 KiB** is saved to a temporary file and the model gets the path plus a preview (`COPILOT_LARGE_OUTPUT_THRESHOLD_BYTES`) | Fresh window; `contextTier: "inherit"` inherits the parent's tier. Repository custom instructions are **not** inherited unless the agent sets `include-custom-instructions: true` |
|
|
95
|
+
| Cursor | Nothing documented — the context ring and breakdown tray are UI | Automatic once the window is full; `/summarize` (alias `/compress`). The injected summary text is not published | Not documented (UNVERIFIED) | Fresh window only; `model: inherit` inherits the **model**, not the conversation |
|
|
96
|
+
| OpenCode | No | `compaction.auto` is on, keeping ~15,000 recent tokens (`keep.tokens`) with `buffer` at 10% of the limit. The model sees the summary **as past conversation** — there is no distinct notice, and later compactions update the same summary. Native provider compaction substitutes an opaque encrypted item | Long tool output is shortened beside the summary; no marker string documented | UNVERIFIED |
|
|
97
|
+
| Kimi CLI | No — `/usage` is a user command | Automatic compression as the conversation approaches the window limit, plus `/compact [hint]`. `/undo` cannot cross a compaction | Not documented (UNVERIFIED) | Isolated context per sub-agent (its own event stream); built-ins `coder`, `explore`, `plan`. `/btw` runs in a **forked** sub-agent that sees the conversation |
|
|
98
|
+
| ZCode | No — the usage-stats panel shows the context composition as UI | `/compact` is user-initiated; automatic compaction is not documented | Not documented (UNVERIFIED) | UNVERIFIED |
|
|
99
|
+
| DeepSeek Harness (dsh) | No — `ctx.tokenMeter` deliberately adds no model-visible surface | At `floor(min(W × 0.8, W − O − 65536))`. The model receives the checkpoint preamble plus `<compacted-summary>`; `/compact` is a user command | `[... tool result middle pruned ...]` for results over 8,192 code points (4,096 head / 1,024 tail) once compaction pressure qualifies, plus a spill notice naming the omitted bytes and the complete-result path (12,500 estimated tokens in `dsh-base`) | `subagent` starts fresh; `subagent_fork` is seeded with the conversation so far |
|
|
100
|
+
|
|
101
|
+
Two consequences worth remembering:
|
|
102
|
+
|
|
103
|
+
- **A budget readout reaching the model is the exception, not the rule.** Only Codex documents one, and it is off by default. Everywhere else, act on the truncation, spill, or compaction notice your harness actually shows.
|
|
104
|
+
- **With no signal, ask the user instead of guessing.** Say what you are about to consume, that this harness gives you no budget readout, and the options with their costs, then do what they choose — falling back to the reversible option only when no one can answer.
|
|
68
105
|
|
|
69
106
|
## Sub-Agent Platform Equivalents
|
|
70
107
|
|
|
71
|
-
This plugin defines 4 sub-agents. On platforms without an `Agent` tool:
|
|
108
|
+
This plugin defines 4 sub-agents. On platforms without an `Agent` tool (including DeepSeek Harness):
|
|
72
109
|
|
|
73
110
|
| Claude Code Agent | Alternative Approach |
|
|
74
111
|
|-------------------|---------------------|
|
|
@@ -81,7 +118,7 @@ This plugin defines 4 sub-agents. On platforms without an `Agent` tool:
|
|
|
81
118
|
|
|
82
119
|
Load this when:
|
|
83
120
|
|
|
84
|
-
- The session is NOT running on Claude Code (check environment: `$CLAUDE_PLUGIN_ROOT`, `$CODEX_CLI`, `$CURSOR_PLUGIN_ROOT`, etc
|
|
121
|
+
- The session is NOT running on Claude Code (check environment: `$CLAUDE_PLUGIN_ROOT`, `$CODEX_CLI`, `$CURSOR_PLUGIN_ROOT`, etc.; a dsh session runs under `$DSH_HOME` and sees none of those)
|
|
85
122
|
- An agent prompt references a tool you don't recognize
|
|
86
123
|
- You need to translate a skill or workflow instruction to your platform
|
|
87
124
|
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
# Proactive Suggestion Phrasings
|
|
2
|
+
|
|
3
|
+
Ready-to-use wording for the suggestions the bootstrap skill describes. Suggest once per
|
|
4
|
+
task, say what the check actually finds rather than which tool runs it, and drop it if the
|
|
5
|
+
user declines.
|
|
6
|
+
|
|
7
|
+
| Pattern you observe | Suggest |
|
|
8
|
+
| --- | --- |
|
|
9
|
+
| Refactoring a state machine (splitting/merging states, changing transitions) | "Before you start, would you like me to run logic-primitive verification on the refactoring? I can extract the current state machine from code, compare it against your plan, and flag regressions, deadlocks, or behavioral deltas before you change a line." |
|
|
10
|
+
| A new state machine or protocol with ≥3 states | "I can run an adversarial verification on this design — 22 automated checks (8 structural S1–S8 plus 14 adversarial A1–A14) for deadlocks, unreachable states, race conditions, guard completeness, and invariant violations." |
|
|
11
|
+
| A state enum plus a switch-case dispatcher | "I notice a state machine here. I can model it and run completeness checks — missing transitions, absorbing error loops, unreachable states." |
|
|
12
|
+
| "always" / "never" / "guaranteed" about behavior | "That's a behavioral invariant. I can model it and look for a counter-example — the shortest event sequence that violates 'X always happens before Y'." |
|
|
13
|
+
| A PR or diff touching a state machine file | "I can extract the before/after models and verify that no regressions were introduced." |
|
|
14
|
+
| A crash or lockup in a stateful module | "This might be a state-machine completeness issue. I can model it and check for deadlocks, unreachable states, or event-ordering problems behind the lockup." |
|
|
15
|
+
| Multiple independent modules, files, or dimensions | "These are independent. I can dispatch parallel sub-agents per module and synthesize the results." |
|
|
16
|
+
| A Detailed Change Plan with no design review | "Would you like the plan fact-checked against the codebase first? It catches API mismatches, missing modules, and mechanism-feasibility gaps before you write code." |
|
|
17
|
+
|
|
18
|
+
## Rules
|
|
19
|
+
|
|
20
|
+
- **Suggest once per task**, not repeatedly. If the user declines, don't push.
|
|
21
|
+
- **Say what the check finds**, not which tool runs it. If logicprobe is not installed,
|
|
22
|
+
offer the built-in `fact-check` skill instead — "I can check every claim in the plan
|
|
23
|
+
against the codebase" — and note that behavioral/model claims then degrade to manual
|
|
24
|
+
confirmation.
|
|
25
|
+
- **Estimate cost** when it matters: a lightweight check is seconds; a full model
|
|
26
|
+
verification pass generates and runs a script.
|
|
27
|
+
- **Respect the decision**: these are tools, not requirements.
|