dsh-logicprobe 0.5.5 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.en-US.md +6 -3
- package/README.md +5 -2
- package/lib/compose-tool.js +43 -0
- package/lib/concurrency.js +11 -1
- package/lib/engine.js +2449 -1699
- package/lib/export-tool.js +45 -0
- package/lib/exporters.js +389 -0
- package/lib/index.js +48 -25
- package/lib/tool.js +58 -57
- package/lib/types/compose-tool.d.ts +6 -0
- package/lib/types/concurrency.d.ts +2 -0
- package/lib/types/engine.d.ts +255 -189
- package/lib/types/export-tool.d.ts +7 -0
- package/lib/types/exporters.d.ts +17 -0
- package/lib/types/tool.d.ts +5 -4
- package/package.json +106 -102
- package/skills/logicprobe/SKILL.md +14 -8
- package/skills/logicprobe/references/concurrency-risk-guide.md +2 -0
- package/skills/logicprobe/references/dsh-model-schema.md +266 -204
- package/skills/logicprobe/references/gap-routing-guide.md +36 -0
- package/skills/logicprobe/references/logic-verification-guide.md +56 -4
- package/skills/logicprobe/references/logicprobe-engine.py +3028 -0
- package/skills/logicprobe/references/verification-harness.py +153 -0
- package/src/compose-tool.ts +46 -0
- package/src/concurrency.ts +13 -1
- package/src/engine.ts +2568 -1778
- package/src/export-tool.ts +47 -0
- package/src/exporters.ts +347 -0
- package/src/index.ts +349 -315
- package/src/tool.ts +61 -60
package/package.json
CHANGED
|
@@ -1,102 +1,106 @@
|
|
|
1
|
-
{
|
|
2
|
-
"name": "dsh-logicprobe",
|
|
3
|
-
"version": "0.
|
|
4
|
-
"description": "Design document & plan claim verification — enumerate claims, verify against codebase facts, then escalate to state-machine verification (S1-S8/A1-
|
|
5
|
-
"type": "module",
|
|
6
|
-
"main": "lib/index.js",
|
|
7
|
-
"types": "lib/types/index.d.ts",
|
|
8
|
-
"exports": {
|
|
9
|
-
".": {
|
|
10
|
-
"types": "./lib/types/index.d.ts",
|
|
11
|
-
"default": "./lib/index.js"
|
|
12
|
-
},
|
|
13
|
-
"./cordis.patch.yml": "./cordis.patch.yml",
|
|
14
|
-
"./package.json": "./package.json"
|
|
15
|
-
},
|
|
16
|
-
"files": [
|
|
17
|
-
"lib",
|
|
18
|
-
"src",
|
|
19
|
-
"skills",
|
|
20
|
-
"cordis.patch.yml"
|
|
21
|
-
],
|
|
22
|
-
"dsh": {
|
|
23
|
-
"engines": {
|
|
24
|
-
"dsh": ">=0.1.0-rc.7"
|
|
25
|
-
},
|
|
26
|
-
"category": "skill",
|
|
27
|
-
"displayName": "逻辑探针",
|
|
28
|
-
"bundle": {
|
|
29
|
-
"patch": "./cordis.patch.yml"
|
|
30
|
-
},
|
|
31
|
-
"compatibility": {
|
|
32
|
-
"dsh": "^0.1.0-rc.7 || ^0.1.1-rc.1 || ^0.1.2-alpha.2 || ^0.1.2-alpha.3",
|
|
33
|
-
"dshReleases": {
|
|
34
|
-
"0.1.0-rc.7": "compatible",
|
|
35
|
-
"0.1.0-rc.8": "compatible",
|
|
36
|
-
"0.1.1-rc.1": "compatible",
|
|
37
|
-
"0.1.1-rc.2": "compatible",
|
|
38
|
-
"0.1.2-alpha.2": "compatible",
|
|
39
|
-
"0.1.2-alpha.3": "compatible"
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
"
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
"
|
|
51
|
-
"
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
"
|
|
55
|
-
"
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
"@deepseek-ai/
|
|
59
|
-
"@deepseek-ai/dsh-
|
|
60
|
-
"@deepseek-ai/
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
"@deepseek-ai/
|
|
64
|
-
"@deepseek-ai/
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
"@deepseek-ai/
|
|
68
|
-
"@deepseek-ai/dsh-
|
|
69
|
-
"@deepseek-ai/dsh-
|
|
70
|
-
"@deepseek-ai/dsh-
|
|
71
|
-
"@deepseek-ai/dsh-
|
|
72
|
-
"@deepseek-ai/dsh-
|
|
73
|
-
"@deepseek-ai/dsh-
|
|
74
|
-
"@deepseek-ai/dsh-
|
|
75
|
-
"@deepseek-ai/
|
|
76
|
-
"@
|
|
77
|
-
"
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
"
|
|
81
|
-
"
|
|
82
|
-
},
|
|
83
|
-
"
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
"
|
|
91
|
-
"
|
|
92
|
-
"
|
|
93
|
-
"
|
|
94
|
-
"
|
|
95
|
-
"
|
|
96
|
-
"
|
|
97
|
-
"
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
"
|
|
101
|
-
|
|
102
|
-
|
|
1
|
+
{
|
|
2
|
+
"name": "dsh-logicprobe",
|
|
3
|
+
"version": "0.6.0",
|
|
4
|
+
"description": "Design document & plan claim verification — enumerate claims, verify against codebase facts, then escalate to state-machine verification (S1-S8/A1-A12, including budget/worst-case path-cost checks) and data-model verification (DS/DA/DD) for behavioral claims. Supports before/after regression, idempotency/monotonic/sequence/leads-to/atomicity constraints, and concurrency risk mining. Ships a native DeepSeek Harness (dsh) bundle that injects the claim-verification gate into the first model step.",
|
|
5
|
+
"type": "module",
|
|
6
|
+
"main": "lib/index.js",
|
|
7
|
+
"types": "lib/types/index.d.ts",
|
|
8
|
+
"exports": {
|
|
9
|
+
".": {
|
|
10
|
+
"types": "./lib/types/index.d.ts",
|
|
11
|
+
"default": "./lib/index.js"
|
|
12
|
+
},
|
|
13
|
+
"./cordis.patch.yml": "./cordis.patch.yml",
|
|
14
|
+
"./package.json": "./package.json"
|
|
15
|
+
},
|
|
16
|
+
"files": [
|
|
17
|
+
"lib",
|
|
18
|
+
"src",
|
|
19
|
+
"skills",
|
|
20
|
+
"cordis.patch.yml"
|
|
21
|
+
],
|
|
22
|
+
"dsh": {
|
|
23
|
+
"engines": {
|
|
24
|
+
"dsh": ">=0.1.0-rc.7"
|
|
25
|
+
},
|
|
26
|
+
"category": "skill",
|
|
27
|
+
"displayName": "逻辑探针",
|
|
28
|
+
"bundle": {
|
|
29
|
+
"patch": "./cordis.patch.yml"
|
|
30
|
+
},
|
|
31
|
+
"compatibility": {
|
|
32
|
+
"dsh": "^0.1.0-rc.7 || ^0.1.1-rc.1 || ^0.1.2-alpha.2 || ^0.1.2-alpha.3 || ^0.1.2-alpha.4 || ^0.1.2-alpha.5 || ^0.1.2-rc.1",
|
|
33
|
+
"dshReleases": {
|
|
34
|
+
"0.1.0-rc.7": "compatible",
|
|
35
|
+
"0.1.0-rc.8": "compatible",
|
|
36
|
+
"0.1.1-rc.1": "compatible",
|
|
37
|
+
"0.1.1-rc.2": "compatible",
|
|
38
|
+
"0.1.2-alpha.2": "compatible",
|
|
39
|
+
"0.1.2-alpha.3": "compatible",
|
|
40
|
+
"0.1.2-alpha.4": "compatible",
|
|
41
|
+
"0.1.2-alpha.5": "compatible",
|
|
42
|
+
"0.1.2-rc.1": "compatible"
|
|
43
|
+
},
|
|
44
|
+
"profiles": [
|
|
45
|
+
"headless"
|
|
46
|
+
]
|
|
47
|
+
}
|
|
48
|
+
},
|
|
49
|
+
"scripts": {
|
|
50
|
+
"build": "tsc -p tsconfig.json",
|
|
51
|
+
"typecheck": "tsc -p tsconfig.json --noEmit",
|
|
52
|
+
"bump": "node scripts/bump-version.mjs",
|
|
53
|
+
"test:engine": "npm run build && node tests/engine/run.mjs && node tests/data-engine/run.mjs && node tests/concurrency/run.mjs && node tests/apply-smoke.mjs && node tests/exporters/run.mjs && node tests/external/run.mjs && node tests/python/run.mjs",
|
|
54
|
+
"test:full": "npm run build && node tests/full-suite.mjs",
|
|
55
|
+
"test:python": "npm run build && node tests/python/run.mjs"
|
|
56
|
+
},
|
|
57
|
+
"peerDependencies": {
|
|
58
|
+
"@deepseek-ai/cordis": "^4.0.1",
|
|
59
|
+
"@deepseek-ai/dsh-agent": "^0.1.0-rc.6",
|
|
60
|
+
"@deepseek-ai/dsh-llm": "^0.1.0-rc.6",
|
|
61
|
+
"@deepseek-ai/dsh-session": "^0.1.0-rc.6",
|
|
62
|
+
"@deepseek-ai/dsh-skill-filesystem": "^0.1.0-rc.8",
|
|
63
|
+
"@deepseek-ai/dsh-tools": "^0.1.0-rc.6",
|
|
64
|
+
"@deepseek-ai/schemastery": "^3.18.1"
|
|
65
|
+
},
|
|
66
|
+
"devDependencies": {
|
|
67
|
+
"@deepseek-ai/cordis": "^4.0.1",
|
|
68
|
+
"@deepseek-ai/dsh-agent": "^0.1.0-rc.6",
|
|
69
|
+
"@deepseek-ai/dsh-cordis-host-runner": "^0.1.0-rc.8",
|
|
70
|
+
"@deepseek-ai/dsh-home-paths": "^0.1.0-rc.8",
|
|
71
|
+
"@deepseek-ai/dsh-llm": "^0.1.0-rc.6",
|
|
72
|
+
"@deepseek-ai/dsh-scope": "^0.1.0-rc.6",
|
|
73
|
+
"@deepseek-ai/dsh-session": "^0.1.0-rc.6",
|
|
74
|
+
"@deepseek-ai/dsh-skill": "^0.1.0-rc.6",
|
|
75
|
+
"@deepseek-ai/dsh-skill-filesystem": "^0.1.0-rc.8",
|
|
76
|
+
"@deepseek-ai/dsh-system-prompt": "^0.1.0-rc.6",
|
|
77
|
+
"@deepseek-ai/dsh-timeout": "^0.1.0-rc.6",
|
|
78
|
+
"@deepseek-ai/dsh-tools": "^0.1.0-rc.6",
|
|
79
|
+
"@deepseek-ai/schemastery": "^3.18.1",
|
|
80
|
+
"@types/node": "^26.3.0",
|
|
81
|
+
"typescript": "^7.0.2"
|
|
82
|
+
},
|
|
83
|
+
"author": {
|
|
84
|
+
"name": "Amethyst Luna",
|
|
85
|
+
"url": "https://github.com/AmethystLuna"
|
|
86
|
+
},
|
|
87
|
+
"license": "MIT",
|
|
88
|
+
"repository": "https://github.com/AmethystLuna/logicprobe",
|
|
89
|
+
"keywords": [
|
|
90
|
+
"verification",
|
|
91
|
+
"fact-check",
|
|
92
|
+
"design-review",
|
|
93
|
+
"plan-review",
|
|
94
|
+
"state-machine",
|
|
95
|
+
"logic-verification",
|
|
96
|
+
"adversarial",
|
|
97
|
+
"refactoring",
|
|
98
|
+
"model-checking",
|
|
99
|
+
"agentskills",
|
|
100
|
+
"plugin",
|
|
101
|
+
"dsh"
|
|
102
|
+
],
|
|
103
|
+
"engines": {
|
|
104
|
+
"node": ">=20"
|
|
105
|
+
}
|
|
106
|
+
}
|
|
@@ -44,7 +44,7 @@ This step is NOT skippable — it creates an explicit, auditable record of what
|
|
|
44
44
|
|-------|----------|
|
|
45
45
|
| LIGHTWEIGHT | All 5 checklist items (file paths, API/type names, line numbers, behavioral claims, mechanism feasibility) answered in context with explicit results per item |
|
|
46
46
|
| STANDARD | Phase 1-2: enumerate every verifiable claim → verify each against codebase with evidence |
|
|
47
|
-
| ESCALATED | Full pipeline: Phase 1-5 including Logic Primitive Verification (Phase 2a + 2b,
|
|
47
|
+
| ESCALATED | Full pipeline: Phase 1-5 including Logic Primitive Verification (Phase 2a + 2b, 22 checks) |
|
|
48
48
|
|
|
49
49
|
### Plan Verification Block
|
|
50
50
|
|
|
@@ -131,12 +131,13 @@ When Phase 2 triggers escalation, do NOT proceed to Phase 3 until the verificati
|
|
|
131
131
|
```text
|
|
132
132
|
Document claims → Extract model → Runtime check:
|
|
133
133
|
├── DSH + `logicprobe_verify` tool available → build Model schema v1 (references/dsh-model-schema.md) → call the tool → structured report
|
|
134
|
-
├── Python available
|
|
134
|
+
├── Python available + LogicModelV1 JSON at hand → references/logicprobe-engine.py verify model.json (S1-S8/A1-A14/D1-D4; subcommands compose / export add C1-C2 and UPPAAL/TLA+/PRISM/SPIN output)
|
|
135
|
+
├── Python available + model only as extracted dicts → fill in references/verification-harness.py → run → report
|
|
135
136
|
└── No Python → Manual Verification Mode (see references/logic-verification-guide.md#manual-verification-mode)
|
|
136
137
|
|
|
137
138
|
Refactoring variant:
|
|
138
139
|
Old code + Refactoring plan → Extract BEFORE model + AFTER model
|
|
139
|
-
→ Run pipeline on AFTER model (
|
|
140
|
+
→ Run pipeline on AFTER model (22 checks)
|
|
140
141
|
→ Compare BEFORE vs AFTER: behavioral preservation, regression, complexity delta
|
|
141
142
|
→ Flag any invariant that held in BEFORE but fails in AFTER
|
|
142
143
|
```
|
|
@@ -148,7 +149,7 @@ When the document under review is a refactoring plan (modifying existing state m
|
|
|
148
149
|
1. **Extract the BEFORE model** from the existing codebase (not the plan — verify what the code actually does, not what the plan claims it does)
|
|
149
150
|
2. **Extract the AFTER model** from the refactoring plan
|
|
150
151
|
3. **Show both tables AND their model narratives** to the user side by side and confirm the delta is intentional
|
|
151
|
-
4. **Run Phase 2a + 2b on the AFTER model** — same
|
|
152
|
+
4. **Run Phase 2a + 2b on the AFTER model** — same 22 checks as new design
|
|
152
153
|
5. **Compare BEFORE vs AFTER**:
|
|
153
154
|
|
|
154
155
|
| Check | Method | Severity if Violated |
|
|
@@ -171,9 +172,9 @@ When the document under review is a refactoring plan (modifying existing state m
|
|
|
171
172
|
|
|
172
173
|
Do NOT attempt to install Python — the user's embedded development machine may be air-gapped or locked down.
|
|
173
174
|
|
|
174
|
-
In DSH, prefer the native `logicprobe_verify` tool (model JSON, structured guard DSL, path-aware invariants) — see `references/dsh-model-schema.md`. For non-DSH hosts, the
|
|
175
|
+
In DSH, prefer the native `logicprobe_verify` tool (model JSON, structured guard DSL, path-aware invariants) — see `references/dsh-model-schema.md`. For non-DSH hosts: when the model is already a LogicModelV1 JSON, run the standalone JSON engine `references/logicprobe-engine.py` (verify | compose | export — an exact Python mirror of the DSH tools, cross-checked byte-for-byte by tests/python/run.mjs); when the model exists only as extracted dicts, fill in the reusable template `references/verification-harness.py`. For detailed probe patterns, model extraction methodology, and manual verification procedures, load `references/logic-verification-guide.md`.
|
|
175
176
|
|
|
176
|
-
### Phase 2a: Structural Primitives (
|
|
177
|
+
### Phase 2a: Structural Primitives (8 Checks)
|
|
177
178
|
|
|
178
179
|
Run these FIRST. They establish basic well-formedness before adversarial probing.
|
|
179
180
|
|
|
@@ -188,7 +189,7 @@ Run these FIRST. They establish basic well-formedness before adversarial probing
|
|
|
188
189
|
| S7 | **Invariant validity** | Does every reachable state satisfy the plan's stated "always/never/guaranteed" assertions? | Error — plan claim is false |
|
|
189
190
|
| S8 | **Monotonic variables** | If a variable is declared `monotonic: inc/dec`, do all updates respect that direction? | Error — counter can move backwards |
|
|
190
191
|
|
|
191
|
-
### Phase 2b: Adversarial Probes (
|
|
192
|
+
### Phase 2b: Adversarial Probes (14 Attacks)
|
|
192
193
|
|
|
193
194
|
Run these SECOND. Each probe actively tries to BREAK the model. If any probe succeeds (finds a violation), the plan has a behavior gap.
|
|
194
195
|
|
|
@@ -197,7 +198,7 @@ Run these SECOND. Each probe actively tries to BREAK the model. If any probe suc
|
|
|
197
198
|
| A1 | **Unexpected event** | For each state S, inject every event E where no transition is defined for (S, E). Log whether the model silently ignores or crashes. | "All events are handled in all states" |
|
|
198
199
|
| A2 | **Race interleaving** | For every pair of concurrent events (E1, E2), simulate arrival in both orders: E1-then-E2 vs E2-then-E1. Flag if terminal state differs. | "Behavior is independent of event ordering" |
|
|
199
200
|
| A3 | **Order permutation** | For N independent events, permute arrival order. Flag if different permutations produce different final states or violate invariants. | "Outcome is order-independent" |
|
|
200
|
-
| A4 | **Pair symmetry** | Match every `start/stop`, `lock/unlock`, `alloc/free` pair. Flag if any state allows a path where a pair is unbalanced (start without stop, lock without unlock). | "Resources are always released" |
|
|
201
|
+
| A4 | **Pair symmetry** | Match every `start/stop`, `lock/unlock`, `alloc/free` pair. Flag if any state allows a path where a pair is unbalanced (start without stop, lock without unlock). State `onEntry`/`onExit` actions are treated as implicit acquire/release. | "Resources are always released" |
|
|
201
202
|
| A5 | **Boundary blast** | Probe counters at 0, 1, max-1, max, max+1. Probe timestamps at 0, tick_wraparound. Flag overflow, underflow, or undefined behavior. | "Handles all counter/timer values" |
|
|
202
203
|
| A6 | **Resource injection** | Simulate `malloc→NULL`, `queue→full`, `semaphore→timeout` at each state that calls them. Flag if any state has no recovery path. | "Graceful degradation under resource pressure" |
|
|
203
204
|
| A7 | **Minimal counter-example** | For any invariant that fails, find the SHORTEST event sequence that violates it (BFS from init to violating state). Output the exact path. | "This invariant holds" → refuted by shortest path |
|
|
@@ -205,6 +206,9 @@ Run these SECOND. Each probe actively tries to BREAK the model. If any probe suc
|
|
|
205
206
|
| A9 | **Leads-to** | From a declared source state, every path must eventually reach the target state. | "This state always progresses to completion" |
|
|
206
207
|
| A10 | **Sequence order** | Events declared in a sequence must occur in the specified order. | "backup before modify before commit" |
|
|
207
208
|
| A11 | **Atomicity** | Once an atomic event starts, the machine must commit or roll back before leaving the scope or terminating. | "all-or-nothing transaction" |
|
|
209
|
+
| A12 | **Budget (worst-case path cost)** | With `cost` on transitions (absent = 1) and a `budget` invariant, every reachable path must stay within budget; reports the shortest over-budget counterexample and flags reachable positive-cost cycles as unbounded. | "Worst-case path cost ≤ budget" |
|
|
210
|
+
| A13 | **Probability reachability (DTMC)** | With `weight` on transitions (absent = 1) and a `probability` invariant, P(ever hitting the target) must satisfy the bound; solved by value iteration over the absorbing chain. | "≥ 90% of runs reach SAFE" |
|
|
211
|
+
| A14 | **Deadline (discrete tick clock)** | State `maxTicks` plus top-level `tickEvents`: a tick step may not keep a state resident past its deadline; reports the over-residency path. | "must leave within 2 ticks" |
|
|
208
212
|
|
|
209
213
|
### Integration Back to Phase 3
|
|
210
214
|
|
|
@@ -323,6 +327,8 @@ logicprobe does **not** prove concurrency safety. It mines documents and plans f
|
|
|
323
327
|
|
|
324
328
|
In DSH, use the `logicprobe_concurrency_scan` tool. For manual review, follow `references/concurrency-risk-guide.md`.
|
|
325
329
|
|
|
330
|
+
For routing claims in dimensions logicprobe does not verify — hard real time (deadlines/periods), preemptive concurrency, hybrid control stability, probabilistic reliability, and execution-cost budgets — to dedicated tools, see `references/gap-routing-guide.md`. In `logicprobe_verify` reports, matching models carry informational `coverageNotes` with the same routing.
|
|
331
|
+
|
|
326
332
|
---
|
|
327
333
|
|
|
328
334
|
## Rules
|
|
@@ -52,3 +52,5 @@ Manual checklist:
|
|
|
52
52
|
- Java: JCStress, Java PathFinder
|
|
53
53
|
- General: TLA+, SPIN, Alloy
|
|
54
54
|
- Rust: loom, Shuttle
|
|
55
|
+
|
|
56
|
+
Other semantic dimensions (hard real time, hybrid control, probability/reliability, execution cost) route through the same pattern — see `references/gap-routing-guide.md`.
|