dsh-logicprobe 0.7.1 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.en-US.md +89 -58
- package/README.md +76 -46
- package/lib/index.js +11 -4
- package/lib/json-value.js +18 -0
- package/lib/types/json-value.d.ts +21 -0
- package/lib/types/uml-tool.d.ts +14 -0
- package/lib/types/uml.d.ts +167 -0
- package/lib/uml-tool.js +104 -0
- package/lib/uml.js +1544 -0
- package/package.json +11 -10
- package/skills/logicprobe/SKILL.md +36 -1
- package/skills/logicprobe/references/logicprobe-engine.py +1586 -2
- package/skills/logicprobe/references/uml-modeling-guide.md +166 -0
- package/src/compose-tool.ts +2 -1
- package/src/concurrency-tool.ts +2 -1
- package/src/data-tool.ts +2 -1
- package/src/export-tool.ts +2 -1
- package/src/index.ts +11 -4
- package/src/json-value.ts +20 -0
- package/src/tool.ts +2 -1
- package/src/uml-tool.ts +106 -0
- package/src/uml.ts +1461 -0
- package/skills/logicprobe/references/__pycache__/logicprobe-engine.cpython-310.pyc +0 -0
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "dsh-logicprobe",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.8.0",
|
|
4
4
|
"description": "Design document & plan claim verification — enumerate claims, verify against codebase facts, then escalate to state-machine verification (S1-S8/A1-A12, including budget/worst-case path-cost checks) and data-model verification (DS/DA/DD) for behavioral claims. Supports before/after regression, idempotency/monotonic/sequence/leads-to/atomicity constraints, and concurrency risk mining. Ships a native DeepSeek Harness (dsh) bundle that injects the claim-verification gate into the first model step.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "lib/index.js",
|
|
@@ -38,7 +38,7 @@
|
|
|
38
38
|
]
|
|
39
39
|
},
|
|
40
40
|
"compatibility": {
|
|
41
|
-
"dsh": "^0.1.0-rc.7 || ^0.1.1-rc.1 || ^0.1.2-alpha.2 || ^0.1.2-alpha.3 || ^0.1.2-alpha.4 || ^0.1.2-alpha.5 || ^0.1.2-rc.1 || ^0.1.3-alpha.1 || ^0.1.3-alpha.2 || ^0.1.5-alpha.1 || ^0.1.5-rc.1 || ^0.1.5-alpha.2 || ^0.1.5-rc.2 || ^0.1.5-rc.3 || ^0.1.6-alpha.1 || ^0.1.6-alpha.2 || ^0.1.7-alpha.1 || ^0.1.7-alpha.2 || ^0.1.7-rc.1 || ^0.1.7-rc.2 || ^0.2.0-rc.1",
|
|
41
|
+
"dsh": "^0.1.0-rc.7 || ^0.1.1-rc.1 || ^0.1.2-alpha.2 || ^0.1.2-alpha.3 || ^0.1.2-alpha.4 || ^0.1.2-alpha.5 || ^0.1.2-rc.1 || ^0.1.3-alpha.1 || ^0.1.3-alpha.2 || ^0.1.5-alpha.1 || ^0.1.5-rc.1 || ^0.1.5-alpha.2 || ^0.1.5-rc.2 || ^0.1.5-rc.3 || ^0.1.6-alpha.1 || ^0.1.6-alpha.2 || ^0.1.7-alpha.1 || ^0.1.7-alpha.2 || ^0.1.7-rc.1 || ^0.1.7-rc.2 || ^0.2.0-rc.1 || ^0.2.1-alpha.1",
|
|
42
42
|
"dshReleases": {
|
|
43
43
|
"0.1.0-rc.7": "compatible",
|
|
44
44
|
"0.1.0-rc.8": "compatible",
|
|
@@ -63,7 +63,8 @@
|
|
|
63
63
|
"0.1.7-rc.1": "compatible",
|
|
64
64
|
"0.1.7-rc.2": "compatible",
|
|
65
65
|
"0.2.0-rc.1": "compatible",
|
|
66
|
-
"0.2.0-rc.2": "compatible"
|
|
66
|
+
"0.2.0-rc.2": "compatible",
|
|
67
|
+
"0.2.1-alpha.1": "compatible"
|
|
67
68
|
},
|
|
68
69
|
"profiles": [
|
|
69
70
|
"headless",
|
|
@@ -75,17 +76,17 @@
|
|
|
75
76
|
"build": "tsc -p tsconfig.json && node scripts/build-client.mjs",
|
|
76
77
|
"typecheck": "tsc -p tsconfig.json --noEmit",
|
|
77
78
|
"bump": "node scripts/bump-version.mjs",
|
|
78
|
-
"test:engine": "npm run build && node tests/engine/run.mjs && node tests/data-engine/run.mjs && node tests/concurrency/run.mjs && node tests/apply-smoke.mjs && node tests/dsh-client-half.test.mjs && node tests/dsh-volatile-fallback.test.mjs && node tests/exporters/run.mjs && node tests/external/run.mjs && node tests/python/run.mjs",
|
|
79
|
+
"test:engine": "npm run build && node tests/engine/run.mjs && node tests/data-engine/run.mjs && node tests/concurrency/run.mjs && node tests/uml/run.mjs && node tests/apply-smoke.mjs && node tests/dsh-client-half.test.mjs && node tests/dsh-volatile-fallback.test.mjs && node tests/exporters/run.mjs && node tests/external/run.mjs && node tests/python/run.mjs",
|
|
79
80
|
"test:full": "npm run build && node tests/full-suite.mjs",
|
|
80
81
|
"test:python": "npm run build && node tests/python/run.mjs"
|
|
81
82
|
},
|
|
82
83
|
"peerDependencies": {
|
|
83
84
|
"@deepseek-ai/cordis": "^4.0.4",
|
|
84
|
-
"@deepseek-ai/dsh-agent": "^0.1.0-rc.6 || ^0.2.0-rc.1",
|
|
85
|
-
"@deepseek-ai/dsh-llm": "^0.1.0-rc.6 || ^0.2.0-rc.1",
|
|
86
|
-
"@deepseek-ai/dsh-session": "^0.1.0-rc.6 || ^0.2.0-rc.1",
|
|
87
|
-
"@deepseek-ai/dsh-skill-filesystem": "^0.1.0-rc.8 || ^0.2.0-rc.1",
|
|
88
|
-
"@deepseek-ai/dsh-tools": "^0.1.0-rc.6 || ^0.2.0-rc.1",
|
|
85
|
+
"@deepseek-ai/dsh-agent": "^0.1.0-rc.6 || ^0.2.0-rc.1 || ^0.2.1-alpha.1",
|
|
86
|
+
"@deepseek-ai/dsh-llm": "^0.1.0-rc.6 || ^0.2.0-rc.1 || ^0.2.1-alpha.1",
|
|
87
|
+
"@deepseek-ai/dsh-session": "^0.1.0-rc.6 || ^0.2.0-rc.1 || ^0.2.1-alpha.1",
|
|
88
|
+
"@deepseek-ai/dsh-skill-filesystem": "^0.1.0-rc.8 || ^0.2.0-rc.1 || ^0.2.1-alpha.1",
|
|
89
|
+
"@deepseek-ai/dsh-tools": "^0.1.0-rc.6 || ^0.2.0-rc.1 || ^0.2.1-alpha.1",
|
|
89
90
|
"@deepseek-ai/schemastery": "^3.18.4"
|
|
90
91
|
},
|
|
91
92
|
"devDependencies": {
|
|
@@ -102,7 +103,7 @@
|
|
|
102
103
|
"@deepseek-ai/dsh-timeout": "^0.1.0-rc.6",
|
|
103
104
|
"@deepseek-ai/dsh-tools": "^0.1.0-rc.6",
|
|
104
105
|
"@deepseek-ai/schemastery": "^3.18.4",
|
|
105
|
-
"@types/node": "^26.6.
|
|
106
|
+
"@types/node": "^26.6.3",
|
|
106
107
|
"typescript": "^7.0.2"
|
|
107
108
|
},
|
|
108
109
|
"author": {
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: logicprobe
|
|
3
|
-
description: "Use when reviewing design documents, architecture specs, technical proposals, or refactoring plans that make claims about API names, file locations, enum values, or mechanism feasibility. When the document contains state machines, protocol logic, or behavioral claims (≥3 states, ACK/NACK/retry sequences, 'always'/'never'/'guaranteed' assertions, or refactoring that modifies state topology), escalate into logic-primitive verification — generate and run executable models to check mathematical completeness before trusting any claim. For refactoring specifically, the pipeline compares before/after models to verify behavioral preservation and regression freedom. ALSO proactively SUGGEST this skill (do not require) when a user asks code-level behavioral questions — 'check this timing for bugs', 'could this state machine deadlock', 'is this retry limit safe' — since plan-level verification has usually already been done."
|
|
3
|
+
description: "Use when reviewing design documents, architecture specs, technical proposals, or refactoring plans that make claims about API names, file locations, enum values, or mechanism feasibility. When the document contains state machines, protocol logic, or behavioral claims (≥3 states, ACK/NACK/retry sequences, 'always'/'never'/'guaranteed' assertions, or refactoring that modifies state topology), escalate into logic-primitive verification — generate and run executable models to check mathematical completeness before trusting any claim. For refactoring specifically, the pipeline compares before/after models to verify behavioral preservation and regression freedom. ALSO use when the task is to model a code flow as UML, or to audit a diagram somebody drew. That covers reverse-engineering a handler and documenting a protocol. The skill renders the model as a state, activity or sequence diagram, reads a hand-drawn diagram back into a model, and reviews the modelling: unreachable states, dead ends, ambiguous branches, documentation gaps, and diagram-versus-model fidelity. ALSO proactively SUGGEST this skill (do not require) when a user asks code-level behavioral questions — 'check this timing for bugs', 'could this state machine deadlock', 'is this retry limit safe' — since plan-level verification has usually already been done."
|
|
4
4
|
---
|
|
5
5
|
|
|
6
6
|
# Logic Probe
|
|
@@ -140,6 +140,11 @@ Refactoring variant:
|
|
|
140
140
|
→ Run pipeline on AFTER model (22 checks)
|
|
141
141
|
→ Compare BEFORE vs AFTER: behavioral preservation, regression, complexity delta
|
|
142
142
|
→ Flag any invariant that held in BEFORE but fails in AFTER
|
|
143
|
+
|
|
144
|
+
UML modelling variant (the task is to draw a flow, not to check a claim):
|
|
145
|
+
Code flow → model with a citation per element → logicprobe_uml action=render → diagram
|
|
146
|
+
→ logicprobe_uml action=review → modelling findings + render/parse round-trip fidelity
|
|
147
|
+
→ logicprobe_verify → the behavioural checks (S1-S8 / A1-A14)
|
|
143
148
|
```
|
|
144
149
|
|
|
145
150
|
### Refactoring Verification Mode
|
|
@@ -307,6 +312,35 @@ If the user says yes, extract the model from the existing code (not a plan docum
|
|
|
307
312
|
|
|
308
313
|
This covers the gap where behavioral verification is useful even when no design document is being reviewed.
|
|
309
314
|
|
|
315
|
+
## UML Modelling and Modelling Review
|
|
316
|
+
|
|
317
|
+
Use this when the task is to **draw** a code flow rather than to check a claim. Typical cases are reverse-engineering a handler, documenting a protocol exchange, and auditing a diagram somebody else drew.
|
|
318
|
+
|
|
319
|
+
A diagram is a model, and a model can be wrong. A tidy flow chart is not evidence about the code. A diagram that disagrees with its own model is worse than no diagram, because a reader will believe it.
|
|
320
|
+
|
|
321
|
+
Pipeline (DSH):
|
|
322
|
+
|
|
323
|
+
```text
|
|
324
|
+
code flow → model (citation per element) → logicprobe_uml action=render → diagram source
|
|
325
|
+
→ logicprobe_uml action=review → modelling findings + round-trip fidelity
|
|
326
|
+
→ logicprobe_verify → behaviour (S1-S8 structural, A1-A14 adversarial)
|
|
327
|
+
```
|
|
328
|
+
|
|
329
|
+
- **render** — model to diagram. Mermaid covers `state`, `activity` and `sequence`. PlantUML covers `state` and `sequence`. Any construct the notation cannot carry becomes a warning, never a silent drop. PlantUML activity is refused instead of approximated.
|
|
330
|
+
- **parse** — diagram to model. It reads Mermaid and PlantUML state or activity text, so a hand-drawn diagram can be verified like any other model. A sequence diagram is refused, because a trace cannot reconstruct a machine.
|
|
331
|
+
- **review** — it answers one of three questions. Give it a model: is the machine well-modelled? Give it a diagram: what does the diagram say? Give it both: does the diagram match the model?
|
|
332
|
+
|
|
333
|
+
The review reports structural defects and documentation gaps. The structural defects are unreachable states, dead ends, ambiguous or non-exhaustive branches, self-loops with no exit, and duplicate transitions. The documentation gaps are a missing narrative, unbounded variables, states the reader cannot map back to code, and label drift. It also runs the fidelity check. Any structural difference between the diagram and its model is `UML017_ROUND_TRIP_MISMATCH`.
|
|
334
|
+
|
|
335
|
+
Rules for this mode:
|
|
336
|
+
|
|
337
|
+
1. **A diagram is not evidence.** Every state, event, guard and action needs a citation. Use `file:line` for code and a section reference for a document. Present the citations with the diagram.
|
|
338
|
+
2. **Review before you present.** Run `logicprobe_uml action=review` and fix the error findings first. An ambiguous or dead-ended diagram misleads every later reader.
|
|
339
|
+
3. **The review never replaces verification.** It covers the modelling. S1-S8 and A1-A14 cover the behaviour, and each finding names the check that settles it.
|
|
340
|
+
4. **Keep the narrative with the model.** Write `narrative.states`, `narrative.events` and `narrative.scenarios`. Then the diagram stays readable against the code, and label drift shows up as a finding instead of as a stale picture.
|
|
341
|
+
|
|
342
|
+
The full checklist (codes `UML001` to `UML019`), the directive format generated diagrams carry, a worked example, and the limits of each view are in `references/uml-modeling-guide.md`.
|
|
343
|
+
|
|
310
344
|
### When NOT to Escalate
|
|
311
345
|
|
|
312
346
|
Skip logic-primitive verification when:
|
|
@@ -340,3 +374,4 @@ For routing claims in dimensions logicprobe does not verify — hard real time (
|
|
|
340
374
|
5. **For behavioral claims: verify with code, not reasoning.** If a plan says "always", "never", or "guaranteed", generate and run a model. One counter-example is enough to refute a universal claim.
|
|
341
375
|
6. **Confirm the model before running it** — unless the runtime reports `logicprobe interaction=auto`. Extraction errors are the dominant failure mode of formal verification. In auto mode, substitute evidence-cited extraction + round-trip validation and mark the report `UNCONFIRMED`.
|
|
342
376
|
7. **Don't verify what the code already checks.** If the existing codebase has compile-time assertions, static analysis, or runtime checks for a property, cite those — don't re-verify in a Python model.
|
|
377
|
+
8. **A diagram is a model, not evidence.** When you model a code flow as UML, cite the source for every state, event, guard and action. Run `logicprobe_uml` with `action=review` before you show the diagram. A diagram that does not read back as the model it was drawn from is mis-modelled, however tidy it looks.
|