dsh-logicprobe 0.5.6 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -21
- package/README.en-US.md +6 -3
- package/README.md +5 -2
- package/lib/compose-tool.js +43 -0
- package/lib/concurrency.js +11 -1
- package/lib/engine.js +761 -11
- package/lib/export-tool.js +45 -0
- package/lib/exporters.js +389 -0
- package/lib/index.js +32 -20
- package/lib/tool.js +6 -5
- package/lib/types/compose-tool.d.ts +6 -0
- package/lib/types/concurrency.d.ts +2 -0
- package/lib/types/engine.d.ts +66 -0
- package/lib/types/export-tool.d.ts +7 -0
- package/lib/types/exporters.d.ts +17 -0
- package/lib/types/tool.d.ts +5 -4
- package/package.json +106 -103
- package/skills/logicprobe/SKILL.md +14 -8
- package/skills/logicprobe/references/concurrency-risk-guide.md +2 -0
- package/skills/logicprobe/references/dsh-model-schema.md +266 -204
- package/skills/logicprobe/references/gap-routing-guide.md +36 -0
- package/skills/logicprobe/references/logic-verification-guide.md +56 -4
- package/skills/logicprobe/references/logicprobe-engine.py +3028 -0
- package/skills/logicprobe/references/verification-harness.py +153 -0
- package/src/compose-tool.ts +46 -0
- package/src/concurrency.ts +13 -1
- package/src/engine.ts +2568 -1778
- package/src/export-tool.ts +47 -0
- package/src/exporters.ts +347 -0
- package/src/index.ts +349 -337
- package/src/tool.ts +61 -60
- package/skills/logicprobe/references/__pycache__/verification-harness.cpython-312.pyc +0 -0
- package/skills/logicprobe-datamodel/references/__pycache__/data-model-harness.cpython-312.pyc +0 -0
|
@@ -4,6 +4,15 @@
|
|
|
4
4
|
Fill in the MODEL section below with states, transitions, invariants extracted from the plan.
|
|
5
5
|
Run: python3 verification-harness.py
|
|
6
6
|
Output: structured verification report for Phase 3 gap analysis.
|
|
7
|
+
|
|
8
|
+
Covers all 22 checks: S1-S8 (structural), A1-A14 (adversarial), plus D1-D4
|
|
9
|
+
(before/after regression) when BEFORE_STATES is filled in. A12 (budget),
|
|
10
|
+
A13 (probability reachability) and A14 (deadline) use the optional config
|
|
11
|
+
lists below; leave them empty to skip the corresponding probe.
|
|
12
|
+
|
|
13
|
+
For multi-machine composition (C1/C2) and external-tool export (UPPAAL/TLA+/
|
|
14
|
+
PRISM/SPIN), use the standalone JSON-driven engine instead:
|
|
15
|
+
references/logicprobe-engine.py (verify | compose | export)
|
|
7
16
|
"""
|
|
8
17
|
import sys
|
|
9
18
|
from collections import deque
|
|
@@ -61,6 +70,24 @@ SEQUENCES: list[list[str]] = []
|
|
|
61
70
|
# Atomic groups (for A11): {"events": [...], "commit": "...", "rollback": "..."}
|
|
62
71
|
ATOMIC_GROUPS: list[dict] = []
|
|
63
72
|
|
|
73
|
+
# Transition costs for A12 budget (optional): {(from_state, event): cost}; absent entries count as 1
|
|
74
|
+
TRANSITION_COSTS: dict[tuple[str, str], int] = {}
|
|
75
|
+
|
|
76
|
+
# Budget invariants for A12 (optional): {"id": "...", "description": "...", "budget": N}
|
|
77
|
+
BUDGETS: list[dict] = []
|
|
78
|
+
|
|
79
|
+
# Probability weights for A13 DTMC (optional): {(from_state, event): weight}; absent = 1, weight 0 never fires
|
|
80
|
+
TRANSITION_WEIGHTS: dict[tuple[str, str], int] = {}
|
|
81
|
+
|
|
82
|
+
# Probability reachability claims for A13 (optional): {"id": "...", "description": "...", "target": "STATE", "op": ">=", "p": 0.9}
|
|
83
|
+
PROBABILITY_INVARIANTS: list[dict] = []
|
|
84
|
+
|
|
85
|
+
# Events that advance the discrete clock for A14 (optional): {"tick"}
|
|
86
|
+
TICK_EVENTS: set[str] = set()
|
|
87
|
+
|
|
88
|
+
# Per-state residency deadlines for A14 (optional): {"BUSY": 2} means BUSY must be left within 2 ticks of entry
|
|
89
|
+
STATE_MAX_TICKS: dict[str, int] = {}
|
|
90
|
+
|
|
64
91
|
# Before model for D1-D4 comparison (optional)
|
|
65
92
|
BEFORE_STATES: dict[str, dict[str, str]] = {}
|
|
66
93
|
BEFORE_INIT: str = "INIT"
|
|
@@ -646,6 +673,129 @@ def A11_atomicity():
|
|
|
646
673
|
return {'pass': len(findings) == 0, 'findings': findings, 'detail': 'Atomicity invariants hold' if not findings else f'{len(findings)} atomicity violations'}
|
|
647
674
|
|
|
648
675
|
|
|
676
|
+
def A12_budget():
|
|
677
|
+
"""A12: worst-case path cost vs declared budgets."""
|
|
678
|
+
if not BUDGETS:
|
|
679
|
+
return {'pass': True, 'findings': [], 'detail': 'No budgets declared — skipped'}
|
|
680
|
+
findings = []
|
|
681
|
+
for budget in BUDGETS:
|
|
682
|
+
budget_val = budget.get('budget', 0)
|
|
683
|
+
# BFS over accumulated cost; first over-budget path is a witness.
|
|
684
|
+
queue = deque([(INIT, 0, [])])
|
|
685
|
+
best_cost = {INIT: 0}
|
|
686
|
+
violation = None
|
|
687
|
+
while queue and violation is None:
|
|
688
|
+
state, cost, path = queue.popleft()
|
|
689
|
+
for event, nxt in STATES.get(state, {}).items():
|
|
690
|
+
edge_cost = TRANSITION_COSTS.get((state, event), 1)
|
|
691
|
+
total = cost + edge_cost
|
|
692
|
+
new_path = path + [(state, event, nxt)]
|
|
693
|
+
if total > budget_val:
|
|
694
|
+
violation = (new_path, total)
|
|
695
|
+
break
|
|
696
|
+
if best_cost.get(nxt) is None or total < best_cost[nxt]:
|
|
697
|
+
best_cost[nxt] = total
|
|
698
|
+
queue.append((nxt, total, new_path))
|
|
699
|
+
if violation is not None:
|
|
700
|
+
findings.append({'code': 'A12_BUDGET_OVER', 'severity': 'error',
|
|
701
|
+
'message': f"Budget '{budget.get('id')}' exceeded: path cost {violation[1]} over budget {budget_val}",
|
|
702
|
+
'evidence': {'budget': budget, 'path': violation[0], 'totalCost': violation[1]}})
|
|
703
|
+
return {'pass': len(findings) == 0, 'findings': findings,
|
|
704
|
+
'detail': 'All budgets respected' if not findings else f'{len(findings)} budget violations'}
|
|
705
|
+
|
|
706
|
+
|
|
707
|
+
def A13_probability():
|
|
708
|
+
"""A13: P(ever reaching target) from INIT under the DTMC induced by weights."""
|
|
709
|
+
if not PROBABILITY_INVARIANTS:
|
|
710
|
+
return {'pass': True, 'findings': [], 'detail': 'No probability invariants — skipped'}
|
|
711
|
+
reachable = set()
|
|
712
|
+
queue = deque([INIT])
|
|
713
|
+
while queue:
|
|
714
|
+
state = queue.popleft()
|
|
715
|
+
if state in reachable:
|
|
716
|
+
continue
|
|
717
|
+
reachable.add(state)
|
|
718
|
+
for nxt in STATES.get(state, {}).values():
|
|
719
|
+
if nxt not in reachable:
|
|
720
|
+
queue.append(nxt)
|
|
721
|
+
states = sorted(reachable)
|
|
722
|
+
findings = []
|
|
723
|
+
eps = 1e-9
|
|
724
|
+
for inv in PROBABILITY_INVARIANTS:
|
|
725
|
+
target = inv.get('target')
|
|
726
|
+
# value iteration over the reachable absorbing chain
|
|
727
|
+
prob = {s: 0.0 for s in states}
|
|
728
|
+
for s in states:
|
|
729
|
+
if s == target:
|
|
730
|
+
prob[s] = 1.0
|
|
731
|
+
for _ in range(20000):
|
|
732
|
+
delta = 0.0
|
|
733
|
+
for s in states:
|
|
734
|
+
if s == target:
|
|
735
|
+
continue
|
|
736
|
+
total_w = 0
|
|
737
|
+
acc = 0.0
|
|
738
|
+
for event, nxt in STATES.get(s, {}).items():
|
|
739
|
+
w = TRANSITION_WEIGHTS.get((s, event), 1)
|
|
740
|
+
if w <= 0:
|
|
741
|
+
continue
|
|
742
|
+
total_w += w
|
|
743
|
+
acc += w * prob[nxt]
|
|
744
|
+
newp = acc / total_w if total_w > 0 else 0.0
|
|
745
|
+
delta = max(delta, abs(newp - prob[s]))
|
|
746
|
+
prob[s] = newp
|
|
747
|
+
if delta < eps:
|
|
748
|
+
break
|
|
749
|
+
p_hit = prob.get(INIT, 0.0)
|
|
750
|
+
op = inv.get('op', '>=')
|
|
751
|
+
bound = inv.get('p', 0.0)
|
|
752
|
+
violated = (op == '>=' and p_hit < bound - eps) or (op == '>' and p_hit <= bound + eps) or (op == '<=' and p_hit > bound + eps) or (op == '<' and p_hit >= bound - eps)
|
|
753
|
+
if violated:
|
|
754
|
+
findings.append({'code': 'A13_PROBABILITY_VIOLATION', 'severity': 'error',
|
|
755
|
+
'message': f"P(hit {target}) = {p_hit:.6f} does not satisfy {op} {bound}",
|
|
756
|
+
'evidence': {'invariant': inv, 'computed': p_hit}})
|
|
757
|
+
return {'pass': len(findings) == 0, 'findings': findings,
|
|
758
|
+
'detail': 'Probability invariants hold' if not findings else f'{len(findings)} probability violations'}
|
|
759
|
+
|
|
760
|
+
|
|
761
|
+
def A14_deadline():
|
|
762
|
+
"""A14: a state with a maxTicks deadline may not be kept resident past it by tick events."""
|
|
763
|
+
if not STATE_MAX_TICKS:
|
|
764
|
+
return {'pass': True, 'findings': [], 'detail': 'No state deadlines declared — skipped'}
|
|
765
|
+
findings = []
|
|
766
|
+
if not TICK_EVENTS:
|
|
767
|
+
return {'pass': False, 'findings': [{'code': 'A14_NO_TICK_EVENTS', 'severity': 'warning',
|
|
768
|
+
'message': 'maxTicks declared but no tick events — cannot verify'}],
|
|
769
|
+
'detail': 'Missing tick events'}
|
|
770
|
+
# Residency ticks, mirroring logicprobe A14: entering a state resets to 0;
|
|
771
|
+
# a tick that stays in the state advances the count; a tick that leaves or any
|
|
772
|
+
# non-tick transition to another state resets; a non-tick self-loop keeps it.
|
|
773
|
+
visited = set()
|
|
774
|
+
queue = deque([(INIT, 0, [])])
|
|
775
|
+
while queue:
|
|
776
|
+
state, ticks, path = queue.popleft()
|
|
777
|
+
key = (state, ticks)
|
|
778
|
+
if key in visited:
|
|
779
|
+
continue
|
|
780
|
+
visited.add(key)
|
|
781
|
+
limit = STATE_MAX_TICKS.get(state)
|
|
782
|
+
if limit is not None and ticks > limit:
|
|
783
|
+
findings.append({'code': 'A14_DEADLINE_MISS', 'severity': 'error',
|
|
784
|
+
'message': f'State {state} resident for {ticks} ticks, limit {limit}',
|
|
785
|
+
'evidence': {'state': state, 'maxTicks': limit, 'ticks': ticks, 'path': path}})
|
|
786
|
+
continue
|
|
787
|
+
for event, nxt in STATES.get(state, {}).items():
|
|
788
|
+
if event in TICK_EVENTS:
|
|
789
|
+
nxt_ticks = ticks + 1 if nxt == state else 0
|
|
790
|
+
elif nxt == state:
|
|
791
|
+
nxt_ticks = ticks
|
|
792
|
+
else:
|
|
793
|
+
nxt_ticks = 0
|
|
794
|
+
queue.append((nxt, nxt_ticks, path + [(state, event, nxt)]))
|
|
795
|
+
return {'pass': len(findings) == 0, 'findings': findings,
|
|
796
|
+
'detail': 'All deadlines respected' if not findings else f'{len(findings)} deadline misses'}
|
|
797
|
+
|
|
798
|
+
|
|
649
799
|
def D1_behavioral_preservation():
|
|
650
800
|
findings = []
|
|
651
801
|
if not BEFORE_STATES:
|
|
@@ -752,6 +902,9 @@ def run_all():
|
|
|
752
902
|
("A9 Leads-To", A9_leads_to),
|
|
753
903
|
("A10 Sequence Order", A10_sequence_order),
|
|
754
904
|
("A11 Atomicity", A11_atomicity),
|
|
905
|
+
("A12 Budget", A12_budget),
|
|
906
|
+
("A13 Probability Reachability", A13_probability),
|
|
907
|
+
("A14 Deadline", A14_deadline),
|
|
755
908
|
]
|
|
756
909
|
|
|
757
910
|
for name, probe_fn in probes_2b:
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
import { defineTool, type JsonValue } from '@deepseek-ai/dsh-tools'
|
|
2
|
+
import { runCompositionVerification } from './engine.js'
|
|
3
|
+
|
|
4
|
+
export const LOGICPROBE_COMPOSE_TOOL_NAME = 'logicprobe_compose_verify'
|
|
5
|
+
|
|
6
|
+
/**
|
|
7
|
+
* DSH tool wrapping multi-machine composition verification:
|
|
8
|
+
* product-space reachability with rendezvous handshakes (C1 deadlock, C2 sync).
|
|
9
|
+
*/
|
|
10
|
+
export const logicProbeComposeTool = defineTool({
|
|
11
|
+
name: LOGICPROBE_COMPOSE_TOOL_NAME,
|
|
12
|
+
description:
|
|
13
|
+
'Run composition verification over two or more LogicModelV1 state machines (logicprobe). Pass machines as an array of models and optionally rendezvous: a list of handshake events that must fire simultaneously across all machines declaring them (at least two participants, all jointly enabled, guards held; a terminal machine stops participating). Non-rendezvous events advance exactly one firing machine. Returns C1 composition-deadlock findings (a reachable state where no machine can advance while at least one is not terminal) and C2 rendezvous-never-fires findings. Each machine validates against the same schema as logicprobe_verify.',
|
|
14
|
+
parameters: {
|
|
15
|
+
machines: {
|
|
16
|
+
type: 'json',
|
|
17
|
+
required: true,
|
|
18
|
+
description: 'Array of LogicModelV1 state-machine models to compose (two or more).',
|
|
19
|
+
},
|
|
20
|
+
rendezvous: {
|
|
21
|
+
type: 'json',
|
|
22
|
+
description: 'Optional array of handshake event names shared by the machines.',
|
|
23
|
+
},
|
|
24
|
+
maxStates: {
|
|
25
|
+
type: 'integer',
|
|
26
|
+
description: 'Maximum composite states to explore. Default 10000.',
|
|
27
|
+
},
|
|
28
|
+
},
|
|
29
|
+
output: {
|
|
30
|
+
schema: {
|
|
31
|
+
type: 'json',
|
|
32
|
+
description: 'Composition verification report with C1/C2 findings.',
|
|
33
|
+
},
|
|
34
|
+
render(_args, value) {
|
|
35
|
+
return [{ type: 'text' as const, text: JSON.stringify(value, null, 2) }]
|
|
36
|
+
},
|
|
37
|
+
},
|
|
38
|
+
timeoutMs: 10_000,
|
|
39
|
+
isConcurrencySafe: () => true,
|
|
40
|
+
async execute(args) {
|
|
41
|
+
return runCompositionVerification(args.machines as unknown[], {
|
|
42
|
+
rendezvous: args.rendezvous as string[] | undefined,
|
|
43
|
+
maxStates: args.maxStates,
|
|
44
|
+
}) as unknown as JsonValue
|
|
45
|
+
},
|
|
46
|
+
})
|
package/src/concurrency.ts
CHANGED
|
@@ -5,6 +5,8 @@ export interface ConcurrencyFinding {
|
|
|
5
5
|
line?: number
|
|
6
6
|
snippet?: string
|
|
7
7
|
keyword: string
|
|
8
|
+
/** Route to dedicated verification tools when an absolute claim is detected (logicprobe does not prove concurrency safety). */
|
|
9
|
+
suggestions?: string[]
|
|
8
10
|
}
|
|
9
11
|
|
|
10
12
|
export interface ConcurrencyScanReport {
|
|
@@ -64,6 +66,15 @@ export function runConcurrencyScan(text: string): ConcurrencyScanReport {
|
|
|
64
66
|
const lines = text.split(/\r?\n/)
|
|
65
67
|
const findings: ConcurrencyFinding[] = []
|
|
66
68
|
const seen = new Set<string>()
|
|
69
|
+
const suggestionsFor = (label: string): string[] => {
|
|
70
|
+
if (/thread-safe|lock-free|wait-free|race-free|no data race|data race|race condition|thread safety/.test(label)) {
|
|
71
|
+
return ['TSan / Helgrind (runtime data-race detection)', 'CBMC / static analysis (proof-oriented)', 'TLA+ (exhaustive interleaving model)']
|
|
72
|
+
}
|
|
73
|
+
if (/interrupt-safe|ISR-safe|interrupt context|critical section|disable_irq|enable_irq/.test(label)) {
|
|
74
|
+
return ['RTOS-aware analysis (interrupt latency, priority inversion)', 'CBMC (state-machine + ISR interleaving proof)', 'TLA+ (preemption/exclusion model)']
|
|
75
|
+
}
|
|
76
|
+
return ['Dedicated concurrency verification (TSan, CBMC, TLA+, or RTOS-specific analysis)']
|
|
77
|
+
}
|
|
67
78
|
lines.forEach((line, index) => {
|
|
68
79
|
const lineNumber = index + 1
|
|
69
80
|
const lower = line.toLowerCase()
|
|
@@ -76,11 +87,12 @@ export function runConcurrencyScan(text: string): ConcurrencyScanReport {
|
|
|
76
87
|
code: rule.absolute ? 'CONCURRENCY_ABSOLUTE_CLAIM' : 'CONCURRENCY_KEYWORD',
|
|
77
88
|
severity: rule.absolute ? 'error' : 'warning',
|
|
78
89
|
message: rule.absolute
|
|
79
|
-
? 'Concurrency safety claim "' + rule.label + '" detected;
|
|
90
|
+
? 'Concurrency safety claim "' + rule.label + '" detected; logicprobe does not verify concurrency — this needs dedicated verification.'
|
|
80
91
|
: 'Concurrency-related term "' + rule.label + '" detected; review whether the plan addresses this risk.',
|
|
81
92
|
line: lineNumber,
|
|
82
93
|
snippet: line.trim().slice(0, 200),
|
|
83
94
|
keyword: rule.label,
|
|
95
|
+
...(rule.absolute ? { suggestions: suggestionsFor(rule.label) } : {}),
|
|
84
96
|
}
|
|
85
97
|
findings.push(finding)
|
|
86
98
|
}
|