dsh-logicprobe 0.4.0 → 0.5.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (33) hide show
  1. package/README.en-US.md +10 -3
  2. package/README.md +10 -3
  3. package/lib/concurrency-tool.js +34 -0
  4. package/lib/concurrency.js +76 -0
  5. package/lib/data-engine.js +930 -0
  6. package/lib/data-tool.js +61 -0
  7. package/lib/engine.js +258 -0
  8. package/lib/index.js +37 -23
  9. package/lib/tool.js +1 -1
  10. package/lib/types/concurrency-tool.d.ts +8 -0
  11. package/lib/types/concurrency.d.ts +20 -0
  12. package/lib/types/data-engine.d.ts +199 -0
  13. package/lib/types/data-tool.d.ts +10 -0
  14. package/lib/types/engine.d.ts +20 -0
  15. package/package.json +82 -81
  16. package/skills/logicprobe/SKILL.md +285 -268
  17. package/skills/logicprobe/references/__pycache__/verification-harness.cpython-312.pyc +0 -0
  18. package/skills/logicprobe/references/concurrency-risk-guide.md +54 -0
  19. package/skills/logicprobe/references/dsh-model-schema.md +145 -129
  20. package/skills/logicprobe/references/logic-verification-guide.md +463 -413
  21. package/skills/logicprobe/references/verification-harness.py +806 -582
  22. package/skills/logicprobe-datamodel/SKILL.md +124 -0
  23. package/skills/logicprobe-datamodel/references/__pycache__/data-model-harness.cpython-312.pyc +0 -0
  24. package/skills/logicprobe-datamodel/references/data-model-guide.md +62 -0
  25. package/skills/logicprobe-datamodel/references/data-model-harness.py +528 -0
  26. package/skills/logicprobe-datamodel/references/data-model-schema.md +128 -0
  27. package/src/concurrency-tool.ts +37 -0
  28. package/src/concurrency.ts +102 -0
  29. package/src/data-engine.ts +1001 -0
  30. package/src/data-tool.ts +65 -0
  31. package/src/engine.ts +234 -0
  32. package/src/index.ts +315 -301
  33. package/src/tool.ts +60 -60
package/README.en-US.md CHANGED
@@ -13,9 +13,11 @@ Design documents are not truth — code is. A claim-verification skill that chec
13
13
  | Phase | What |
14
14
  |-------|------|
15
15
  | Phase 1-2 | Enumerate every verifiable claim (API names, file paths, enum values, counts, mechanism feasibility) → verify each against the codebase with evidence |
16
- | Phase 2a | **7 structural checks** on extracted state-machine models: reachability, deadlock, liveness, determinism, event/guard completeness, invariant validity |
17
- | Phase 2b | **7 adversarial probes**: unexpected events, race interleaving, order permutation, pair symmetry (lock/unlock), boundary blast, resource injection, minimal counter-example |
16
+ | Phase 2a | **8 structural checks** on extracted state-machine models: reachability, deadlock, liveness, determinism, event/guard completeness, invariant validity, monotonic variables |
17
+ | Phase 2b | **11 adversarial probes**: unexpected events, race interleaving, order permutation, pair symmetry (lock/unlock), boundary blast, resource injection, minimal counter-example, idempotent replay, leads-to, sequence, atomicity |
18
18
  | Refactoring | Before/after model comparison — behavioral preservation, invariant continuity, deadlock regression, complexity claims |
19
+ | Data models | DataModelV1 verification — DS/DA/DD checks, migration coverage, copy consistency, before/after breaking-change regression |
20
+ | Concurrency risk mining | Scans documents/plans for concurrency safety claims (thread-safe, lock-free, race condition, interrupt safety, etc.) and flags them for dedicated verification |
19
21
  | Output | Structured findings with exact file:line evidence, severity classification, correction direction — never inline fixes |
20
22
 
21
23
  The model is always shown as a transition table and **confirmed with the user before running** — extraction errors are the dominant failure mode.
@@ -64,6 +66,8 @@ Native dsh support ships as a cordis plugin bundle at the repository root (the r
64
66
 
65
67
  - The skill is discovered as-is by dsh's `skill-filesystem` provider (Agent Skills open standard) — zero code.
66
68
  - The bundle injects the claim-verification gate (1% Rule / Red Flags / proactive suggestion) into the first model step of every agent session — the dsh-native counterpart of the Claude `SessionStart` hook. It also registers a model-visible catalog entry (`cordis_inspect`), a native `logicprobe_verify` tool (`ctx.tools`), and a policy-aware `logicprobe:mode` context (`ctx.systemPrompt`).
69
+ - `logicprobe_datamodel_verify` adds data-model verification: DataModelV1, migration coverage, copy consistency, and DD1-DD4 before/after data regression.
70
+ - `logicprobe_concurrency_scan` mines documents/plans for concurrency risk claims (thread-safe, lock-free, race condition, mutex, etc.) and flags them for dedicated verification.
67
71
  - Together with the embedded-workbench bundle's Plan Verification Gate, this closes the claim-verification loop in dsh.
68
72
 
69
73
  Install: see [`.dsh/INSTALL.md`](.dsh/INSTALL.md) (four options, from plain skill copy to `dsh plugin add`).
@@ -77,10 +81,13 @@ The plugin auto-injects a capability notification into the first model step. The
77
81
  - **Design doc / plan review** — "Review this design document" → claim enumeration and codebase verification
78
82
  - **Behavioral questions** — "could this state machine deadlock", "is this retry limit safe", "check this timing for bugs" → the skill is proactively suggested (not auto-loaded) as an optional verification pass
79
83
  - **Refactoring plans** — the pipeline compares before/after models to flag undocumented behavioral changes
84
+ - **Data model / migration review** — "is this migration non-breaking", "does this copy cover all required fields" → use the `logicprobe-datamodel` skill
80
85
 
81
86
  The skill auto-classifies depth (LIGHTWEIGHT / STANDARD / ESCALATED) from plan features in Phase 0, and appends a `## Plan Verification` summary block as the audit trail.
82
87
 
83
- In DSH, prefer the native `logicprobe_verify` tool (see `skills/logicprobe/references/dsh-model-schema.md`). Python remains optional for non-DSH hosts: when available, the reusable harness at `references/verification-harness.py` runs the checks; when not (air-gapped machines), the guide at `references/logic-verification-guide.md` provides a manual verification mode.
88
+ In DSH, prefer the native `logicprobe_verify` tool for state machines and `logicprobe_datamodel_verify` for data models (see the schema references under each skill). Python remains optional for non-DSH hosts: state-machine checks use `skills/logicprobe/references/verification-harness.py`; data-model checks use `skills/logicprobe-datamodel/references/data-model-harness.py`. When Python is unavailable, the corresponding guide provides a manual verification mode.
89
+
90
+ Sample models are available under [`examples/`](examples/README.md): order state-machine before/after, e-commerce data model, and User field migration.
84
91
 
85
92
  ## Codex CLI
86
93
 
package/README.md CHANGED
@@ -11,9 +11,11 @@
11
11
  | 阶段 | 内容 |
12
12
  |------|------|
13
13
  | Phase 1-2 | 枚举每个可验证声称(API 名、文件路径、枚举值、数量、机制可行性)→ 逐条对照代码库给出证据 |
14
- | Phase 2a | 对提取的状态机模型执行 **7 项结构检查**:可达性、死锁、活性、确定性、事件/守卫完备性、不变量有效性 |
15
- | Phase 2b | **7 种对抗探针**:意外事件、竞态交错、顺序置换、配对对称(lock/unlock)、边界轰炸、资源注入、最小反例 |
14
+ | Phase 2a | 对提取的状态机模型执行 **8 项结构检查**:可达性、死锁、活性、确定性、事件/守卫完备性、不变量有效性、单调变量 |
15
+ | Phase 2b | **11 种对抗探针**:意外事件、竞态交错、顺序置换、配对对称(lock/unlock)、边界轰炸、资源注入、最小反例、幂等重放、必达、顺序、原子性 |
16
16
  | 重构模式 | 前后模型对比——行为保持、不变量连续性、死锁回归、复杂度声称 |
17
+ | 数据模型模式 | DataModelV1 数据模型验证——DS/DA/DD 检查,迁移覆盖、copy 一致性、before/after 破坏性变更回归 |
18
+ | 并发风险挖掘 | 扫描文档/计划中的并发安全声称(thread-safe、lock-free、race condition、中断安全等),标记需要专用验证 |
17
19
  | 输出 | 结构化发现:精确 file:line 证据、严重性分级、修正方向——绝不在核查中直接改代码 |
18
20
 
19
21
  模型永远先以转换表形式展示并**经用户确认后才运行**——模型提取错误是验证的头号失败模式。
@@ -63,6 +65,8 @@ git clone https://github.com/AmethystLuna/logicprobe.git ~/.claude/plugins/dev/l
63
65
  - 技能遵循 Agent Skills 开放标准,被 dsh 的 `skill-filesystem` provider 原样发现——零代码。
64
66
  - bundle 将 claim 验证门禁(1% Rule / Red Flags / 主动建议)注入每个 agent 会话的第一个模型步骤——是 Claude `SessionStart` hook 在 dsh 的原生对应物,并注册模型可见目录条目(`cordis_inspect`)、原生工具 `logicprobe_verify`(`ctx.tools`)以及策略感知上下文 `logicprobe:mode`(`ctx.systemPrompt`)。
65
67
  - `logicprobe_verify` 支持 `beforeModel` + `stateMapping` 的 BEFORE/AFTER 对比(D1-D4),可直接验证重构/迁移的行为保持、不变量连续性、回归增量和死锁/活性回归。
68
+ - `logicprobe_datamodel_verify` 新增数据模型验证:DataModelV1、迁移覆盖、copy 一致性、DD1-DD4 before/after 数据回归。
69
+ - `logicprobe_concurrency_scan` 扫描文档/计划中的并发风险声称(thread-safe、lock-free、race condition、mutex 等),标记需要专用并发验证。
66
70
  - 与 embedded-workbench bundle 的 Plan Verification Gate 配合,在 dsh 中闭环了 claim 验证链路。
67
71
 
68
72
  安装:参见 [`.dsh/INSTALL.md`](.dsh/INSTALL.md)(四种方式,从纯技能拷贝到 `dsh plugin add`)。
@@ -76,10 +80,13 @@ git clone https://github.com/AmethystLuna/logicprobe.git ~/.claude/plugins/dev/l
76
80
  - **设计文档 / 计划审查** — "Review this design document" → 声称枚举与代码库核查
77
81
  - **行为类问题** — "could this state machine deadlock"、"is this retry limit safe"、"check this timing for bugs" → 主动建议(不自动加载)作为可选验证
78
82
  - **重构计划** — 管线对比前后模型,标记计划未声明的行为变化
83
+ - **数据模型/迁移审查** — "is this migration non-breaking"、"does this copy cover all required fields" → 使用 `logicprobe-datamodel` 技能
79
84
 
80
85
  技能在 Phase 0 依据计划特征自动分级(LIGHTWEIGHT / STANDARD / ESCALATED),并在计划文件追加 `## Plan Verification` 摘要块作为审计痕迹。
81
86
 
82
- Python 可选:可用时使用 `references/verification-harness.py` 自动执行检查;不可用(如离线开发机)时,`references/logic-verification-guide.md` 提供手动验证模式。
87
+ Python 可选:状态机验证使用 `skills/logicprobe/references/verification-harness.py`,数据模型验证使用 `skills/logicprobe-datamodel/references/data-model-harness.py`;不可用(如离线开发机)时,对应 guide 提供手动验证模式。
88
+
89
+ 示例模型见 [`examples/`](examples/README.md):订单状态机 before/after、电商数据模型、User 字段迁移。
83
90
 
84
91
  ## Codex CLI
85
92
 
@@ -0,0 +1,34 @@
1
+ import { defineTool } from '@deepseek-ai/dsh-tools';
2
+ import { runConcurrencyScan } from './concurrency.js';
3
+ export const LOGICPROBE_CONCURRENCY_SCAN_TOOL_NAME = 'logicprobe_concurrency_scan';
4
+ /**
5
+ * Model-visible DSH tool that mines design documents/plans for concurrency-related
6
+ * claims and risk keywords. It does not prove concurrency safety; it flags terms
7
+ * such as "thread-safe", "lock-free", "race condition", "atomic", "mutex", etc.,
8
+ * so the model can either provide dedicated evidence or mark the claim unverified.
9
+ */
10
+ export const logicProbeConcurrencyScanTool = defineTool({
11
+ name: LOGICPROBE_CONCURRENCY_SCAN_TOOL_NAME,
12
+ description: 'Scan a document or plan text for concurrency risk points. Use ONLY after confirming the target actually has concurrency requirements or behavior (threads, async tasks, interrupts, shared state, parallel execution). If the target is purely sequential, do not call this tool. Returns findings for keywords like thread-safe, lock-free, data race, race condition, atomic, synchronized, mutex, semaphore, shared variable, reentrant, interrupt-safe. Absolute claims (thread-safe, lock-free, no data race) are flagged as errors requiring dedicated verification.',
13
+ parameters: {
14
+ text: {
15
+ type: 'string',
16
+ required: true,
17
+ description: 'Document or plan text to scan for concurrency-related claims.',
18
+ },
19
+ },
20
+ output: {
21
+ schema: {
22
+ type: 'json',
23
+ description: 'Concurrency scan report with findings and summary.',
24
+ },
25
+ render(_args, value) {
26
+ return [{ type: 'text', text: JSON.stringify(value, null, 2) }];
27
+ },
28
+ },
29
+ timeoutMs: 10_000,
30
+ isConcurrencySafe: () => true,
31
+ async execute(args) {
32
+ return runConcurrencyScan(args.text);
33
+ },
34
+ });
@@ -0,0 +1,76 @@
1
+ const KEYWORD_RULES = [
2
+ { pattern: /\bthread\s*-?\s*safe\b/i, label: 'thread-safe', absolute: true },
3
+ { pattern: /\block\s*-?\s*free\b/i, label: 'lock-free', absolute: true },
4
+ { pattern: /\bwait\s*-?\s*free\b/i, label: 'wait-free', absolute: true },
5
+ { pattern: /\bno\s+data\s+race\b/i, label: 'no data race', absolute: true },
6
+ { pattern: /\brace\s*-?\s*free\b/i, label: 'race-free', absolute: true },
7
+ { pattern: /\bdata\s+race\b/i, label: 'data race', absolute: false },
8
+ { pattern: /\brace\s+condition\b/i, label: 'race condition', absolute: false },
9
+ { pattern: /\bthread\s*-?\s*safety\b/i, label: 'thread safety', absolute: false },
10
+ { pattern: /\bconcurrent\b/i, label: 'concurrent', absolute: false },
11
+ { pattern: /\bparallel\b/i, label: 'parallel', absolute: false },
12
+ { pattern: /\bmulti-?threaded\b/i, label: 'multi-threaded', absolute: false },
13
+ { pattern: /\bmultithreaded\b/i, label: 'multithreaded', absolute: false },
14
+ { pattern: /\batomic\b/i, label: 'atomic', absolute: false },
15
+ { pattern: /\bsynchronized\b/i, label: 'synchronized', absolute: false },
16
+ { pattern: /\bmutex\b/i, label: 'mutex', absolute: false },
17
+ { pattern: /\bsemaphore\b/i, label: 'semaphore', absolute: false },
18
+ { pattern: /\bspinlock\b/i, label: 'spinlock', absolute: false },
19
+ { pattern: /\bshared\s+variable\b/i, label: 'shared variable', absolute: false },
20
+ { pattern: /\bshared\s+memory\b/i, label: 'shared memory', absolute: false },
21
+ { pattern: /\bglobal\s+state\b/i, label: 'global state', absolute: false },
22
+ { pattern: /\breentrant\b/i, label: 'reentrant', absolute: false },
23
+ { pattern: /\binterrupt\s*-?\s*safe\b/i, label: 'interrupt-safe', absolute: true },
24
+ { pattern: /\bISR\s*-?\s*safe\b/i, label: 'ISR-safe', absolute: true },
25
+ { pattern: /\binterrupt\s+safety\b/i, label: 'interrupt safety', absolute: false },
26
+ { pattern: /\binterrupt\s+context\b/i, label: 'interrupt context', absolute: false },
27
+ { pattern: /\bISR\b/i, label: 'ISR', absolute: false },
28
+ { pattern: /\bIRQ\b/i, label: 'IRQ', absolute: false },
29
+ { pattern: /\bNMI\b/i, label: 'NMI', absolute: false },
30
+ { pattern: /\bcritical\s+section\b/i, label: 'critical section', absolute: false },
31
+ { pattern: /\bdisable_irq\b/i, label: 'disable_irq', absolute: false },
32
+ { pattern: /\benable_irq\b/i, label: 'enable_irq', absolute: false },
33
+ { pattern: /\bspin_lock_irqsave\b/i, label: 'spin_lock_irqsave', absolute: false },
34
+ ];
35
+ export function runConcurrencyScan(text) {
36
+ const lines = text.split(/\r?\n/);
37
+ const findings = [];
38
+ const seen = new Set();
39
+ lines.forEach((line, index) => {
40
+ const lineNumber = index + 1;
41
+ const lower = line.toLowerCase();
42
+ for (const rule of KEYWORD_RULES) {
43
+ if (!rule.pattern.test(line))
44
+ continue;
45
+ const key = rule.label + ':' + lineNumber;
46
+ if (seen.has(key))
47
+ continue;
48
+ seen.add(key);
49
+ const finding = {
50
+ code: rule.absolute ? 'CONCURRENCY_ABSOLUTE_CLAIM' : 'CONCURRENCY_KEYWORD',
51
+ severity: rule.absolute ? 'error' : 'warning',
52
+ message: rule.absolute
53
+ ? 'Concurrency safety claim "' + rule.label + '" detected; this requires dedicated verification (TSan, model checker, or explicit proof).'
54
+ : 'Concurrency-related term "' + rule.label + '" detected; review whether the plan addresses this risk.',
55
+ line: lineNumber,
56
+ snippet: line.trim().slice(0, 200),
57
+ keyword: rule.label,
58
+ };
59
+ findings.push(finding);
60
+ }
61
+ });
62
+ const errors = findings.filter((finding) => finding.severity === 'error').length;
63
+ const warnings = findings.filter((finding) => finding.severity === 'warning').length;
64
+ const absoluteClaims = findings.filter((finding) => finding.code === 'CONCURRENCY_ABSOLUTE_CLAIM').length;
65
+ return {
66
+ ok: true,
67
+ findings,
68
+ summary: {
69
+ lines: lines.length,
70
+ keywords: findings.length,
71
+ absoluteClaims,
72
+ warnings,
73
+ errors,
74
+ },
75
+ };
76
+ }