dsh-logicprobe 0.3.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.en-US.md +237 -0
- package/README.md +234 -0
- package/cordis.patch.yml +12 -0
- package/lib/engine.js +1103 -0
- package/lib/index.js +277 -0
- package/lib/tool.js +45 -0
- package/lib/types/engine.d.ts +133 -0
- package/lib/types/index.d.ts +46 -0
- package/lib/types/tool.d.ts +8 -0
- package/package.json +81 -0
- package/skills/logicprobe/SKILL.md +266 -0
- package/skills/logicprobe/references/dsh-model-schema.md +104 -0
- package/skills/logicprobe/references/logic-verification-guide.md +413 -0
- package/skills/logicprobe/references/verification-harness.py +582 -0
- package/src/engine.ts +1164 -0
- package/src/index.ts +301 -0
- package/src/tool.ts +48 -0
package/lib/index.js
ADDED
|
@@ -0,0 +1,277 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* logicprobe — DeepSeek Harness native plugin for the Logic Probe toolbox.
|
|
3
|
+
* Injects the session-start gate text (claim-verification doctrine, 1% Rule,
|
|
4
|
+
* Red Flags, proactive suggestion) into the first model step of every agent
|
|
5
|
+
* session, mirroring the SessionStart hook the Claude Code plugin installs.
|
|
6
|
+
* The skill ships in this package's `skills/` directory and is registered at
|
|
7
|
+
* apply time into dsh's `ctx.skills` registry through the standard filesystem
|
|
8
|
+
* provider, so it appears in every session catalog without a manual copy step.
|
|
9
|
+
*
|
|
10
|
+
* Injection listens on agent/pre-step and appends the gate to the FIRST
|
|
11
|
+
* model step that runs, once per session (guarded by the session's durable
|
|
12
|
+
* history). Session-start inbox injection was dropped: a blank-session preset
|
|
13
|
+
* switch (agentPreset.select -> recompose) can clear the inbox before the
|
|
14
|
+
* first step, losing the gate for the whole session. The pre-step decision is
|
|
15
|
+
* the durable path - anchored/bootstrap presets that strip first-step injected
|
|
16
|
+
* reminders (skill catalog, AGENTS.md, gate plugins) simply defer this message
|
|
17
|
+
* to the first step after their promotion, and the history guard re-injects it
|
|
18
|
+
* there. The default gate text is the dsh-native adaptation of
|
|
19
|
+
* `hooks/session-start-content.md`: behavior rules
|
|
20
|
+
* (1% Rule / Red Flags / proactive suggestion) stay in sync, while
|
|
21
|
+
* presentation is adapted to dsh's native skill catalog — the trigger list
|
|
22
|
+
* lives in the skill description, not duplicated in the gate. Deployments
|
|
23
|
+
* override via Config.
|
|
24
|
+
*
|
|
25
|
+
* @module logicprobe-dsh
|
|
26
|
+
*/
|
|
27
|
+
import { fileURLToPath } from 'node:url';
|
|
28
|
+
import z from '@deepseek-ai/schemastery';
|
|
29
|
+
import { createUserMessage } from '@deepseek-ai/dsh-llm';
|
|
30
|
+
import { FileSystemSkillProvider } from '@deepseek-ai/dsh-skill-filesystem';
|
|
31
|
+
import { logicProbeVerifyTool } from './tool.js';
|
|
32
|
+
import { ENGINE_SCHEMA_VERSION } from './engine.js';
|
|
33
|
+
export const name = 'logicprobe';
|
|
34
|
+
// Skills are contributed through the registry service, which dsh-base always
|
|
35
|
+
// mounts before bundle rows such as this one apply.
|
|
36
|
+
export const inject = ['skills'];
|
|
37
|
+
// Absolute path of the package's shipped skills directory. `lib/index.js`
|
|
38
|
+
// lives one level below the package root, so `../skills` from the module URL
|
|
39
|
+
// lands on `<package>/skills` regardless of where the package was installed.
|
|
40
|
+
const SKILLS_DIR = fileURLToPath(new URL('../skills', import.meta.url));
|
|
41
|
+
const GATE_PLUGIN_ID = 'logicprobe';
|
|
42
|
+
const DEFAULT_GATE_CONTENT = `<EXTREMELY_IMPORTANT>
|
|
43
|
+
Plugin logicprobe is active. Documents are not truth — code is. Verify every verifiable claim before accepting or acting on any design.
|
|
44
|
+
|
|
45
|
+
**1% Rule**: If there is even a 1% chance the logicprobe skill applies — reviewing design documents, architecture specs, technical proposals, or refactoring plans that make claims about API names, file locations, enum values, mechanism feasibility, state machines, protocol logic, or behavioral guarantees ("always"/"never"/"guaranteed") — load it with the skill tool before responding. The cost of loading is trivial compared to the cost of a false claim.
|
|
46
|
+
|
|
47
|
+
**Red Flags** — if you think any of these, STOP. You are rationalizing:
|
|
48
|
+
|
|
49
|
+
| You think | Reality |
|
|
50
|
+
|-----------|---------|
|
|
51
|
+
| "This plan is too simple to verify" | The skill auto-classifies depth (LIGHTWEIGHT / STANDARD / ESCALATED). You don't decide. |
|
|
52
|
+
| "I already know the file paths are correct" | Organic verification leaves no audit trail. Run Phase 0, append the "## Plan Verification" block. |
|
|
53
|
+
| "I'll verify while implementing" | Verification happens before implementation, not during. |
|
|
54
|
+
| "I can check this with reasoning alone" | Behavioral claims are verified with code/models, not intuition. One counter-example refutes a universal claim. |
|
|
55
|
+
|
|
56
|
+
**Native verification path**: In dsh, prefer the \`logicprobe_verify\` tool for executable state-machine checks. The skill's Python harness remains the fallback for non-dsh hosts.
|
|
57
|
+
|
|
58
|
+
**Proactive suggestion**: When a user asks code-level behavioral questions — "could this state machine deadlock", "is this retry limit safe", "check this timing sequence for bugs" — suggest logicprobe as an optional verification pass (do not auto-escalate).
|
|
59
|
+
</EXTREMELY_IMPORTANT>`;
|
|
60
|
+
export const Config = z.object({
|
|
61
|
+
enabled: z.boolean().default(true),
|
|
62
|
+
gateContent: z.string().default(DEFAULT_GATE_CONTENT),
|
|
63
|
+
interaction: z.union(['ask', 'auto', 'follow-approval']).default('follow-approval'),
|
|
64
|
+
});
|
|
65
|
+
function gateMessage(text) {
|
|
66
|
+
return createUserMessage({
|
|
67
|
+
content: [{ type: 'text', text }],
|
|
68
|
+
// `form` omitted — an undeclared context is the documented default.
|
|
69
|
+
source: { kind: 'plugin', plugin: GATE_PLUGIN_ID },
|
|
70
|
+
});
|
|
71
|
+
}
|
|
72
|
+
function lastApprovalPolicy(session) {
|
|
73
|
+
const events = session.events;
|
|
74
|
+
for (let index = events.length - 1; index >= 0; index -= 1) {
|
|
75
|
+
const event = events[index];
|
|
76
|
+
if (event.type === 'approval/policy') {
|
|
77
|
+
return event.data?.policy === 'never' ? 'never' : 'ask';
|
|
78
|
+
}
|
|
79
|
+
}
|
|
80
|
+
return undefined;
|
|
81
|
+
}
|
|
82
|
+
function planModeActive(session) {
|
|
83
|
+
const events = session.events;
|
|
84
|
+
for (let index = events.length - 1; index >= 0; index -= 1) {
|
|
85
|
+
const event = events[index];
|
|
86
|
+
if (event.type === 'plan/mode')
|
|
87
|
+
return event.data?.active === true;
|
|
88
|
+
}
|
|
89
|
+
return false;
|
|
90
|
+
}
|
|
91
|
+
function resolveInteraction(config, session) {
|
|
92
|
+
if (config.interaction === 'ask' || config.interaction === 'auto')
|
|
93
|
+
return config.interaction;
|
|
94
|
+
return lastApprovalPolicy(session) === 'never' ? 'auto' : 'ask';
|
|
95
|
+
}
|
|
96
|
+
function modeContextText(config, session) {
|
|
97
|
+
const interaction = resolveInteraction(config, session);
|
|
98
|
+
const lines = [
|
|
99
|
+
'logicprobe: use the `logicprobe_verify` tool for executable state-machine verification.',
|
|
100
|
+
interaction === 'auto'
|
|
101
|
+
? 'logicprobe interaction=auto: do NOT call ask_user_question for model confirmation; run round-trip validation of the extracted transition table and mark the result UNCONFIRMED.'
|
|
102
|
+
: 'logicprobe interaction=ask: show the extracted transition table and get user confirmation before running verification.',
|
|
103
|
+
];
|
|
104
|
+
if (planModeActive(session)) {
|
|
105
|
+
lines.push('Plan mode active: before exit_plan_mode, run logicprobe Phase 0 and append the "## Plan Verification" block to the plan file.');
|
|
106
|
+
}
|
|
107
|
+
return lines.join(' ');
|
|
108
|
+
}
|
|
109
|
+
/**
|
|
110
|
+
* Model-visible catalog entry (cordis_inspect_list / cordis_inspect_query):
|
|
111
|
+
* lets the model read this plugin's runtime status without guessing. Mirrors
|
|
112
|
+
* the registration pattern of the official dsh-tool-cordis host providers.
|
|
113
|
+
*/
|
|
114
|
+
function inspectProvider(config, isToolRegistered) {
|
|
115
|
+
return {
|
|
116
|
+
manifest: {
|
|
117
|
+
id: 'logicprobe',
|
|
118
|
+
description: 'Session-start gate injection and native verification tooling for the Logic Probe toolbox — folds the claim-verification doctrine (1% Rule / Red Flags / proactive suggestion) into the first model step of every agent session and registers the logicprobe_verify tool.',
|
|
119
|
+
methods: [
|
|
120
|
+
{
|
|
121
|
+
name: 'status',
|
|
122
|
+
description: 'Read gate injection status, interaction mode, tool registration state, and engine schema version.',
|
|
123
|
+
inputSchema: {
|
|
124
|
+
type: 'object',
|
|
125
|
+
properties: {},
|
|
126
|
+
additionalProperties: false,
|
|
127
|
+
},
|
|
128
|
+
outputSchema: {
|
|
129
|
+
type: 'object',
|
|
130
|
+
description: 'Gate-injection plugin status.',
|
|
131
|
+
properties: {
|
|
132
|
+
enabled: { type: 'boolean', description: 'Whether the gate folds into the first model step.' },
|
|
133
|
+
gateContentLength: { type: 'integer', description: 'Length in characters of the injected gate text.' },
|
|
134
|
+
interaction: { type: 'string', enum: ['ask', 'auto', 'follow-approval'], description: 'Configured interaction mode. follow-approval resolves per session from approval/policy.' },
|
|
135
|
+
toolRegistered: { type: 'boolean', description: 'Whether the logicprobe_verify tool is registered on ctx.tools.' },
|
|
136
|
+
engineSchemaVersion: { type: 'integer', description: 'Model schema version the bundled verification engine accepts.' },
|
|
137
|
+
},
|
|
138
|
+
required: ['enabled', 'gateContentLength', 'interaction', 'toolRegistered', 'engineSchemaVersion'],
|
|
139
|
+
additionalProperties: false,
|
|
140
|
+
},
|
|
141
|
+
},
|
|
142
|
+
],
|
|
143
|
+
},
|
|
144
|
+
query: async (method) => {
|
|
145
|
+
if (method === 'status') {
|
|
146
|
+
return {
|
|
147
|
+
enabled: config.enabled,
|
|
148
|
+
gateContentLength: config.gateContent.length,
|
|
149
|
+
interaction: config.interaction,
|
|
150
|
+
toolRegistered: isToolRegistered(),
|
|
151
|
+
engineSchemaVersion: ENGINE_SCHEMA_VERSION,
|
|
152
|
+
};
|
|
153
|
+
}
|
|
154
|
+
return null;
|
|
155
|
+
},
|
|
156
|
+
};
|
|
157
|
+
}
|
|
158
|
+
export function apply(ctx, config) {
|
|
159
|
+
// Optional services are registered opportunistically; base-bundle rows can
|
|
160
|
+
// mount after this row applies, so registration is retried on the first
|
|
161
|
+
// agent/pre-step — by then the app is fully booted.
|
|
162
|
+
let providerRegistered = false;
|
|
163
|
+
let toolRegistered = false;
|
|
164
|
+
let modeContextRegistered = false;
|
|
165
|
+
const registerProvider = () => {
|
|
166
|
+
if (providerRegistered)
|
|
167
|
+
return;
|
|
168
|
+
const inspect = ctx.get('cordisInspect');
|
|
169
|
+
if (inspect === undefined)
|
|
170
|
+
return;
|
|
171
|
+
try {
|
|
172
|
+
ctx.effect(() => inspect.register(inspectProvider(config, () => toolRegistered)), 'logicprobe: inspect provider');
|
|
173
|
+
providerRegistered = true;
|
|
174
|
+
}
|
|
175
|
+
catch (err) {
|
|
176
|
+
console.warn('[logicprobe] inspect provider registration failed', err);
|
|
177
|
+
}
|
|
178
|
+
};
|
|
179
|
+
const registerTool = () => {
|
|
180
|
+
if (toolRegistered)
|
|
181
|
+
return;
|
|
182
|
+
const tools = ctx.get('tools');
|
|
183
|
+
if (tools === undefined)
|
|
184
|
+
return;
|
|
185
|
+
try {
|
|
186
|
+
ctx.effect(() => tools.register(logicProbeVerifyTool), 'logicprobe: verify tool');
|
|
187
|
+
toolRegistered = true;
|
|
188
|
+
}
|
|
189
|
+
catch (err) {
|
|
190
|
+
console.warn('[logicprobe] logicprobe_verify tool registration failed', err);
|
|
191
|
+
}
|
|
192
|
+
};
|
|
193
|
+
const registerModeContext = () => {
|
|
194
|
+
if (modeContextRegistered)
|
|
195
|
+
return;
|
|
196
|
+
const systemPrompt = ctx.get('systemPrompt');
|
|
197
|
+
if (systemPrompt === undefined)
|
|
198
|
+
return;
|
|
199
|
+
try {
|
|
200
|
+
ctx.effect(() => systemPrompt.context({
|
|
201
|
+
name: 'logicprobe:mode',
|
|
202
|
+
order: 118,
|
|
203
|
+
text: (context) => {
|
|
204
|
+
const agent = context.agent;
|
|
205
|
+
if (agent === undefined)
|
|
206
|
+
return '';
|
|
207
|
+
return modeContextText(config, agent.session);
|
|
208
|
+
},
|
|
209
|
+
}), 'logicprobe: system prompt context');
|
|
210
|
+
modeContextRegistered = true;
|
|
211
|
+
}
|
|
212
|
+
catch (err) {
|
|
213
|
+
console.warn('[logicprobe] system prompt context registration failed', err);
|
|
214
|
+
}
|
|
215
|
+
};
|
|
216
|
+
const registerIntegrations = () => {
|
|
217
|
+
registerProvider();
|
|
218
|
+
registerTool();
|
|
219
|
+
registerModeContext();
|
|
220
|
+
};
|
|
221
|
+
registerIntegrations();
|
|
222
|
+
// Ship the bundled skill through the registry: reuse the standard
|
|
223
|
+
// filesystem provider over this package's own `skills/` directory, so
|
|
224
|
+
// catalog discovery, frontmatter parsing, and SKILL.md loading behave
|
|
225
|
+
// exactly like project/user skills while the plugin stays self-contained.
|
|
226
|
+
// Registration lands in the global registry layer (this row mounts at the
|
|
227
|
+
// profile root), so every agent preset sees the skill. `registerProvider`
|
|
228
|
+
// returns the effect disposer; its teardown unregisters and invalidates.
|
|
229
|
+
ctx.skills.registerProvider((control) => {
|
|
230
|
+
return new FileSystemSkillProvider(ctx, control, {
|
|
231
|
+
providerName: 'logicprobe',
|
|
232
|
+
includeDefaultRoots: false,
|
|
233
|
+
customSkillDirs: [SKILLS_DIR],
|
|
234
|
+
});
|
|
235
|
+
});
|
|
236
|
+
if (!config.enabled)
|
|
237
|
+
return;
|
|
238
|
+
// Inject the gate once per session on the FIRST model step that runs,
|
|
239
|
+
// instead of at session-start: session-start injection lands in the agent's
|
|
240
|
+
// inbox, which a blank-session preset switch (agentPreset.select ->
|
|
241
|
+
// recompose) can clear before the first step - the gate would then be lost
|
|
242
|
+
// for the whole session. The pre-step decision is the durable path a
|
|
243
|
+
// first-step injection takes: the gate is appended to the first step's
|
|
244
|
+
// decision and enters session history there, so every later step (and a
|
|
245
|
+
// resume) skips it. Anchored/bootstrap presets that strip first-step
|
|
246
|
+
// injected reminders (skill catalog, AGENTS.md, gate plugins) simply defer
|
|
247
|
+
// this message to the first step after their promotion - the history guard
|
|
248
|
+
// re-injects it there, so the gate still lands exactly once per session.
|
|
249
|
+
ctx.on('agent/pre-step', async ({ agent }, next) => {
|
|
250
|
+
const decision = await next();
|
|
251
|
+
if (decision.kind === 'reject')
|
|
252
|
+
return decision;
|
|
253
|
+
registerIntegrations();
|
|
254
|
+
if (gateInHistory(agent.session))
|
|
255
|
+
return decision;
|
|
256
|
+
return {
|
|
257
|
+
kind: 'enter',
|
|
258
|
+
messages: [...decision.messages, gateMessage(config.gateContent)],
|
|
259
|
+
};
|
|
260
|
+
});
|
|
261
|
+
}
|
|
262
|
+
/**
|
|
263
|
+
* Whether the gate already entered this session's durable history. The
|
|
264
|
+
* pre-step listener re-appends the gate until it does; once a step committed
|
|
265
|
+
* it, every later step (and a resume of a session that kept it) skips the
|
|
266
|
+
* injection. A session whose gate was dropped before any step ran (e.g. an
|
|
267
|
+
* inbox cleared by a blank-session preset switch) simply re-injects on the
|
|
268
|
+
* first step that runs.
|
|
269
|
+
*/
|
|
270
|
+
function gateInHistory(session) {
|
|
271
|
+
return session.events.some((event) => {
|
|
272
|
+
if (event.type !== 'user/message')
|
|
273
|
+
return false;
|
|
274
|
+
const source = event.data.source;
|
|
275
|
+
return source.kind === 'plugin' && source.plugin === GATE_PLUGIN_ID;
|
|
276
|
+
});
|
|
277
|
+
}
|
package/lib/tool.js
ADDED
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
import { defineTool } from '@deepseek-ai/dsh-tools';
|
|
2
|
+
import { runVerification } from './engine.js';
|
|
3
|
+
export const LOGICPROBE_VERIFY_TOOL_NAME = 'logicprobe_verify';
|
|
4
|
+
/**
|
|
5
|
+
* Model-visible DSH tool wrapping the bundled TypeScript verification engine.
|
|
6
|
+
* The model passes a LogicModelV1 object; the engine validates it and returns
|
|
7
|
+
* the 14-check report. This is the dsh-native replacement for hand-filling the
|
|
8
|
+
* Python template shipped in the skill references.
|
|
9
|
+
*/
|
|
10
|
+
export const logicProbeVerifyTool = defineTool({
|
|
11
|
+
name: LOGICPROBE_VERIFY_TOOL_NAME,
|
|
12
|
+
description: 'Run executable state-machine verification (logicprobe). Takes a LogicModelV1 object with schemaVersion=1, init, states ({id, terminal?}), transitions ({from, event, to, guard?, updates?}), variables?, invariants?, concurrentPairs?, boundaryChecks?, resourcePairs?. Guards are structured ({variable, op, value} | {all} | {any} | {not}); invariants support never-states, var-in-range, and event-before-state. Returns a report with S1-S7 structural checks and A1-A7 adversarial probes including shortest counterexample paths. See skills/logicprobe/references/dsh-model-schema.md.',
|
|
13
|
+
parameters: {
|
|
14
|
+
model: {
|
|
15
|
+
type: 'json',
|
|
16
|
+
required: true,
|
|
17
|
+
description: 'LogicModelV1 state machine model to verify.',
|
|
18
|
+
},
|
|
19
|
+
maxStates: {
|
|
20
|
+
type: 'integer',
|
|
21
|
+
description: 'Maximum runtime states to explore. Default 10000.',
|
|
22
|
+
},
|
|
23
|
+
maxPermutationEvents: {
|
|
24
|
+
type: 'integer',
|
|
25
|
+
description: 'Maximum event count for A3 order permutation. Default 5.',
|
|
26
|
+
},
|
|
27
|
+
},
|
|
28
|
+
output: {
|
|
29
|
+
schema: {
|
|
30
|
+
type: 'json',
|
|
31
|
+
description: 'logicprobe verification report with summary and per-check findings.',
|
|
32
|
+
},
|
|
33
|
+
render(_args, value) {
|
|
34
|
+
return [{ type: 'text', text: JSON.stringify(value, null, 2) }];
|
|
35
|
+
},
|
|
36
|
+
},
|
|
37
|
+
timeoutMs: 10_000,
|
|
38
|
+
isConcurrencySafe: () => true,
|
|
39
|
+
async execute(args) {
|
|
40
|
+
return runVerification(args.model, {
|
|
41
|
+
maxStates: args.maxStates,
|
|
42
|
+
maxPermutationEvents: args.maxPermutationEvents,
|
|
43
|
+
});
|
|
44
|
+
},
|
|
45
|
+
});
|
|
@@ -0,0 +1,133 @@
|
|
|
1
|
+
export declare const ENGINE_SCHEMA_VERSION = 1;
|
|
2
|
+
export type VarValue = number | boolean;
|
|
3
|
+
export type GuardOp = '==' | '!=' | '<' | '<=' | '>' | '>=';
|
|
4
|
+
export interface LeafGuard {
|
|
5
|
+
variable: string;
|
|
6
|
+
op: GuardOp;
|
|
7
|
+
value: VarValue;
|
|
8
|
+
}
|
|
9
|
+
export interface AllGuard {
|
|
10
|
+
all: GuardNode[];
|
|
11
|
+
}
|
|
12
|
+
export interface AnyGuard {
|
|
13
|
+
any: GuardNode[];
|
|
14
|
+
}
|
|
15
|
+
export interface NotGuard {
|
|
16
|
+
not: GuardNode;
|
|
17
|
+
}
|
|
18
|
+
export type GuardNode = LeafGuard | AllGuard | AnyGuard | NotGuard;
|
|
19
|
+
export interface StateSpec {
|
|
20
|
+
id: string;
|
|
21
|
+
terminal?: boolean;
|
|
22
|
+
}
|
|
23
|
+
export interface UpdateSpec {
|
|
24
|
+
variable: string;
|
|
25
|
+
op: 'set' | 'inc' | 'dec';
|
|
26
|
+
value?: number;
|
|
27
|
+
}
|
|
28
|
+
export interface TransitionSpec {
|
|
29
|
+
from: string;
|
|
30
|
+
event: string;
|
|
31
|
+
to: string;
|
|
32
|
+
/** Absent guard is the else/default branch for the same (from, event) group. */
|
|
33
|
+
guard?: GuardNode;
|
|
34
|
+
updates?: UpdateSpec[];
|
|
35
|
+
}
|
|
36
|
+
export interface VariableSpec {
|
|
37
|
+
name: string;
|
|
38
|
+
kind: 'integer' | 'boolean';
|
|
39
|
+
init: number | boolean;
|
|
40
|
+
min?: number;
|
|
41
|
+
max?: number;
|
|
42
|
+
}
|
|
43
|
+
export type InvariantSpec = {
|
|
44
|
+
id: string;
|
|
45
|
+
description: string;
|
|
46
|
+
kind: 'never-states';
|
|
47
|
+
states: string[];
|
|
48
|
+
} | {
|
|
49
|
+
id: string;
|
|
50
|
+
description: string;
|
|
51
|
+
kind: 'var-in-range';
|
|
52
|
+
variable: string;
|
|
53
|
+
min?: number;
|
|
54
|
+
max?: number;
|
|
55
|
+
} | {
|
|
56
|
+
id: string;
|
|
57
|
+
description: string;
|
|
58
|
+
kind: 'event-before-state';
|
|
59
|
+
event: string;
|
|
60
|
+
state: string;
|
|
61
|
+
};
|
|
62
|
+
export interface ResourcePairSpec {
|
|
63
|
+
resource: string;
|
|
64
|
+
acquireEvent: string;
|
|
65
|
+
releaseEvent: string;
|
|
66
|
+
failEvent?: string;
|
|
67
|
+
}
|
|
68
|
+
export interface BoundaryCheckSpec {
|
|
69
|
+
variable: string;
|
|
70
|
+
values: number[];
|
|
71
|
+
}
|
|
72
|
+
export interface LogicModelV1 {
|
|
73
|
+
schemaVersion: 1;
|
|
74
|
+
init: string;
|
|
75
|
+
states: StateSpec[];
|
|
76
|
+
transitions: TransitionSpec[];
|
|
77
|
+
variables?: VariableSpec[];
|
|
78
|
+
invariants?: InvariantSpec[];
|
|
79
|
+
concurrentPairs?: [string, string][];
|
|
80
|
+
boundaryChecks?: BoundaryCheckSpec[];
|
|
81
|
+
resourcePairs?: ResourcePairSpec[];
|
|
82
|
+
}
|
|
83
|
+
export interface VerificationOptions {
|
|
84
|
+
maxStates?: number;
|
|
85
|
+
maxPermutationEvents?: number;
|
|
86
|
+
}
|
|
87
|
+
export interface PathStep {
|
|
88
|
+
from: string;
|
|
89
|
+
event: string;
|
|
90
|
+
to: string;
|
|
91
|
+
}
|
|
92
|
+
export interface Finding {
|
|
93
|
+
code: string;
|
|
94
|
+
severity: 'error' | 'warning';
|
|
95
|
+
message: string;
|
|
96
|
+
path?: PathStep[];
|
|
97
|
+
evidence?: Record<string, unknown>;
|
|
98
|
+
}
|
|
99
|
+
export interface CheckResult {
|
|
100
|
+
id: string;
|
|
101
|
+
name: string;
|
|
102
|
+
status: 'pass' | 'fail' | 'skip';
|
|
103
|
+
detail: string;
|
|
104
|
+
findings: Finding[];
|
|
105
|
+
}
|
|
106
|
+
export interface VerificationReport {
|
|
107
|
+
ok: boolean;
|
|
108
|
+
schemaVersion: 1;
|
|
109
|
+
modelHash: string;
|
|
110
|
+
summary: {
|
|
111
|
+
states: number;
|
|
112
|
+
transitions: number;
|
|
113
|
+
errors: number;
|
|
114
|
+
warnings: number;
|
|
115
|
+
checksRun: number;
|
|
116
|
+
truncated?: boolean;
|
|
117
|
+
};
|
|
118
|
+
checks: CheckResult[];
|
|
119
|
+
}
|
|
120
|
+
export interface RuntimeState {
|
|
121
|
+
state: string;
|
|
122
|
+
vars: Record<string, VarValue>;
|
|
123
|
+
}
|
|
124
|
+
export declare function modelHash(model: LogicModelV1): string;
|
|
125
|
+
export declare function validateModel(input: unknown): {
|
|
126
|
+
ok: true;
|
|
127
|
+
model: LogicModelV1;
|
|
128
|
+
} | {
|
|
129
|
+
ok: false;
|
|
130
|
+
errors: string[];
|
|
131
|
+
};
|
|
132
|
+
export declare function guardVariables(guard: GuardNode | undefined): string[];
|
|
133
|
+
export declare function runVerification(input: unknown, options?: VerificationOptions): VerificationReport;
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* logicprobe — DeepSeek Harness native plugin for the Logic Probe toolbox.
|
|
3
|
+
* Injects the session-start gate text (claim-verification doctrine, 1% Rule,
|
|
4
|
+
* Red Flags, proactive suggestion) into the first model step of every agent
|
|
5
|
+
* session, mirroring the SessionStart hook the Claude Code plugin installs.
|
|
6
|
+
* The skill ships in this package's `skills/` directory and is registered at
|
|
7
|
+
* apply time into dsh's `ctx.skills` registry through the standard filesystem
|
|
8
|
+
* provider, so it appears in every session catalog without a manual copy step.
|
|
9
|
+
*
|
|
10
|
+
* Injection listens on agent/pre-step and appends the gate to the FIRST
|
|
11
|
+
* model step that runs, once per session (guarded by the session's durable
|
|
12
|
+
* history). Session-start inbox injection was dropped: a blank-session preset
|
|
13
|
+
* switch (agentPreset.select -> recompose) can clear the inbox before the
|
|
14
|
+
* first step, losing the gate for the whole session. The pre-step decision is
|
|
15
|
+
* the durable path - anchored/bootstrap presets that strip first-step injected
|
|
16
|
+
* reminders (skill catalog, AGENTS.md, gate plugins) simply defer this message
|
|
17
|
+
* to the first step after their promotion, and the history guard re-injects it
|
|
18
|
+
* there. The default gate text is the dsh-native adaptation of
|
|
19
|
+
* `hooks/session-start-content.md`: behavior rules
|
|
20
|
+
* (1% Rule / Red Flags / proactive suggestion) stay in sync, while
|
|
21
|
+
* presentation is adapted to dsh's native skill catalog — the trigger list
|
|
22
|
+
* lives in the skill description, not duplicated in the gate. Deployments
|
|
23
|
+
* override via Config.
|
|
24
|
+
*
|
|
25
|
+
* @module logicprobe-dsh
|
|
26
|
+
*/
|
|
27
|
+
import type { Context } from '@deepseek-ai/cordis';
|
|
28
|
+
import z from '@deepseek-ai/schemastery';
|
|
29
|
+
export declare const name = "logicprobe";
|
|
30
|
+
export declare const inject: string[];
|
|
31
|
+
export type InteractionMode = 'ask' | 'auto' | 'follow-approval';
|
|
32
|
+
export interface Config {
|
|
33
|
+
enabled: boolean;
|
|
34
|
+
gateContent: string;
|
|
35
|
+
interaction: InteractionMode;
|
|
36
|
+
}
|
|
37
|
+
export declare const Config: z<Schemastery.ObjectS<{
|
|
38
|
+
enabled: z<boolean, boolean>;
|
|
39
|
+
gateContent: z<string, string>;
|
|
40
|
+
interaction: z<"ask" | "auto" | "follow-approval", "ask" | "auto" | "follow-approval">;
|
|
41
|
+
}>, Schemastery.ObjectT<{
|
|
42
|
+
enabled: z<boolean, boolean>;
|
|
43
|
+
gateContent: z<string, string>;
|
|
44
|
+
interaction: z<"ask" | "auto" | "follow-approval", "ask" | "auto" | "follow-approval">;
|
|
45
|
+
}>>;
|
|
46
|
+
export declare function apply(ctx: Context, config: Config): void;
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
export declare const LOGICPROBE_VERIFY_TOOL_NAME = "logicprobe_verify";
|
|
2
|
+
/**
|
|
3
|
+
* Model-visible DSH tool wrapping the bundled TypeScript verification engine.
|
|
4
|
+
* The model passes a LogicModelV1 object; the engine validates it and returns
|
|
5
|
+
* the 14-check report. This is the dsh-native replacement for hand-filling the
|
|
6
|
+
* Python template shipped in the skill references.
|
|
7
|
+
*/
|
|
8
|
+
export declare const logicProbeVerifyTool: import("@deepseek-ai/dsh-tools").ToolDefinition;
|
package/package.json
ADDED
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "dsh-logicprobe",
|
|
3
|
+
"version": "0.3.1",
|
|
4
|
+
"description": "Design document & plan claim verification — enumerate claims, verify against codebase facts, then escalate to logic-primitive verification (7 structural checks + 7 adversarial probes) for behavioral claims. Before/after model comparison for refactoring regression detection. Ships a native DeepSeek Harness (dsh) bundle that injects the claim-verification gate into the first model step.",
|
|
5
|
+
"type": "module",
|
|
6
|
+
"main": "lib/index.js",
|
|
7
|
+
"types": "lib/types/index.d.ts",
|
|
8
|
+
"exports": {
|
|
9
|
+
".": {
|
|
10
|
+
"types": "./lib/types/index.d.ts",
|
|
11
|
+
"default": "./lib/index.js"
|
|
12
|
+
},
|
|
13
|
+
"./cordis.patch.yml": "./cordis.patch.yml",
|
|
14
|
+
"./package.json": "./package.json"
|
|
15
|
+
},
|
|
16
|
+
"files": [
|
|
17
|
+
"lib",
|
|
18
|
+
"src",
|
|
19
|
+
"skills",
|
|
20
|
+
"cordis.patch.yml"
|
|
21
|
+
],
|
|
22
|
+
"dsh": {
|
|
23
|
+
"category": "skill",
|
|
24
|
+
"displayName": "逻辑探针",
|
|
25
|
+
"bundle": {
|
|
26
|
+
"patch": "./cordis.patch.yml"
|
|
27
|
+
}
|
|
28
|
+
},
|
|
29
|
+
"scripts": {
|
|
30
|
+
"build": "tsc -p tsconfig.json",
|
|
31
|
+
"typecheck": "tsc -p tsconfig.json --noEmit",
|
|
32
|
+
"test:engine": "npm run build && node tests/engine/run.mjs && node tests/apply-smoke.mjs"
|
|
33
|
+
},
|
|
34
|
+
"dependencies": {},
|
|
35
|
+
"peerDependencies": {
|
|
36
|
+
"@deepseek-ai/cordis": "^4.0.1",
|
|
37
|
+
"@deepseek-ai/dsh-agent": "^0.1.0-rc.6",
|
|
38
|
+
"@deepseek-ai/dsh-llm": "^0.0.1-rc.1",
|
|
39
|
+
"@deepseek-ai/dsh-session": "^0.0.1-rc.1",
|
|
40
|
+
"@deepseek-ai/dsh-skill-filesystem": "^0.1.0-rc.6",
|
|
41
|
+
"@deepseek-ai/schemastery": "^3.18.1",
|
|
42
|
+
"@deepseek-ai/dsh-tools": "^0.1.0-rc.6"
|
|
43
|
+
},
|
|
44
|
+
"devDependencies": {
|
|
45
|
+
"@deepseek-ai/cordis": "^4.0.1",
|
|
46
|
+
"@deepseek-ai/dsh-agent": "^0.1.0-rc.6",
|
|
47
|
+
"@deepseek-ai/dsh-cordis-host-runner": "^0.1.0-rc.6",
|
|
48
|
+
"@deepseek-ai/dsh-home-paths": "^0.1.0-rc.6",
|
|
49
|
+
"@deepseek-ai/dsh-llm": "^0.0.1-rc.1",
|
|
50
|
+
"@deepseek-ai/dsh-scope": "^0.1.0-rc.6",
|
|
51
|
+
"@deepseek-ai/dsh-session": "^0.0.1-rc.1",
|
|
52
|
+
"@deepseek-ai/dsh-skill": "^0.1.0-rc.6",
|
|
53
|
+
"@deepseek-ai/dsh-skill-filesystem": "^0.1.0-rc.6",
|
|
54
|
+
"@deepseek-ai/schemastery": "^3.18.1",
|
|
55
|
+
"@deepseek-ai/dsh-timeout": "^0.0.1-rc.1",
|
|
56
|
+
"@types/node": "^20.0.0",
|
|
57
|
+
"typescript": "^5.0.0",
|
|
58
|
+
"@deepseek-ai/dsh-tools": "^0.1.0-rc.6",
|
|
59
|
+
"@deepseek-ai/dsh-system-prompt": "^0.1.0-rc.6"
|
|
60
|
+
},
|
|
61
|
+
"author": {
|
|
62
|
+
"name": "Amethyst Luna",
|
|
63
|
+
"url": "https://github.com/AmethystLuna"
|
|
64
|
+
},
|
|
65
|
+
"license": "MIT",
|
|
66
|
+
"repository": "https://github.com/AmethystLuna/logicprobe",
|
|
67
|
+
"keywords": [
|
|
68
|
+
"verification",
|
|
69
|
+
"fact-check",
|
|
70
|
+
"design-review",
|
|
71
|
+
"plan-review",
|
|
72
|
+
"state-machine",
|
|
73
|
+
"logic-verification",
|
|
74
|
+
"adversarial",
|
|
75
|
+
"refactoring",
|
|
76
|
+
"model-checking",
|
|
77
|
+
"agentskills",
|
|
78
|
+
"plugin",
|
|
79
|
+
"dsh"
|
|
80
|
+
]
|
|
81
|
+
}
|