dsh-logicprobe 0.3.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/index.ts ADDED
@@ -0,0 +1,301 @@
1
+ /**
2
+ * logicprobe — DeepSeek Harness native plugin for the Logic Probe toolbox.
3
+ * Injects the session-start gate text (claim-verification doctrine, 1% Rule,
4
+ * Red Flags, proactive suggestion) into the first model step of every agent
5
+ * session, mirroring the SessionStart hook the Claude Code plugin installs.
6
+ * The skill ships in this package's `skills/` directory and is registered at
7
+ * apply time into dsh's `ctx.skills` registry through the standard filesystem
8
+ * provider, so it appears in every session catalog without a manual copy step.
9
+ *
10
+ * Injection listens on agent/pre-step and appends the gate to the FIRST
11
+ * model step that runs, once per session (guarded by the session's durable
12
+ * history). Session-start inbox injection was dropped: a blank-session preset
13
+ * switch (agentPreset.select -> recompose) can clear the inbox before the
14
+ * first step, losing the gate for the whole session. The pre-step decision is
15
+ * the durable path - anchored/bootstrap presets that strip first-step injected
16
+ * reminders (skill catalog, AGENTS.md, gate plugins) simply defer this message
17
+ * to the first step after their promotion, and the history guard re-injects it
18
+ * there. The default gate text is the dsh-native adaptation of
19
+ * `hooks/session-start-content.md`: behavior rules
20
+ * (1% Rule / Red Flags / proactive suggestion) stay in sync, while
21
+ * presentation is adapted to dsh's native skill catalog — the trigger list
22
+ * lives in the skill description, not duplicated in the gate. Deployments
23
+ * override via Config.
24
+ *
25
+ * @module logicprobe-dsh
26
+ */
27
+
28
+ import { fileURLToPath } from 'node:url'
29
+ import type { Context } from '@deepseek-ai/cordis'
30
+ import z from '@deepseek-ai/schemastery'
31
+ import { createUserMessage } from '@deepseek-ai/dsh-llm'
32
+ import type { Session, UserMessage } from '@deepseek-ai/dsh-session'
33
+ import type { HostCordisInspectProviderRegistration } from '@deepseek-ai/dsh-cordis-host-runner'
34
+ import type { AssembleContext, PromptContext } from '@deepseek-ai/dsh-system-prompt'
35
+ import { FileSystemSkillProvider } from '@deepseek-ai/dsh-skill-filesystem'
36
+ import { logicProbeVerifyTool } from './tool.js'
37
+ import { ENGINE_SCHEMA_VERSION } from './engine.js'
38
+
39
+ export const name = 'logicprobe'
40
+
41
+ // Skills are contributed through the registry service, which dsh-base always
42
+ // mounts before bundle rows such as this one apply.
43
+ export const inject = ['skills']
44
+
45
+ // Absolute path of the package's shipped skills directory. `lib/index.js`
46
+ // lives one level below the package root, so `../skills` from the module URL
47
+ // lands on `<package>/skills` regardless of where the package was installed.
48
+ const SKILLS_DIR = fileURLToPath(new URL('../skills', import.meta.url))
49
+
50
+ const GATE_PLUGIN_ID = 'logicprobe'
51
+
52
+ export type InteractionMode = 'ask' | 'auto' | 'follow-approval'
53
+
54
+ const DEFAULT_GATE_CONTENT = `<EXTREMELY_IMPORTANT>
55
+ Plugin logicprobe is active. Documents are not truth — code is. Verify every verifiable claim before accepting or acting on any design.
56
+
57
+ **1% Rule**: If there is even a 1% chance the logicprobe skill applies — reviewing design documents, architecture specs, technical proposals, or refactoring plans that make claims about API names, file locations, enum values, mechanism feasibility, state machines, protocol logic, or behavioral guarantees ("always"/"never"/"guaranteed") — load it with the skill tool before responding. The cost of loading is trivial compared to the cost of a false claim.
58
+
59
+ **Red Flags** — if you think any of these, STOP. You are rationalizing:
60
+
61
+ | You think | Reality |
62
+ |-----------|---------|
63
+ | "This plan is too simple to verify" | The skill auto-classifies depth (LIGHTWEIGHT / STANDARD / ESCALATED). You don't decide. |
64
+ | "I already know the file paths are correct" | Organic verification leaves no audit trail. Run Phase 0, append the "## Plan Verification" block. |
65
+ | "I'll verify while implementing" | Verification happens before implementation, not during. |
66
+ | "I can check this with reasoning alone" | Behavioral claims are verified with code/models, not intuition. One counter-example refutes a universal claim. |
67
+
68
+ **Native verification path**: In dsh, prefer the \`logicprobe_verify\` tool for executable state-machine checks. The skill's Python harness remains the fallback for non-dsh hosts.
69
+
70
+ **Proactive suggestion**: When a user asks code-level behavioral questions — "could this state machine deadlock", "is this retry limit safe", "check this timing sequence for bugs" — suggest logicprobe as an optional verification pass (do not auto-escalate).
71
+ </EXTREMELY_IMPORTANT>`
72
+
73
+ export interface Config {
74
+ enabled: boolean
75
+ gateContent: string
76
+ interaction: InteractionMode
77
+ }
78
+
79
+ export const Config = z.object({
80
+ enabled: z.boolean().default(true),
81
+ gateContent: z.string().default(DEFAULT_GATE_CONTENT),
82
+ interaction: z.union(['ask', 'auto', 'follow-approval']).default('follow-approval'),
83
+ })
84
+
85
+ function gateMessage(text: string): UserMessage {
86
+ return createUserMessage({
87
+ content: [{ type: 'text', text }],
88
+ // `form` omitted — an undeclared context is the documented default.
89
+ source: { kind: 'plugin', plugin: GATE_PLUGIN_ID },
90
+ })
91
+ }
92
+
93
+ interface SessionEventLike {
94
+ type: string
95
+ data?: Record<string, unknown>
96
+ }
97
+
98
+ function lastApprovalPolicy(session: Session): 'ask' | 'never' | undefined {
99
+ const events = session.events as readonly SessionEventLike[]
100
+ for (let index = events.length - 1; index >= 0; index -= 1) {
101
+ const event = events[index]
102
+ if (event.type === 'approval/policy') {
103
+ return event.data?.policy === 'never' ? 'never' : 'ask'
104
+ }
105
+ }
106
+ return undefined
107
+ }
108
+
109
+ function planModeActive(session: Session): boolean {
110
+ const events = session.events as readonly SessionEventLike[]
111
+ for (let index = events.length - 1; index >= 0; index -= 1) {
112
+ const event = events[index]
113
+ if (event.type === 'plan/mode') return event.data?.active === true
114
+ }
115
+ return false
116
+ }
117
+
118
+ function resolveInteraction(config: Config, session: Session): 'ask' | 'auto' {
119
+ if (config.interaction === 'ask' || config.interaction === 'auto') return config.interaction
120
+ return lastApprovalPolicy(session) === 'never' ? 'auto' : 'ask'
121
+ }
122
+
123
+ function modeContextText(config: Config, session: Session): string {
124
+ const interaction = resolveInteraction(config, session)
125
+ const lines = [
126
+ 'logicprobe: use the `logicprobe_verify` tool for executable state-machine verification.',
127
+ interaction === 'auto'
128
+ ? 'logicprobe interaction=auto: do NOT call ask_user_question for model confirmation; run round-trip validation of the extracted transition table and mark the result UNCONFIRMED.'
129
+ : 'logicprobe interaction=ask: show the extracted transition table and get user confirmation before running verification.',
130
+ ]
131
+ if (planModeActive(session)) {
132
+ lines.push('Plan mode active: before exit_plan_mode, run logicprobe Phase 0 and append the "## Plan Verification" block to the plan file.')
133
+ }
134
+ return lines.join(' ')
135
+ }
136
+
137
+ interface ToolRegistryLike {
138
+ register(definition: unknown): () => void
139
+ }
140
+
141
+ interface SystemPromptLike {
142
+ context(contribution: PromptContext): () => void
143
+ }
144
+
145
+ /**
146
+ * Model-visible catalog entry (cordis_inspect_list / cordis_inspect_query):
147
+ * lets the model read this plugin's runtime status without guessing. Mirrors
148
+ * the registration pattern of the official dsh-tool-cordis host providers.
149
+ */
150
+ function inspectProvider(config: Config, isToolRegistered: () => boolean): HostCordisInspectProviderRegistration {
151
+ return {
152
+ manifest: {
153
+ id: 'logicprobe',
154
+ description: 'Session-start gate injection and native verification tooling for the Logic Probe toolbox — folds the claim-verification doctrine (1% Rule / Red Flags / proactive suggestion) into the first model step of every agent session and registers the logicprobe_verify tool.',
155
+ methods: [
156
+ {
157
+ name: 'status',
158
+ description: 'Read gate injection status, interaction mode, tool registration state, and engine schema version.',
159
+ inputSchema: {
160
+ type: 'object',
161
+ properties: {},
162
+ additionalProperties: false,
163
+ },
164
+ outputSchema: {
165
+ type: 'object',
166
+ description: 'Gate-injection plugin status.',
167
+ properties: {
168
+ enabled: { type: 'boolean', description: 'Whether the gate folds into the first model step.' },
169
+ gateContentLength: { type: 'integer', description: 'Length in characters of the injected gate text.' },
170
+ interaction: { type: 'string', enum: ['ask', 'auto', 'follow-approval'], description: 'Configured interaction mode. follow-approval resolves per session from approval/policy.' },
171
+ toolRegistered: { type: 'boolean', description: 'Whether the logicprobe_verify tool is registered on ctx.tools.' },
172
+ engineSchemaVersion: { type: 'integer', description: 'Model schema version the bundled verification engine accepts.' },
173
+ },
174
+ required: ['enabled', 'gateContentLength', 'interaction', 'toolRegistered', 'engineSchemaVersion'],
175
+ additionalProperties: false,
176
+ },
177
+ },
178
+ ],
179
+ },
180
+ query: async (method) => {
181
+ if (method === 'status') {
182
+ return {
183
+ enabled: config.enabled,
184
+ gateContentLength: config.gateContent.length,
185
+ interaction: config.interaction,
186
+ toolRegistered: isToolRegistered(),
187
+ engineSchemaVersion: ENGINE_SCHEMA_VERSION,
188
+ }
189
+ }
190
+ return null
191
+ },
192
+ }
193
+ }
194
+
195
+ export function apply(ctx: Context, config: Config): void {
196
+ // Optional services are registered opportunistically; base-bundle rows can
197
+ // mount after this row applies, so registration is retried on the first
198
+ // agent/pre-step — by then the app is fully booted.
199
+ let providerRegistered = false
200
+ let toolRegistered = false
201
+ let modeContextRegistered = false
202
+ const registerProvider = (): void => {
203
+ if (providerRegistered) return
204
+ const inspect = ctx.get('cordisInspect')
205
+ if (inspect === undefined) return
206
+ try {
207
+ ctx.effect(() => inspect.register(inspectProvider(config, () => toolRegistered)), 'logicprobe: inspect provider')
208
+ providerRegistered = true
209
+ } catch (err) {
210
+ console.warn('[logicprobe] inspect provider registration failed', err)
211
+ }
212
+ }
213
+ const registerTool = (): void => {
214
+ if (toolRegistered) return
215
+ const tools = ctx.get('tools') as ToolRegistryLike | undefined
216
+ if (tools === undefined) return
217
+ try {
218
+ ctx.effect(() => tools.register(logicProbeVerifyTool), 'logicprobe: verify tool')
219
+ toolRegistered = true
220
+ } catch (err) {
221
+ console.warn('[logicprobe] logicprobe_verify tool registration failed', err)
222
+ }
223
+ }
224
+ const registerModeContext = (): void => {
225
+ if (modeContextRegistered) return
226
+ const systemPrompt = ctx.get('systemPrompt') as SystemPromptLike | undefined
227
+ if (systemPrompt === undefined) return
228
+ try {
229
+ ctx.effect(() => systemPrompt.context({
230
+ name: 'logicprobe:mode',
231
+ order: 118,
232
+ text: (context: AssembleContext) => {
233
+ const agent = (context as AssembleContext & { agent?: { session: Session } }).agent
234
+ if (agent === undefined) return ''
235
+ return modeContextText(config, agent.session)
236
+ },
237
+ }), 'logicprobe: system prompt context')
238
+ modeContextRegistered = true
239
+ } catch (err) {
240
+ console.warn('[logicprobe] system prompt context registration failed', err)
241
+ }
242
+ }
243
+ const registerIntegrations = (): void => {
244
+ registerProvider()
245
+ registerTool()
246
+ registerModeContext()
247
+ }
248
+ registerIntegrations()
249
+ // Ship the bundled skill through the registry: reuse the standard
250
+ // filesystem provider over this package's own `skills/` directory, so
251
+ // catalog discovery, frontmatter parsing, and SKILL.md loading behave
252
+ // exactly like project/user skills while the plugin stays self-contained.
253
+ // Registration lands in the global registry layer (this row mounts at the
254
+ // profile root), so every agent preset sees the skill. `registerProvider`
255
+ // returns the effect disposer; its teardown unregisters and invalidates.
256
+ ctx.skills.registerProvider((control) => {
257
+ return new FileSystemSkillProvider(ctx, control, {
258
+ providerName: 'logicprobe',
259
+ includeDefaultRoots: false,
260
+ customSkillDirs: [SKILLS_DIR],
261
+ })
262
+ })
263
+ if (!config.enabled) return
264
+ // Inject the gate once per session on the FIRST model step that runs,
265
+ // instead of at session-start: session-start injection lands in the agent's
266
+ // inbox, which a blank-session preset switch (agentPreset.select ->
267
+ // recompose) can clear before the first step - the gate would then be lost
268
+ // for the whole session. The pre-step decision is the durable path a
269
+ // first-step injection takes: the gate is appended to the first step's
270
+ // decision and enters session history there, so every later step (and a
271
+ // resume) skips it. Anchored/bootstrap presets that strip first-step
272
+ // injected reminders (skill catalog, AGENTS.md, gate plugins) simply defer
273
+ // this message to the first step after their promotion - the history guard
274
+ // re-injects it there, so the gate still lands exactly once per session.
275
+ ctx.on('agent/pre-step', async ({ agent }, next) => {
276
+ const decision = await next()
277
+ if (decision.kind === 'reject') return decision
278
+ registerIntegrations()
279
+ if (gateInHistory(agent.session)) return decision
280
+ return {
281
+ kind: 'enter',
282
+ messages: [...decision.messages, gateMessage(config.gateContent)],
283
+ }
284
+ })
285
+ }
286
+
287
+ /**
288
+ * Whether the gate already entered this session's durable history. The
289
+ * pre-step listener re-appends the gate until it does; once a step committed
290
+ * it, every later step (and a resume of a session that kept it) skips the
291
+ * injection. A session whose gate was dropped before any step ran (e.g. an
292
+ * inbox cleared by a blank-session preset switch) simply re-injects on the
293
+ * first step that runs.
294
+ */
295
+ function gateInHistory(session: Session): boolean {
296
+ return session.events.some((event) => {
297
+ if (event.type !== 'user/message') return false
298
+ const source = event.data.source
299
+ return source.kind === 'plugin' && source.plugin === GATE_PLUGIN_ID
300
+ })
301
+ }
package/src/tool.ts ADDED
@@ -0,0 +1,48 @@
1
+ import { defineTool, type JsonValue } from '@deepseek-ai/dsh-tools'
2
+ import { runVerification } from './engine.js'
3
+
4
+ export const LOGICPROBE_VERIFY_TOOL_NAME = 'logicprobe_verify'
5
+
6
+ /**
7
+ * Model-visible DSH tool wrapping the bundled TypeScript verification engine.
8
+ * The model passes a LogicModelV1 object; the engine validates it and returns
9
+ * the 14-check report. This is the dsh-native replacement for hand-filling the
10
+ * Python template shipped in the skill references.
11
+ */
12
+ export const logicProbeVerifyTool = defineTool({
13
+ name: LOGICPROBE_VERIFY_TOOL_NAME,
14
+ description:
15
+ 'Run executable state-machine verification (logicprobe). Takes a LogicModelV1 object with schemaVersion=1, init, states ({id, terminal?}), transitions ({from, event, to, guard?, updates?}), variables?, invariants?, concurrentPairs?, boundaryChecks?, resourcePairs?. Guards are structured ({variable, op, value} | {all} | {any} | {not}); invariants support never-states, var-in-range, and event-before-state. Returns a report with S1-S7 structural checks and A1-A7 adversarial probes including shortest counterexample paths. See skills/logicprobe/references/dsh-model-schema.md.',
16
+ parameters: {
17
+ model: {
18
+ type: 'json',
19
+ required: true,
20
+ description: 'LogicModelV1 state machine model to verify.',
21
+ },
22
+ maxStates: {
23
+ type: 'integer',
24
+ description: 'Maximum runtime states to explore. Default 10000.',
25
+ },
26
+ maxPermutationEvents: {
27
+ type: 'integer',
28
+ description: 'Maximum event count for A3 order permutation. Default 5.',
29
+ },
30
+ },
31
+ output: {
32
+ schema: {
33
+ type: 'json',
34
+ description: 'logicprobe verification report with summary and per-check findings.',
35
+ },
36
+ render(_args, value) {
37
+ return [{ type: 'text' as const, text: JSON.stringify(value, null, 2) }]
38
+ },
39
+ },
40
+ timeoutMs: 10_000,
41
+ isConcurrencySafe: () => true,
42
+ async execute(args) {
43
+ return runVerification(args.model, {
44
+ maxStates: args.maxStates,
45
+ maxPermutationEvents: args.maxPermutationEvents,
46
+ }) as unknown as JsonValue
47
+ },
48
+ })