deepclause-pi 0.4.0 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +4 -4
- package/dist/diagram/extract.d.ts +2 -0
- package/dist/diagram/extract.js +223 -3
- package/dist/diagram/grade.d.ts +1 -1
- package/dist/diagram/grade.js +4 -4
- package/dist/index.d.ts +9 -0
- package/dist/index.js +103 -81
- package/package.json +1 -1
- package/skills/handbook-dml/SKILL.md +174 -46
- package/src/assets/AGENTS.md +43 -7
- package/src/diagram/extract.ts +225 -3
- package/src/diagram/grade.ts +4 -4
- package/src/index.ts +120 -92
package/dist/index.js
CHANGED
|
@@ -275,6 +275,18 @@ function viewerVendorAssetPath() {
|
|
|
275
275
|
function publishResult(pi, content, details) {
|
|
276
276
|
pi.sendMessage({ customType: "deepclause-result", content, display: true, details });
|
|
277
277
|
}
|
|
278
|
+
/**
|
|
279
|
+
* Ask the user a DeepClause question.
|
|
280
|
+
*
|
|
281
|
+
* Pi's interactive text-input dialog renders only its title: the placeholder
|
|
282
|
+
* argument is ignored by `ExtensionInputComponent`. Passing the question as the
|
|
283
|
+
* placeholder (as the extension used to) made it invisible, so put the whole
|
|
284
|
+
* question in the title and label which run is asking.
|
|
285
|
+
*/
|
|
286
|
+
export function requestDeepClauseInput(ctx, label, prompt, signal) {
|
|
287
|
+
const heading = label ? `DeepClause input — ${label}` : "DeepClause input";
|
|
288
|
+
return ctx.ui.input(`${heading}\n\n${prompt}`, undefined, { signal });
|
|
289
|
+
}
|
|
278
290
|
export default function deepClauseExtension(pi) {
|
|
279
291
|
let activeController;
|
|
280
292
|
let activeDescription;
|
|
@@ -284,6 +296,26 @@ export default function deepClauseExtension(pi) {
|
|
|
284
296
|
let planCommitRegistered = false;
|
|
285
297
|
let planningTransaction;
|
|
286
298
|
let pendingAgentStep;
|
|
299
|
+
/**
|
|
300
|
+
* Claim the single DeepClause execution slot synchronously, before any await.
|
|
301
|
+
* Without this, parallel `dc_run` calls all pass the `activeController` check
|
|
302
|
+
* while the first one is still awaiting its setup, then run concurrently and
|
|
303
|
+
* fight over the one-slot pi input dialog (leaving earlier runs hung).
|
|
304
|
+
*/
|
|
305
|
+
const claimExecution = (description) => {
|
|
306
|
+
if (activeController)
|
|
307
|
+
return undefined;
|
|
308
|
+
const controller = new AbortController();
|
|
309
|
+
activeController = controller;
|
|
310
|
+
activeDescription = description;
|
|
311
|
+
return controller;
|
|
312
|
+
};
|
|
313
|
+
const releaseExecution = (controller) => {
|
|
314
|
+
if (activeController === controller) {
|
|
315
|
+
activeController = undefined;
|
|
316
|
+
activeDescription = undefined;
|
|
317
|
+
}
|
|
318
|
+
};
|
|
287
319
|
const setPlanCommitActive = (enabled) => {
|
|
288
320
|
if (enabled && !planCommitRegistered) {
|
|
289
321
|
pi.registerTool({
|
|
@@ -479,6 +511,7 @@ export default function deepClauseExtension(pi) {
|
|
|
479
511
|
promptGuidelines: [
|
|
480
512
|
"Use dc_run only for existing DML programs when their deterministic logic, constraints, or specialized orchestration is useful; do not use dc_run to compile natural language or create a skill.",
|
|
481
513
|
"Do not call dc_run while another DeepClause execution is active, and do not claim success unless dc_run returns an answer without errors.",
|
|
514
|
+
"If dc_run reports execution_already_active, wait for the active run to finish and then retry this skill instead of reporting failure.",
|
|
482
515
|
],
|
|
483
516
|
parameters: Type.Object({
|
|
484
517
|
skill: Type.String({ description: "Skill name such as example, or a DML path relative to .pi/deepclause/." }),
|
|
@@ -488,41 +521,40 @@ export default function deepClauseExtension(pi) {
|
|
|
488
521
|
})),
|
|
489
522
|
}),
|
|
490
523
|
async execute(_toolCallId, params, signal, onUpdate, ctx) {
|
|
491
|
-
|
|
524
|
+
const controller = claimExecution("dc_run");
|
|
525
|
+
if (!controller) {
|
|
492
526
|
return {
|
|
493
|
-
content: [{ type: "text", text: "DeepClause execution rejected: another execution is
|
|
527
|
+
content: [{ type: "text", text: "DeepClause execution rejected: another execution is active. Wait for it to finish, then call dc_run again for this skill." }],
|
|
494
528
|
details: { success: false, error: "execution_already_active" },
|
|
495
529
|
};
|
|
496
530
|
}
|
|
497
|
-
const paths = await initializeWorkspace(ctx.cwd);
|
|
498
|
-
const config = await loadConfig(paths.config);
|
|
499
|
-
if (!config.modelToolEnabled || !pi.getActiveTools().includes(DC_RUN_TOOL)) {
|
|
500
|
-
return {
|
|
501
|
-
content: [{ type: "text", text: "The dc_run tool is disabled. The user can enable it with /dc-tool enable." }],
|
|
502
|
-
details: { success: false, error: "tool_disabled" },
|
|
503
|
-
};
|
|
504
|
-
}
|
|
505
|
-
const mode = params.context ?? config.contextMode;
|
|
506
|
-
const filePath = await resolveDmlPath(paths, params.skill);
|
|
507
|
-
if (await isContextualPlan(filePath)) {
|
|
508
|
-
return {
|
|
509
|
-
content: [{ type: "text", text: "Contextual DML plans must be started by the user with /dc-run; they cannot start a nested pi agent turn from dc_run." }],
|
|
510
|
-
details: { success: false, error: "interactive_plan_requires_user_run" },
|
|
511
|
-
};
|
|
512
|
-
}
|
|
513
|
-
const skillName = path.relative(paths.root, filePath);
|
|
514
|
-
const initialMessages = buildInitialMessages(ctx.sessionManager.getBranch(), mode, config.branchMessageLimit);
|
|
515
|
-
const controller = new AbortController();
|
|
516
531
|
const cancel = () => controller.abort(signal?.reason ?? new Error("dc_run cancelled"));
|
|
517
532
|
if (signal?.aborted)
|
|
518
533
|
cancel();
|
|
519
534
|
else
|
|
520
535
|
signal?.addEventListener("abort", cancel, { once: true });
|
|
521
|
-
activeController = controller;
|
|
522
|
-
activeDescription = `model tool running ${skillName}`;
|
|
523
|
-
const startedAt = Date.now();
|
|
524
|
-
const progress = [];
|
|
525
536
|
try {
|
|
537
|
+
const paths = await initializeWorkspace(ctx.cwd);
|
|
538
|
+
const config = await loadConfig(paths.config);
|
|
539
|
+
if (!config.modelToolEnabled || !pi.getActiveTools().includes(DC_RUN_TOOL)) {
|
|
540
|
+
return {
|
|
541
|
+
content: [{ type: "text", text: "The dc_run tool is disabled. The user can enable it with /dc-tool enable." }],
|
|
542
|
+
details: { success: false, error: "tool_disabled" },
|
|
543
|
+
};
|
|
544
|
+
}
|
|
545
|
+
const mode = params.context ?? config.contextMode;
|
|
546
|
+
const filePath = await resolveDmlPath(paths, params.skill);
|
|
547
|
+
if (await isContextualPlan(filePath)) {
|
|
548
|
+
return {
|
|
549
|
+
content: [{ type: "text", text: "Contextual DML plans must be started by the user with /dc-run; they cannot start a nested pi agent turn from dc_run." }],
|
|
550
|
+
details: { success: false, error: "interactive_plan_requires_user_run" },
|
|
551
|
+
};
|
|
552
|
+
}
|
|
553
|
+
const skillName = path.relative(paths.root, filePath);
|
|
554
|
+
activeDescription = `model tool running ${skillName}`;
|
|
555
|
+
const initialMessages = buildInitialMessages(ctx.sessionManager.getBranch(), mode, config.branchMessageLimit);
|
|
556
|
+
const startedAt = Date.now();
|
|
557
|
+
const progress = [];
|
|
526
558
|
const result = await executeDml(filePath, params.args ?? [], initialMessages, config, pi, ctx, controller, {
|
|
527
559
|
onEvent: (event) => {
|
|
528
560
|
if (event.type === "output" && event.content)
|
|
@@ -544,7 +576,7 @@ export default function deepClauseExtension(pi) {
|
|
|
544
576
|
onInput: async (prompt, inputSignal) => {
|
|
545
577
|
if (!ctx.hasUI)
|
|
546
578
|
throw new Error("dc_run cannot request user input without interactive UI");
|
|
547
|
-
const answer = await ctx
|
|
579
|
+
const answer = await requestDeepClauseInput(ctx, skillName, prompt, inputSignal);
|
|
548
580
|
if (answer === undefined)
|
|
549
581
|
throw new Error("Input cancelled");
|
|
550
582
|
return answer;
|
|
@@ -571,13 +603,12 @@ export default function deepClauseExtension(pi) {
|
|
|
571
603
|
const message = error instanceof Error ? error.message : String(error);
|
|
572
604
|
return {
|
|
573
605
|
content: [{ type: "text", text: `DeepClause execution failed: ${message}` }],
|
|
574
|
-
details: { success: false,
|
|
606
|
+
details: { success: false, error: message },
|
|
575
607
|
};
|
|
576
608
|
}
|
|
577
609
|
finally {
|
|
578
610
|
signal?.removeEventListener("abort", cancel);
|
|
579
|
-
|
|
580
|
-
activeDescription = undefined;
|
|
611
|
+
releaseExecution(controller);
|
|
581
612
|
}
|
|
582
613
|
},
|
|
583
614
|
});
|
|
@@ -596,11 +627,11 @@ export default function deepClauseExtension(pi) {
|
|
|
596
627
|
pi.registerTool({
|
|
597
628
|
name: DC_DIAGRAM_TOOL,
|
|
598
629
|
label: "Create DeepClause Diagram",
|
|
599
|
-
description: "Create a presentation-grade or specification-grade Mermaid diagram from any .dml file, write a self-contained offline viewer under .pi/deepclause/diagrams/, and open it.",
|
|
630
|
+
description: "Create a presentation-grade or specification-grade Mermaid diagram from any .dml file, write a self-contained offline viewer under .pi/deepclause/diagrams/, and open it. The specification grade preserves the core decision logic and rule facts.",
|
|
600
631
|
promptSnippet: "Create a presentation- or specification-grade diagram from a DML file",
|
|
601
632
|
promptGuidelines: [
|
|
602
633
|
"Use dc_diagram whenever the user asks for a diagram, flowchart, or visual of a .dml file; pass the exact path the user named.",
|
|
603
|
-
"Choose grade=presentation for slides and overviews and grade=specification for engineering detail; use grade=both only when the user asks for both.",
|
|
634
|
+
"Choose grade=presentation for slides and overviews and grade=specification for engineering detail (decision predicates, thresholds, rule facts); use grade=both only when the user asks for both.",
|
|
604
635
|
"Do not hand-write Mermaid or run diagram tools yourself; call dc_diagram and report the viewer result.",
|
|
605
636
|
],
|
|
606
637
|
parameters: Type.Object({
|
|
@@ -609,52 +640,47 @@ export default function deepClauseExtension(pi) {
|
|
|
609
640
|
view: Type.Optional(StringEnum(["flow", "sequence"], { description: "Base layout used to seed the grade; default flow." })),
|
|
610
641
|
}),
|
|
611
642
|
async execute(_toolCallId, params, signal, onUpdate, ctx) {
|
|
612
|
-
|
|
643
|
+
const controller = claimExecution("dc_diagram");
|
|
644
|
+
if (!controller) {
|
|
613
645
|
return {
|
|
614
646
|
content: [{ type: "text", text: "Another DeepClause operation is already active; wait for it to finish." }],
|
|
615
647
|
details: { success: false, error: "execution_already_active" },
|
|
616
648
|
};
|
|
617
649
|
}
|
|
618
|
-
const requested = resolveGrade(String(params.grade ?? "")) ?? "presentation";
|
|
619
|
-
const grades = requested === "both" ? ["presentation", "specification"] : [requested];
|
|
620
|
-
const view = params.view === "sequence" ? "sequence" : "flow";
|
|
621
|
-
let sourcePath;
|
|
622
|
-
try {
|
|
623
|
-
sourcePath = await resolveDiagramSource(ctx.cwd, params.dml);
|
|
624
|
-
}
|
|
625
|
-
catch (error) {
|
|
626
|
-
const message = error instanceof Error ? error.message : String(error);
|
|
627
|
-
return { content: [{ type: "text", text: `dc_diagram failed: ${message}` }], details: { success: false, error: message } };
|
|
628
|
-
}
|
|
629
|
-
if (!ctx.model) {
|
|
630
|
-
return {
|
|
631
|
-
content: [{ type: "text", text: "dc_diagram requires an active pi model. Select one and try again." }],
|
|
632
|
-
details: { success: false, error: "no_model" },
|
|
633
|
-
};
|
|
634
|
-
}
|
|
635
|
-
const config = await loadConfig(getPaths(ctx.cwd).config);
|
|
636
|
-
const source = await readFile(sourcePath, "utf8");
|
|
637
|
-
const display = displayPath(ctx.cwd, sourcePath);
|
|
638
|
-
const seed = view === "sequence"
|
|
639
|
-
? renderSequence(display, source)
|
|
640
|
-
: renderDml(display, source, { hideOutput: true });
|
|
641
|
-
const targets = await collectDiagramTargets(ctx.cwd, [sourcePath]);
|
|
642
|
-
const name = diagramNameFor(sourcePath, targets, ctx.cwd);
|
|
643
|
-
const { diagrams, vendor } = await ensureDiagramDir(ctx.cwd, viewerVendorAssetPath());
|
|
644
|
-
const templateText = await bundledViewerTemplate();
|
|
645
|
-
const controller = new AbortController();
|
|
646
650
|
const cancel = () => controller.abort(signal?.reason ?? new Error("dc_diagram cancelled"));
|
|
647
651
|
if (signal?.aborted)
|
|
648
652
|
cancel();
|
|
649
653
|
else
|
|
650
654
|
signal?.addEventListener("abort", cancel, { once: true });
|
|
651
|
-
activeController = controller;
|
|
652
|
-
activeDescription = `diagram ${name} (${grades.join("+")})`;
|
|
653
|
-
const run = (command, args, options) => pi.exec(command, args, options);
|
|
654
|
-
let chrome;
|
|
655
|
-
let chromeResolved = false;
|
|
656
655
|
try {
|
|
656
|
+
const requested = resolveGrade(String(params.grade ?? "")) ?? "presentation";
|
|
657
|
+
const grades = requested === "both" ? ["presentation", "specification"] : [requested];
|
|
658
|
+
const view = params.view === "sequence" ? "sequence" : "flow";
|
|
659
|
+
const sourcePath = await resolveDiagramSource(ctx.cwd, params.dml);
|
|
660
|
+
if (!ctx.model) {
|
|
661
|
+
return {
|
|
662
|
+
content: [{ type: "text", text: "dc_diagram requires an active pi model. Select one and try again." }],
|
|
663
|
+
details: { success: false, error: "no_model" },
|
|
664
|
+
};
|
|
665
|
+
}
|
|
666
|
+
const config = await loadConfig(getPaths(ctx.cwd).config);
|
|
667
|
+
const source = await readFile(sourcePath, "utf8");
|
|
668
|
+
const display = displayPath(ctx.cwd, sourcePath);
|
|
669
|
+
const targets = await collectDiagramTargets(ctx.cwd, [sourcePath]);
|
|
670
|
+
const name = diagramNameFor(sourcePath, targets, ctx.cwd);
|
|
671
|
+
const { diagrams, vendor } = await ensureDiagramDir(ctx.cwd, viewerVendorAssetPath());
|
|
672
|
+
const templateText = await bundledViewerTemplate();
|
|
673
|
+
activeDescription = `diagram ${name} (${grades.join("+")})`;
|
|
674
|
+
const run = (command, args, options) => pi.exec(command, args, options);
|
|
675
|
+
let chrome;
|
|
676
|
+
let chromeResolved = false;
|
|
657
677
|
for (const grade of grades) {
|
|
678
|
+
// The specification seed carries the core decision logic and rule
|
|
679
|
+
// facts; the presentation seed stays small so the model can keep
|
|
680
|
+
// it to a general-audience overview.
|
|
681
|
+
const seed = view === "sequence"
|
|
682
|
+
? renderSequence(display, source)
|
|
683
|
+
: renderDml(display, source, { hideOutput: true, includeLogic: grade === "specification" });
|
|
658
684
|
const result = await polishDiagram({
|
|
659
685
|
grade,
|
|
660
686
|
view,
|
|
@@ -700,8 +726,7 @@ export default function deepClauseExtension(pi) {
|
|
|
700
726
|
}
|
|
701
727
|
finally {
|
|
702
728
|
signal?.removeEventListener("abort", cancel);
|
|
703
|
-
|
|
704
|
-
activeDescription = undefined;
|
|
729
|
+
releaseExecution(controller);
|
|
705
730
|
}
|
|
706
731
|
},
|
|
707
732
|
});
|
|
@@ -768,13 +793,13 @@ export default function deepClauseExtension(pi) {
|
|
|
768
793
|
}
|
|
769
794
|
};
|
|
770
795
|
const runSpecSkill = async (ctx, skill, args = [], options = {}) => {
|
|
771
|
-
const
|
|
772
|
-
|
|
773
|
-
|
|
774
|
-
const controller = new AbortController();
|
|
775
|
-
activeController = controller;
|
|
776
|
-
activeDescription = `running ${skill}`;
|
|
796
|
+
const controller = claimExecution(`running ${skill}`);
|
|
797
|
+
if (!controller)
|
|
798
|
+
throw new Error("Another DeepClause execution is already active");
|
|
777
799
|
try {
|
|
800
|
+
const paths = await initializeWorkspace(ctx.cwd);
|
|
801
|
+
const config = await loadConfig(paths.config);
|
|
802
|
+
const filePath = await resolveDmlPath(paths, skill);
|
|
778
803
|
const result = await executeDml(filePath, args, [], config, pi, ctx, controller, {
|
|
779
804
|
onEvent() { },
|
|
780
805
|
onInput: async () => { throw new Error("spec skills do not request input"); },
|
|
@@ -784,8 +809,7 @@ export default function deepClauseExtension(pi) {
|
|
|
784
809
|
return result.answer ?? "(no answer)";
|
|
785
810
|
}
|
|
786
811
|
finally {
|
|
787
|
-
|
|
788
|
-
activeDescription = undefined;
|
|
812
|
+
releaseExecution(controller);
|
|
789
813
|
}
|
|
790
814
|
};
|
|
791
815
|
pi.on("session_start", async (_event, ctx) => {
|
|
@@ -1194,7 +1218,8 @@ export default function deepClauseExtension(pi) {
|
|
|
1194
1218
|
pi.registerCommand("dc-run", {
|
|
1195
1219
|
description: "Run a DML skill with pi's active model",
|
|
1196
1220
|
handler: async (rawArgs, ctx) => {
|
|
1197
|
-
|
|
1221
|
+
const controller = claimExecution("dc-run");
|
|
1222
|
+
if (!controller) {
|
|
1198
1223
|
ctx.ui.notify("A DeepClause execution is already active", "warning");
|
|
1199
1224
|
return;
|
|
1200
1225
|
}
|
|
@@ -1229,8 +1254,6 @@ export default function deepClauseExtension(pi) {
|
|
|
1229
1254
|
}
|
|
1230
1255
|
const mode = parsed.contextMode ?? config.contextMode;
|
|
1231
1256
|
const initialMessages = buildInitialMessages(ctx.sessionManager.getBranch(), mode, config.branchMessageLimit);
|
|
1232
|
-
const controller = new AbortController();
|
|
1233
|
-
activeController = controller;
|
|
1234
1257
|
const outputLines = [];
|
|
1235
1258
|
const recentEvents = [];
|
|
1236
1259
|
const events = [];
|
|
@@ -1309,7 +1332,7 @@ export default function deepClauseExtension(pi) {
|
|
|
1309
1332
|
onEvent,
|
|
1310
1333
|
onDiagnostic,
|
|
1311
1334
|
onInput: async (prompt, signal) => {
|
|
1312
|
-
const answer = await ctx
|
|
1335
|
+
const answer = await requestDeepClauseInput(ctx, skillName, prompt, signal);
|
|
1313
1336
|
if (answer === undefined)
|
|
1314
1337
|
throw new Error("Input cancelled");
|
|
1315
1338
|
return answer;
|
|
@@ -1340,8 +1363,7 @@ export default function deepClauseExtension(pi) {
|
|
|
1340
1363
|
ctx.ui.notify(message, activeController?.signal.aborted ? "warning" : "error");
|
|
1341
1364
|
}
|
|
1342
1365
|
finally {
|
|
1343
|
-
|
|
1344
|
-
activeDescription = undefined;
|
|
1366
|
+
releaseExecution(controller);
|
|
1345
1367
|
ctx.ui.setStatus(STATUS_KEY, undefined);
|
|
1346
1368
|
ctx.ui.setWidget(WIDGET_KEY, undefined);
|
|
1347
1369
|
}
|
package/package.json
CHANGED
|
@@ -1,14 +1,16 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: handbook-dml
|
|
3
|
-
description: Convert a long handbook/SOP into one DeepClause DML skill per workflow —
|
|
3
|
+
description: Convert a long handbook/SOP into one DeepClause DML skill per workflow — a `task/N` subagent for agentic work, the judgment predicates (`choose`/`rate`/`verify`/`probability`/`holds`/`judge`) for bounded classification and calibrated gates, and deterministic Prolog only for mechanical checks, always with fallbacks. Also teaches a tool audit, user confirmation, and how to write the repo-root AGENTS.md policy-routing table so pi calls the skills automatically via dc_run. Use when asked to turn a handbook or procedures manual into executable DML, update a handbook-derived skill, choose between a classifier, a probability gate, and a full agent task, or wire the policy router.
|
|
4
4
|
---
|
|
5
5
|
|
|
6
6
|
# Handbook → DML (procedures)
|
|
7
7
|
|
|
8
|
-
Turn a long handbook into **one DML skill per workflow/procedure**. Each skill
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
8
|
+
Turn a long handbook into **one DML skill per workflow/procedure**. Each skill
|
|
9
|
+
combines three primitives: agentic `task/N` leaves for work that must act or
|
|
10
|
+
reason across turns, bounded judgment predicates (`choose`/`rate`/`verify`/
|
|
11
|
+
`probability`/`holds`) for classification and calibrated gates, and Prolog only
|
|
12
|
+
for what is genuinely mechanical (arithmetic, counting, exact equality) or a
|
|
13
|
+
hard safety invariant.
|
|
12
14
|
|
|
13
15
|
Before writing DML, read `.pi/deepclause/AGENTS.md` and
|
|
14
16
|
`.pi/deepclause/DML_REFERENCE.md`. They are authoritative for syntax.
|
|
@@ -17,19 +19,117 @@ Before writing DML, read `.pi/deepclause/AGENTS.md` and
|
|
|
17
19
|
|
|
18
20
|
- **Pi authors; DML is the runtime artifact.** Decomposition and authoring happen
|
|
19
21
|
in a normal pi turn. There is no Markdown→DML compiler.
|
|
20
|
-
- **
|
|
21
|
-
|
|
22
|
+
- **Pick the cheapest primitive.** Bounded classifications, ratings, yes/no
|
|
23
|
+
checks, and probabilities use the judgment predicates (`choose/4`, `rate/4`,
|
|
24
|
+
`verify/3`, `probability/3`, `holds/2-3`, `judge/2`). Fresh-context generation
|
|
25
|
+
and critique use `prompt/N`. Multi-turn work with memory, tools, and typed
|
|
26
|
+
outputs uses `task/N`. Use Prolog only where a rule is mechanical or must
|
|
27
|
+
never be wrong. See **Choosing the reasoning primitive** below.
|
|
22
28
|
- **Input is a generic request.** `agent_main(Request)` takes free text; the
|
|
23
29
|
first step is an LLM `task/N` that parses it into a typed `object/1` case (or
|
|
24
30
|
the request is used directly for simple skills).
|
|
25
31
|
- **Forbidden actions become tool scoping.** "Never send" means *no send tool*.
|
|
26
32
|
"Read-only inspection" means the inspect phase gets read tools only.
|
|
27
|
-
- **Every deterministic rule gets
|
|
28
|
-
fails, branch to a `prompt/N`/`task/N
|
|
29
|
-
|
|
33
|
+
- **Every deterministic rule gets a fallback.** If a deterministic check
|
|
34
|
+
fails, branch to a judgment, a `prompt/N`/`task/N`, or the user — do not
|
|
35
|
+
hard-fail.
|
|
30
36
|
- **Ask, don't assume.** Confirm scope, the decomposition, and tool choices with
|
|
31
37
|
the user before authoring.
|
|
32
38
|
|
|
39
|
+
## Choosing the reasoning primitive
|
|
40
|
+
|
|
41
|
+
Before writing flow, ask what the **core question** of each step is. A full
|
|
42
|
+
`task/N` agent loop carries memory, DML tools, and a multi-turn reasoning loop;
|
|
43
|
+
it is the right answer only when the step must act or reason iteratively. A
|
|
44
|
+
bounded question over explicit text should use the judgment predicates instead:
|
|
45
|
+
they are cheaper, the answer is constrained to the labels/levels you supply,
|
|
46
|
+
they never touch DML memory, and they cannot call tools.
|
|
47
|
+
|
|
48
|
+
| Core question | Primitive | Notes |
|
|
49
|
+
| --- | --- | --- |
|
|
50
|
+
| "Which category/team/route?" from a small closed set | `choose/4` (or `judge/2` with `choose`) | The answer is always one of your option atoms |
|
|
51
|
+
| "How severe/frustrated/confident?" (ordered) | `rate/4` | Answer is constrained to your levels |
|
|
52
|
+
| "Is X true?" / "Does it ask for a refund?" | `verify/3`, or `holds/2` when only the boolean matters | Three-valued: `yes` / `no` / `unknown` |
|
|
53
|
+
| "How likely is X?" / "Is it over a threshold?" | `probability/3`, or `holds/3` for a threshold gate | Decision-grade only when the backend reports `calibrated` |
|
|
54
|
+
| Free-form generation, rewrite, or critique of supplied text | `prompt/N` | Fresh context, no tools, no memory |
|
|
55
|
+
| Multi-step job needing tools, memory, or iteration | `task/N` | Typed outputs; scope tools with `with_tools/2` |
|
|
56
|
+
|
|
57
|
+
**Simple classifier.** If the core question reduces to choosing among a fixed
|
|
58
|
+
set of labels, use `choose/4`. Do not spend a `task/N` on it, and do not let the
|
|
59
|
+
model invent a label: the judge is constrained to your options.
|
|
60
|
+
|
|
61
|
+
```prolog
|
|
62
|
+
route(Request, Team) :-
|
|
63
|
+
choose(Request, "Which team should handle this request?",
|
|
64
|
+
[billing-"Charges and refunds", orders-"Delivery and returns", account-"Login and security"],
|
|
65
|
+
Team).
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
**Calibrated probability.** If the decision depends on a probability, say so and
|
|
69
|
+
gate it. The default `llm` backend is *uncalibrated*: a number from it is an
|
|
70
|
+
estimate, not a calibrated probability. Wrap the judgment in
|
|
71
|
+
`require_judgment(calibrated, ...)` so the skill cannot silently run on a
|
|
72
|
+
backend that only estimates, and provide a fallback clause.
|
|
73
|
+
|
|
74
|
+
```prolog
|
|
75
|
+
risk_band(Text, Band) :-
|
|
76
|
+
require_judgment(calibrated,
|
|
77
|
+
probability(Text, "What is the probability this is high risk?", P)),
|
|
78
|
+
( P >= 0.8 -> Band = high
|
|
79
|
+
; P >= 0.4 -> Band = medium
|
|
80
|
+
; Band = low
|
|
81
|
+
).
|
|
82
|
+
|
|
83
|
+
risk_band(Text, Band) :-
|
|
84
|
+
% Uncalibrated fallback: a coarse classifier, never a fake probability.
|
|
85
|
+
choose(Text, "Is this high, medium, or low risk?", [high, medium, low], Band).
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
`holds(Text, Question, Threshold)` is the concise form when you only need the
|
|
89
|
+
threshold decision, and `holds(Text, Question)` is the concise form of a yes/no
|
|
90
|
+
check. `with_judgment(jev, Goal)` pins a scope to a specific backend (for
|
|
91
|
+
example a calibrated one). `require_judgment/2` fails **before** the judgment
|
|
92
|
+
runs, which is what makes the fallback clause above reachable; the runtime emits
|
|
93
|
+
a warning naming the missing capability.
|
|
94
|
+
|
|
95
|
+
**Full agent loop.** Use `task/N` (or `prompt/N`) when the step needs to read
|
|
96
|
+
accumulated memory, call a DML tool, run several model turns, or produce
|
|
97
|
+
free-form text. The judgment layer deliberately cannot do those things.
|
|
98
|
+
|
|
99
|
+
**Batch judgments.** When one step needs more than one or two judgments, send a
|
|
100
|
+
single `judge/2` batch rather than several one-offs:
|
|
101
|
+
|
|
102
|
+
```prolog
|
|
103
|
+
judge(Message, [
|
|
104
|
+
choose("Which team should handle this?", [billing, orders, account]) - Team,
|
|
105
|
+
rate("How frustrated is the customer?", [calm, frustrated, angry]) - Frustration,
|
|
106
|
+
verify("Does the message ask for a refund?") - Refund,
|
|
107
|
+
probability("Is this urgent?") - Urgency
|
|
108
|
+
]).
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
### Judgment rules
|
|
112
|
+
|
|
113
|
+
- `judge/2`, `choose/4`, `rate/4`, `verify/3`, `probability/3`, `holds/2`,
|
|
114
|
+
`holds/3`, `with_judgment/2`, and `require_judgment/2` are runtime special
|
|
115
|
+
predicates. Do not define your own predicates with those names — a common
|
|
116
|
+
collision is a hand-written `verify/3`. Name deterministic helpers
|
|
117
|
+
`verify_task/3`, `check_*`, and so on.
|
|
118
|
+
- `verify` is three-valued. `\+ verify(...)` does **not** mean "the model said
|
|
119
|
+
no"; inspect the returned `yes` / `no` / `unknown`.
|
|
120
|
+
- Keep `State` explicit: pass the exact text or object to judge. The judge sees
|
|
121
|
+
only that state plus the question and never reads DML memory.
|
|
122
|
+
- A `choose`/`rate` answer is constrained to your options/levels, so make them
|
|
123
|
+
exhaustive and mutually exclusive.
|
|
124
|
+
- Never present an `estimated` probability as `calibrated`. Require the
|
|
125
|
+
capability and fall back, or label the output as an estimate.
|
|
126
|
+
- Answers are memoized per run by backend, model, state, and questions, so
|
|
127
|
+
backtracking reuses an answer instead of re-querying the model.
|
|
128
|
+
- The deterministic linter warns when a skill has no `task()` call and suggests
|
|
129
|
+
`task()` for classification. That heuristic predates the judgment layer: a
|
|
130
|
+
skill whose core question is a bounded judgment can legitimately have no
|
|
131
|
+
`task/N`.
|
|
132
|
+
|
|
33
133
|
## Workflow
|
|
34
134
|
|
|
35
135
|
1. **Ingest** the handbook to Markdown (PDF→`pdftotext`, DOCX/HTML→`pandoc`).
|
|
@@ -121,6 +221,13 @@ tool(user_feedback(Prompt, Response), "Ask the user one focused question and ret
|
|
|
121
221
|
exec(ask_user(prompt: Prompt), Result),
|
|
122
222
|
get_dict(user_response, Result, Response).
|
|
123
223
|
|
|
224
|
+
% --- bounded judgments (cheap: no memory, no tools) ------------------------
|
|
225
|
+
% Prefer a judgment over a task/N for a bounded question about explicit text.
|
|
226
|
+
% classify(Request, Kind) :-
|
|
227
|
+
% choose(Request, "Which case type is this?", [<kind_a>, <kind_b>, other], Kind).
|
|
228
|
+
% confirm_requirement(Report) :-
|
|
229
|
+
% holds(Report, "Does the report include the required referral advice?", 0.7).
|
|
230
|
+
|
|
124
231
|
% --- deterministic helpers (mechanical only) -------------------------------
|
|
125
232
|
<compute_or_check>(...). % arithmetic/counts; keep small
|
|
126
233
|
|
|
@@ -136,7 +243,7 @@ agent_main(Request) :-
|
|
|
136
243
|
with_tools([<write tools>], (
|
|
137
244
|
task("Produce the required effects for this case: {ConfirmedCase}.", string(Summary))
|
|
138
245
|
)),
|
|
139
|
-
<optional verification with
|
|
246
|
+
<optional judgment / prompt / deterministic verification with fallback>,
|
|
140
247
|
answer(Final).
|
|
141
248
|
|
|
142
249
|
agent_main(_) :-
|
|
@@ -146,49 +253,59 @@ agent_main(_) :-
|
|
|
146
253
|
Notes:
|
|
147
254
|
|
|
148
255
|
- `task/N` = agentic leaf (memory + DML tools). `prompt/N` = fresh-context
|
|
149
|
-
review. `
|
|
256
|
+
generation/review. `choose`/`rate`/`verify`/`probability` = the judgment layer.
|
|
257
|
+
`with_tools/2` scopes capability per phase.
|
|
258
|
+
- Prefer a judgment over a `task/N` whenever the core question is a bounded
|
|
259
|
+
classification, rating, yes/no check, or probability gate.
|
|
150
260
|
- Build `task/N` descriptions with `format/3` (not `{Var}` interpolation) when
|
|
151
261
|
you embed dynamic values — avoids singleton-variable noise.
|
|
152
262
|
- Mutable facts must be declared `:- dynamic` before `assertz`/`retract`.
|
|
153
263
|
|
|
154
|
-
## Verification (optional
|
|
264
|
+
## Verification (optional)
|
|
265
|
+
|
|
266
|
+
Only add checks when the procedure has observable post-conditions. Match the
|
|
267
|
+
check to the question: a bounded semantic check is a judgment, an open-ended
|
|
268
|
+
read is a model review, and a count is deterministic. Every check needs a
|
|
269
|
+
fallback path so it never hard-fails.
|
|
155
270
|
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
271
|
+
- **Semantic gate (default for bounded checks).** "Does the output recommend
|
|
272
|
+
urgent referral when a danger sign is present?" is a `verify/3` question, or
|
|
273
|
+
`holds/2` when you only need the boolean. Use `holds/3` when the check is a
|
|
274
|
+
probability threshold.
|
|
275
|
+
- **Model review (`prompt/N`).** Tone, completeness, correctness of free text,
|
|
276
|
+
"does this read right", and open-ended rubric interpretation.
|
|
277
|
+
- **Deterministic (only when mechanical).** Counts, exact IDs, arithmetic. Keep
|
|
278
|
+
it tiny, and route failures to a judgment, a review, or the user.
|
|
159
279
|
|
|
160
280
|
```prolog
|
|
161
|
-
%
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
281
|
+
% Semantic gate: a bounded yes/no question about explicit text.
|
|
282
|
+
referral_ok(Report) :-
|
|
283
|
+
holds(Report, "Does the report recommend urgent referral when a danger sign is present?").
|
|
284
|
+
|
|
285
|
+
% Deterministic gate (mechanical only).
|
|
286
|
+
drafts_complete :- findall(_, draft(_,_,_,_), Ds), length(Ds, 2).
|
|
287
|
+
|
|
288
|
+
verify_state(Report, Note) :-
|
|
289
|
+
( referral_ok(Report), drafts_complete
|
|
290
|
+
-> Note = "verification passed"
|
|
291
|
+
; % Fallback: never hard-fail; explain the failure and let pi/user decide.
|
|
292
|
+
format(string(Prompt),
|
|
293
|
+
"Review this outcome against the requirement and explain any gap: ~w. Store 'acceptable' or 'needs-attention' in Verdict and a one-line reason in Reason.",
|
|
294
|
+
[Report]),
|
|
295
|
+
prompt(Prompt, string(Verdict), string(Reason)),
|
|
296
|
+
format(string(Note), "fallback verdict ~w: ~w", [Verdict, Reason])
|
|
297
|
+
).
|
|
174
298
|
```
|
|
175
299
|
|
|
176
300
|
In `agent_main`:
|
|
177
301
|
|
|
178
302
|
```prolog
|
|
179
|
-
verify_state(
|
|
180
|
-
(
|
|
181
|
-
; fallback_review(Failed, V, R)
|
|
182
|
-
),
|
|
183
|
-
answer(... report Report + V/R ...).
|
|
303
|
+
verify_state(Report, Note),
|
|
304
|
+
answer(... report + Note ...).
|
|
184
305
|
```
|
|
185
306
|
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
- **Model review** (default): tone, completeness, correctness of free text,
|
|
189
|
-
"does this read right", and anything the rubric phrases as a judgment.
|
|
190
|
-
- **Deterministic** (only when mechanical): counts, exact IDs, arithmetic. Keep
|
|
191
|
-
it tiny, and route failures to the model or the user instead of failing.
|
|
307
|
+
For a calibrated postcondition, wrap `holds(Report, Question, Threshold)` in
|
|
308
|
+
`require_judgment(calibrated, ...)` and keep the same fallback pattern.
|
|
192
309
|
|
|
193
310
|
## Dummy tools
|
|
194
311
|
|
|
@@ -243,10 +360,13 @@ For each generated skill:
|
|
|
243
360
|
— runs clean; verification (if any) passes or the fallback explains.
|
|
244
361
|
2. Inspect phase is read-only, action phase is write-only, and a forbidden
|
|
245
362
|
action has no tool at all.
|
|
246
|
-
3.
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
363
|
+
3. Bounded classifications and checks use the judgment predicates; calibrated
|
|
364
|
+
probabilities are wrapped in `require_judgment(calibrated, ...)` with a
|
|
365
|
+
fallback clause, and `verify/3` results are read as `yes`/`no`/`unknown`.
|
|
366
|
+
4. Deterministic code is limited to arithmetic/counts; every deterministic
|
|
367
|
+
check has a model/judgment fallback branch.
|
|
368
|
+
5. The tool audit was done and its result is recorded in the header/INDEX.
|
|
369
|
+
6. Re-check the invalid-pattern table in `.pi/deepclause/AGENTS.md` (singleton
|
|
250
370
|
variables, `~` vs `{}` interpolation, `Result.field` vs `get_dict/3`, `->`
|
|
251
371
|
committing over generators, `answer/1` last, `:- dynamic` before
|
|
252
372
|
`assertz/retract`).
|
|
@@ -254,12 +374,20 @@ For each generated skill:
|
|
|
254
374
|
`--context=isolated` keeps session text out of the run. The runtime still needs
|
|
255
375
|
a model selected.
|
|
256
376
|
|
|
377
|
+
To exercise calibrated judgments for real, enable the Jev backend
|
|
378
|
+
(`/dc-judge enable`, `/dc-judge default jev`, or `--judge=jev`) and export the
|
|
379
|
+
backend's API key (default `TYPESAFE_API_KEY`) in the shell that launches pi.
|
|
380
|
+
Without it, the `llm` backend is uncalibrated and the
|
|
381
|
+
`require_judgment(calibrated, ...)` branch is skipped in favor of the fallback.
|
|
382
|
+
|
|
257
383
|
## Reference shape
|
|
258
384
|
|
|
259
385
|
A typical procedure skill has: `agent_main(Request)` that parses the request
|
|
260
|
-
with `task/N`, confirms with the user via a `user_feedback` tool loop,
|
|
386
|
+
with `task/N`, confirms with the user via a `user_feedback` tool loop, uses
|
|
387
|
+
bounded judgments (`choose`/`verify`/`holds`) for classification and gates, acts
|
|
261
388
|
through write tools, and reviews with `prompt/N` — with small deterministic
|
|
262
389
|
helpers for arithmetic and a fallback branch instead of hard failures.
|
|
263
390
|
|
|
264
391
|
For DML mechanics, see the bundled example skills in a fresh workspace
|
|
265
|
-
(`example.dml`, `deep_research.dml`)
|
|
392
|
+
(`example.dml`, `deep_research.dml`), `.pi/deepclause/DML_REFERENCE.md` (the
|
|
393
|
+
judgment predicates are documented there), and `.pi/deepclause/AGENTS.md`.
|