deepclause-pi 0.4.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -275,6 +275,18 @@ function viewerVendorAssetPath() {
275
275
  function publishResult(pi, content, details) {
276
276
  pi.sendMessage({ customType: "deepclause-result", content, display: true, details });
277
277
  }
278
+ /**
279
+ * Ask the user a DeepClause question.
280
+ *
281
+ * Pi's interactive text-input dialog renders only its title: the placeholder
282
+ * argument is ignored by `ExtensionInputComponent`. Passing the question as the
283
+ * placeholder (as the extension used to) made it invisible, so put the whole
284
+ * question in the title and label which run is asking.
285
+ */
286
+ export function requestDeepClauseInput(ctx, label, prompt, signal) {
287
+ const heading = label ? `DeepClause input — ${label}` : "DeepClause input";
288
+ return ctx.ui.input(`${heading}\n\n${prompt}`, undefined, { signal });
289
+ }
278
290
  export default function deepClauseExtension(pi) {
279
291
  let activeController;
280
292
  let activeDescription;
@@ -284,6 +296,26 @@ export default function deepClauseExtension(pi) {
284
296
  let planCommitRegistered = false;
285
297
  let planningTransaction;
286
298
  let pendingAgentStep;
299
+ /**
300
+ * Claim the single DeepClause execution slot synchronously, before any await.
301
+ * Without this, parallel `dc_run` calls all pass the `activeController` check
302
+ * while the first one is still awaiting its setup, then run concurrently and
303
+ * fight over the one-slot pi input dialog (leaving earlier runs hung).
304
+ */
305
+ const claimExecution = (description) => {
306
+ if (activeController)
307
+ return undefined;
308
+ const controller = new AbortController();
309
+ activeController = controller;
310
+ activeDescription = description;
311
+ return controller;
312
+ };
313
+ const releaseExecution = (controller) => {
314
+ if (activeController === controller) {
315
+ activeController = undefined;
316
+ activeDescription = undefined;
317
+ }
318
+ };
287
319
  const setPlanCommitActive = (enabled) => {
288
320
  if (enabled && !planCommitRegistered) {
289
321
  pi.registerTool({
@@ -479,6 +511,7 @@ export default function deepClauseExtension(pi) {
479
511
  promptGuidelines: [
480
512
  "Use dc_run only for existing DML programs when their deterministic logic, constraints, or specialized orchestration is useful; do not use dc_run to compile natural language or create a skill.",
481
513
  "Do not call dc_run while another DeepClause execution is active, and do not claim success unless dc_run returns an answer without errors.",
514
+ "If dc_run reports execution_already_active, wait for the active run to finish and then retry this skill instead of reporting failure.",
482
515
  ],
483
516
  parameters: Type.Object({
484
517
  skill: Type.String({ description: "Skill name such as example, or a DML path relative to .pi/deepclause/." }),
@@ -488,41 +521,40 @@ export default function deepClauseExtension(pi) {
488
521
  })),
489
522
  }),
490
523
  async execute(_toolCallId, params, signal, onUpdate, ctx) {
491
- if (activeController) {
524
+ const controller = claimExecution("dc_run");
525
+ if (!controller) {
492
526
  return {
493
- content: [{ type: "text", text: "DeepClause execution rejected: another execution is already active." }],
527
+ content: [{ type: "text", text: "DeepClause execution rejected: another execution is active. Wait for it to finish, then call dc_run again for this skill." }],
494
528
  details: { success: false, error: "execution_already_active" },
495
529
  };
496
530
  }
497
- const paths = await initializeWorkspace(ctx.cwd);
498
- const config = await loadConfig(paths.config);
499
- if (!config.modelToolEnabled || !pi.getActiveTools().includes(DC_RUN_TOOL)) {
500
- return {
501
- content: [{ type: "text", text: "The dc_run tool is disabled. The user can enable it with /dc-tool enable." }],
502
- details: { success: false, error: "tool_disabled" },
503
- };
504
- }
505
- const mode = params.context ?? config.contextMode;
506
- const filePath = await resolveDmlPath(paths, params.skill);
507
- if (await isContextualPlan(filePath)) {
508
- return {
509
- content: [{ type: "text", text: "Contextual DML plans must be started by the user with /dc-run; they cannot start a nested pi agent turn from dc_run." }],
510
- details: { success: false, error: "interactive_plan_requires_user_run" },
511
- };
512
- }
513
- const skillName = path.relative(paths.root, filePath);
514
- const initialMessages = buildInitialMessages(ctx.sessionManager.getBranch(), mode, config.branchMessageLimit);
515
- const controller = new AbortController();
516
531
  const cancel = () => controller.abort(signal?.reason ?? new Error("dc_run cancelled"));
517
532
  if (signal?.aborted)
518
533
  cancel();
519
534
  else
520
535
  signal?.addEventListener("abort", cancel, { once: true });
521
- activeController = controller;
522
- activeDescription = `model tool running ${skillName}`;
523
- const startedAt = Date.now();
524
- const progress = [];
525
536
  try {
537
+ const paths = await initializeWorkspace(ctx.cwd);
538
+ const config = await loadConfig(paths.config);
539
+ if (!config.modelToolEnabled || !pi.getActiveTools().includes(DC_RUN_TOOL)) {
540
+ return {
541
+ content: [{ type: "text", text: "The dc_run tool is disabled. The user can enable it with /dc-tool enable." }],
542
+ details: { success: false, error: "tool_disabled" },
543
+ };
544
+ }
545
+ const mode = params.context ?? config.contextMode;
546
+ const filePath = await resolveDmlPath(paths, params.skill);
547
+ if (await isContextualPlan(filePath)) {
548
+ return {
549
+ content: [{ type: "text", text: "Contextual DML plans must be started by the user with /dc-run; they cannot start a nested pi agent turn from dc_run." }],
550
+ details: { success: false, error: "interactive_plan_requires_user_run" },
551
+ };
552
+ }
553
+ const skillName = path.relative(paths.root, filePath);
554
+ activeDescription = `model tool running ${skillName}`;
555
+ const initialMessages = buildInitialMessages(ctx.sessionManager.getBranch(), mode, config.branchMessageLimit);
556
+ const startedAt = Date.now();
557
+ const progress = [];
526
558
  const result = await executeDml(filePath, params.args ?? [], initialMessages, config, pi, ctx, controller, {
527
559
  onEvent: (event) => {
528
560
  if (event.type === "output" && event.content)
@@ -544,7 +576,7 @@ export default function deepClauseExtension(pi) {
544
576
  onInput: async (prompt, inputSignal) => {
545
577
  if (!ctx.hasUI)
546
578
  throw new Error("dc_run cannot request user input without interactive UI");
547
- const answer = await ctx.ui.input("DeepClause input", prompt, { signal: inputSignal });
579
+ const answer = await requestDeepClauseInput(ctx, skillName, prompt, inputSignal);
548
580
  if (answer === undefined)
549
581
  throw new Error("Input cancelled");
550
582
  return answer;
@@ -571,13 +603,12 @@ export default function deepClauseExtension(pi) {
571
603
  const message = error instanceof Error ? error.message : String(error);
572
604
  return {
573
605
  content: [{ type: "text", text: `DeepClause execution failed: ${message}` }],
574
- details: { success: false, skill: skillName, contextMode: mode, error: message },
606
+ details: { success: false, error: message },
575
607
  };
576
608
  }
577
609
  finally {
578
610
  signal?.removeEventListener("abort", cancel);
579
- activeController = undefined;
580
- activeDescription = undefined;
611
+ releaseExecution(controller);
581
612
  }
582
613
  },
583
614
  });
@@ -596,11 +627,11 @@ export default function deepClauseExtension(pi) {
596
627
  pi.registerTool({
597
628
  name: DC_DIAGRAM_TOOL,
598
629
  label: "Create DeepClause Diagram",
599
- description: "Create a presentation-grade or specification-grade Mermaid diagram from any .dml file, write a self-contained offline viewer under .pi/deepclause/diagrams/, and open it.",
630
+ description: "Create a presentation-grade or specification-grade Mermaid diagram from any .dml file, write a self-contained offline viewer under .pi/deepclause/diagrams/, and open it. The specification grade preserves the core decision logic and rule facts.",
600
631
  promptSnippet: "Create a presentation- or specification-grade diagram from a DML file",
601
632
  promptGuidelines: [
602
633
  "Use dc_diagram whenever the user asks for a diagram, flowchart, or visual of a .dml file; pass the exact path the user named.",
603
- "Choose grade=presentation for slides and overviews and grade=specification for engineering detail; use grade=both only when the user asks for both.",
634
+ "Choose grade=presentation for slides and overviews and grade=specification for engineering detail (decision predicates, thresholds, rule facts); use grade=both only when the user asks for both.",
604
635
  "Do not hand-write Mermaid or run diagram tools yourself; call dc_diagram and report the viewer result.",
605
636
  ],
606
637
  parameters: Type.Object({
@@ -609,52 +640,47 @@ export default function deepClauseExtension(pi) {
609
640
  view: Type.Optional(StringEnum(["flow", "sequence"], { description: "Base layout used to seed the grade; default flow." })),
610
641
  }),
611
642
  async execute(_toolCallId, params, signal, onUpdate, ctx) {
612
- if (activeController) {
643
+ const controller = claimExecution("dc_diagram");
644
+ if (!controller) {
613
645
  return {
614
646
  content: [{ type: "text", text: "Another DeepClause operation is already active; wait for it to finish." }],
615
647
  details: { success: false, error: "execution_already_active" },
616
648
  };
617
649
  }
618
- const requested = resolveGrade(String(params.grade ?? "")) ?? "presentation";
619
- const grades = requested === "both" ? ["presentation", "specification"] : [requested];
620
- const view = params.view === "sequence" ? "sequence" : "flow";
621
- let sourcePath;
622
- try {
623
- sourcePath = await resolveDiagramSource(ctx.cwd, params.dml);
624
- }
625
- catch (error) {
626
- const message = error instanceof Error ? error.message : String(error);
627
- return { content: [{ type: "text", text: `dc_diagram failed: ${message}` }], details: { success: false, error: message } };
628
- }
629
- if (!ctx.model) {
630
- return {
631
- content: [{ type: "text", text: "dc_diagram requires an active pi model. Select one and try again." }],
632
- details: { success: false, error: "no_model" },
633
- };
634
- }
635
- const config = await loadConfig(getPaths(ctx.cwd).config);
636
- const source = await readFile(sourcePath, "utf8");
637
- const display = displayPath(ctx.cwd, sourcePath);
638
- const seed = view === "sequence"
639
- ? renderSequence(display, source)
640
- : renderDml(display, source, { hideOutput: true });
641
- const targets = await collectDiagramTargets(ctx.cwd, [sourcePath]);
642
- const name = diagramNameFor(sourcePath, targets, ctx.cwd);
643
- const { diagrams, vendor } = await ensureDiagramDir(ctx.cwd, viewerVendorAssetPath());
644
- const templateText = await bundledViewerTemplate();
645
- const controller = new AbortController();
646
650
  const cancel = () => controller.abort(signal?.reason ?? new Error("dc_diagram cancelled"));
647
651
  if (signal?.aborted)
648
652
  cancel();
649
653
  else
650
654
  signal?.addEventListener("abort", cancel, { once: true });
651
- activeController = controller;
652
- activeDescription = `diagram ${name} (${grades.join("+")})`;
653
- const run = (command, args, options) => pi.exec(command, args, options);
654
- let chrome;
655
- let chromeResolved = false;
656
655
  try {
656
+ const requested = resolveGrade(String(params.grade ?? "")) ?? "presentation";
657
+ const grades = requested === "both" ? ["presentation", "specification"] : [requested];
658
+ const view = params.view === "sequence" ? "sequence" : "flow";
659
+ const sourcePath = await resolveDiagramSource(ctx.cwd, params.dml);
660
+ if (!ctx.model) {
661
+ return {
662
+ content: [{ type: "text", text: "dc_diagram requires an active pi model. Select one and try again." }],
663
+ details: { success: false, error: "no_model" },
664
+ };
665
+ }
666
+ const config = await loadConfig(getPaths(ctx.cwd).config);
667
+ const source = await readFile(sourcePath, "utf8");
668
+ const display = displayPath(ctx.cwd, sourcePath);
669
+ const targets = await collectDiagramTargets(ctx.cwd, [sourcePath]);
670
+ const name = diagramNameFor(sourcePath, targets, ctx.cwd);
671
+ const { diagrams, vendor } = await ensureDiagramDir(ctx.cwd, viewerVendorAssetPath());
672
+ const templateText = await bundledViewerTemplate();
673
+ activeDescription = `diagram ${name} (${grades.join("+")})`;
674
+ const run = (command, args, options) => pi.exec(command, args, options);
675
+ let chrome;
676
+ let chromeResolved = false;
657
677
  for (const grade of grades) {
678
+ // The specification seed carries the core decision logic and rule
679
+ // facts; the presentation seed stays small so the model can keep
680
+ // it to a general-audience overview.
681
+ const seed = view === "sequence"
682
+ ? renderSequence(display, source)
683
+ : renderDml(display, source, { hideOutput: true, includeLogic: grade === "specification" });
658
684
  const result = await polishDiagram({
659
685
  grade,
660
686
  view,
@@ -700,8 +726,7 @@ export default function deepClauseExtension(pi) {
700
726
  }
701
727
  finally {
702
728
  signal?.removeEventListener("abort", cancel);
703
- activeController = undefined;
704
- activeDescription = undefined;
729
+ releaseExecution(controller);
705
730
  }
706
731
  },
707
732
  });
@@ -768,13 +793,13 @@ export default function deepClauseExtension(pi) {
768
793
  }
769
794
  };
770
795
  const runSpecSkill = async (ctx, skill, args = [], options = {}) => {
771
- const paths = await initializeWorkspace(ctx.cwd);
772
- const config = await loadConfig(paths.config);
773
- const filePath = await resolveDmlPath(paths, skill);
774
- const controller = new AbortController();
775
- activeController = controller;
776
- activeDescription = `running ${skill}`;
796
+ const controller = claimExecution(`running ${skill}`);
797
+ if (!controller)
798
+ throw new Error("Another DeepClause execution is already active");
777
799
  try {
800
+ const paths = await initializeWorkspace(ctx.cwd);
801
+ const config = await loadConfig(paths.config);
802
+ const filePath = await resolveDmlPath(paths, skill);
778
803
  const result = await executeDml(filePath, args, [], config, pi, ctx, controller, {
779
804
  onEvent() { },
780
805
  onInput: async () => { throw new Error("spec skills do not request input"); },
@@ -784,8 +809,7 @@ export default function deepClauseExtension(pi) {
784
809
  return result.answer ?? "(no answer)";
785
810
  }
786
811
  finally {
787
- activeController = undefined;
788
- activeDescription = undefined;
812
+ releaseExecution(controller);
789
813
  }
790
814
  };
791
815
  pi.on("session_start", async (_event, ctx) => {
@@ -1194,7 +1218,8 @@ export default function deepClauseExtension(pi) {
1194
1218
  pi.registerCommand("dc-run", {
1195
1219
  description: "Run a DML skill with pi's active model",
1196
1220
  handler: async (rawArgs, ctx) => {
1197
- if (activeController) {
1221
+ const controller = claimExecution("dc-run");
1222
+ if (!controller) {
1198
1223
  ctx.ui.notify("A DeepClause execution is already active", "warning");
1199
1224
  return;
1200
1225
  }
@@ -1229,8 +1254,6 @@ export default function deepClauseExtension(pi) {
1229
1254
  }
1230
1255
  const mode = parsed.contextMode ?? config.contextMode;
1231
1256
  const initialMessages = buildInitialMessages(ctx.sessionManager.getBranch(), mode, config.branchMessageLimit);
1232
- const controller = new AbortController();
1233
- activeController = controller;
1234
1257
  const outputLines = [];
1235
1258
  const recentEvents = [];
1236
1259
  const events = [];
@@ -1309,7 +1332,7 @@ export default function deepClauseExtension(pi) {
1309
1332
  onEvent,
1310
1333
  onDiagnostic,
1311
1334
  onInput: async (prompt, signal) => {
1312
- const answer = await ctx.ui.input("DeepClause input", prompt, { signal });
1335
+ const answer = await requestDeepClauseInput(ctx, skillName, prompt, signal);
1313
1336
  if (answer === undefined)
1314
1337
  throw new Error("Input cancelled");
1315
1338
  return answer;
@@ -1340,8 +1363,7 @@ export default function deepClauseExtension(pi) {
1340
1363
  ctx.ui.notify(message, activeController?.signal.aborted ? "warning" : "error");
1341
1364
  }
1342
1365
  finally {
1343
- activeController = undefined;
1344
- activeDescription = undefined;
1366
+ releaseExecution(controller);
1345
1367
  ctx.ui.setStatus(STATUS_KEY, undefined);
1346
1368
  ctx.ui.setWidget(WIDGET_KEY, undefined);
1347
1369
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "deepclause-pi",
3
- "version": "0.4.0",
3
+ "version": "0.5.0",
4
4
  "description": "Pi-hosted runtime for DeepClause DML programs",
5
5
  "type": "module",
6
6
  "license": "MIT",
@@ -1,14 +1,16 @@
1
1
  ---
2
2
  name: handbook-dml
3
- description: Convert a long handbook/SOP into one DeepClause DML skill per workflow — an LLM-first subagent (agentic task/N leaves, narrow tool/2 capabilities) that takes a generic natural-language request as input and uses deterministic Prolog only for mechanical checks, with LLM fallback. Also teaches a tool audit, user confirmation, and how to write the repo-root AGENTS.md policy-routing table so pi calls the skills automatically via dc_run. Use when asked to turn a handbook or procedures manual into executable DML, update a handbook-derived skill, or wire the policy router.
3
+ description: Convert a long handbook/SOP into one DeepClause DML skill per workflow — a `task/N` subagent for agentic work, the judgment predicates (`choose`/`rate`/`verify`/`probability`/`holds`/`judge`) for bounded classification and calibrated gates, and deterministic Prolog only for mechanical checks, always with fallbacks. Also teaches a tool audit, user confirmation, and how to write the repo-root AGENTS.md policy-routing table so pi calls the skills automatically via dc_run. Use when asked to turn a handbook or procedures manual into executable DML, update a handbook-derived skill, choose between a classifier, a probability gate, and a full agent task, or wire the policy router.
4
4
  ---
5
5
 
6
6
  # Handbook → DML (procedures)
7
7
 
8
- Turn a long handbook into **one DML skill per workflow/procedure**. Each skill is
9
- an **LLM-first subagent**: agentic `task/N` leaves do the reading, reasoning,
10
- and acting, while Prolog handles only what is genuinely mechanical (arithmetic,
11
- counting, exact equality) or a hard safety invariant.
8
+ Turn a long handbook into **one DML skill per workflow/procedure**. Each skill
9
+ combines three primitives: agentic `task/N` leaves for work that must act or
10
+ reason across turns, bounded judgment predicates (`choose`/`rate`/`verify`/
11
+ `probability`/`holds`) for classification and calibrated gates, and Prolog only
12
+ for what is genuinely mechanical (arithmetic, counting, exact equality) or a
13
+ hard safety invariant.
12
14
 
13
15
  Before writing DML, read `.pi/deepclause/AGENTS.md` and
14
16
  `.pi/deepclause/DML_REFERENCE.md`. They are authoritative for syntax.
@@ -17,19 +19,117 @@ Before writing DML, read `.pi/deepclause/AGENTS.md` and
17
19
 
18
20
  - **Pi authors; DML is the runtime artifact.** Decomposition and authoring happen
19
21
  in a normal pi turn. There is no Markdown→DML compiler.
20
- - **LLM-first.** Rules and flow are `task/N` / `prompt/N` by default. Use Prolog
21
- only where a rule is mechanical or must never be wrong.
22
+ - **Pick the cheapest primitive.** Bounded classifications, ratings, yes/no
23
+ checks, and probabilities use the judgment predicates (`choose/4`, `rate/4`,
24
+ `verify/3`, `probability/3`, `holds/2-3`, `judge/2`). Fresh-context generation
25
+ and critique use `prompt/N`. Multi-turn work with memory, tools, and typed
26
+ outputs uses `task/N`. Use Prolog only where a rule is mechanical or must
27
+ never be wrong. See **Choosing the reasoning primitive** below.
22
28
  - **Input is a generic request.** `agent_main(Request)` takes free text; the
23
29
  first step is an LLM `task/N` that parses it into a typed `object/1` case (or
24
30
  the request is used directly for simple skills).
25
31
  - **Forbidden actions become tool scoping.** "Never send" means *no send tool*.
26
32
  "Read-only inspection" means the inspect phase gets read tools only.
27
- - **Every deterministic rule gets an LLM fallback.** If a deterministic check
28
- fails, branch to a `prompt/N`/`task/N` that reviews, repairs, or asks the user
29
- — do not hard-fail.
33
+ - **Every deterministic rule gets a fallback.** If a deterministic check
34
+ fails, branch to a judgment, a `prompt/N`/`task/N`, or the user — do not
35
+ hard-fail.
30
36
  - **Ask, don't assume.** Confirm scope, the decomposition, and tool choices with
31
37
  the user before authoring.
32
38
 
39
+ ## Choosing the reasoning primitive
40
+
41
+ Before writing flow, ask what the **core question** of each step is. A full
42
+ `task/N` agent loop carries memory, DML tools, and a multi-turn reasoning loop;
43
+ it is the right answer only when the step must act or reason iteratively. A
44
+ bounded question over explicit text should use the judgment predicates instead:
45
+ they are cheaper, the answer is constrained to the labels/levels you supply,
46
+ they never touch DML memory, and they cannot call tools.
47
+
48
+ | Core question | Primitive | Notes |
49
+ | --- | --- | --- |
50
+ | "Which category/team/route?" from a small closed set | `choose/4` (or `judge/2` with `choose`) | The answer is always one of your option atoms |
51
+ | "How severe/frustrated/confident?" (ordered) | `rate/4` | Answer is constrained to your levels |
52
+ | "Is X true?" / "Does it ask for a refund?" | `verify/3`, or `holds/2` when only the boolean matters | Three-valued: `yes` / `no` / `unknown` |
53
+ | "How likely is X?" / "Is it over a threshold?" | `probability/3`, or `holds/3` for a threshold gate | Decision-grade only when the backend reports `calibrated` |
54
+ | Free-form generation, rewrite, or critique of supplied text | `prompt/N` | Fresh context, no tools, no memory |
55
+ | Multi-step job needing tools, memory, or iteration | `task/N` | Typed outputs; scope tools with `with_tools/2` |
56
+
57
+ **Simple classifier.** If the core question reduces to choosing among a fixed
58
+ set of labels, use `choose/4`. Do not spend a `task/N` on it, and do not let the
59
+ model invent a label: the judge is constrained to your options.
60
+
61
+ ```prolog
62
+ route(Request, Team) :-
63
+ choose(Request, "Which team should handle this request?",
64
+ [billing-"Charges and refunds", orders-"Delivery and returns", account-"Login and security"],
65
+ Team).
66
+ ```
67
+
68
+ **Calibrated probability.** If the decision depends on a probability, say so and
69
+ gate it. The default `llm` backend is *uncalibrated*: a number from it is an
70
+ estimate, not a calibrated probability. Wrap the judgment in
71
+ `require_judgment(calibrated, ...)` so the skill cannot silently run on a
72
+ backend that only estimates, and provide a fallback clause.
73
+
74
+ ```prolog
75
+ risk_band(Text, Band) :-
76
+ require_judgment(calibrated,
77
+ probability(Text, "What is the probability this is high risk?", P)),
78
+ ( P >= 0.8 -> Band = high
79
+ ; P >= 0.4 -> Band = medium
80
+ ; Band = low
81
+ ).
82
+
83
+ risk_band(Text, Band) :-
84
+ % Uncalibrated fallback: a coarse classifier, never a fake probability.
85
+ choose(Text, "Is this high, medium, or low risk?", [high, medium, low], Band).
86
+ ```
87
+
88
+ `holds(Text, Question, Threshold)` is the concise form when you only need the
89
+ threshold decision, and `holds(Text, Question)` is the concise form of a yes/no
90
+ check. `with_judgment(jev, Goal)` pins a scope to a specific backend (for
91
+ example a calibrated one). `require_judgment/2` fails **before** the judgment
92
+ runs, which is what makes the fallback clause above reachable; the runtime emits
93
+ a warning naming the missing capability.
94
+
95
+ **Full agent loop.** Use `task/N` (or `prompt/N`) when the step needs to read
96
+ accumulated memory, call a DML tool, run several model turns, or produce
97
+ free-form text. The judgment layer deliberately cannot do those things.
98
+
99
+ **Batch judgments.** When one step needs more than one or two judgments, send a
100
+ single `judge/2` batch rather than several one-offs:
101
+
102
+ ```prolog
103
+ judge(Message, [
104
+ choose("Which team should handle this?", [billing, orders, account]) - Team,
105
+ rate("How frustrated is the customer?", [calm, frustrated, angry]) - Frustration,
106
+ verify("Does the message ask for a refund?") - Refund,
107
+ probability("Is this urgent?") - Urgency
108
+ ]).
109
+ ```
110
+
111
+ ### Judgment rules
112
+
113
+ - `judge/2`, `choose/4`, `rate/4`, `verify/3`, `probability/3`, `holds/2`,
114
+ `holds/3`, `with_judgment/2`, and `require_judgment/2` are runtime special
115
+ predicates. Do not define your own predicates with those names — a common
116
+ collision is a hand-written `verify/3`. Name deterministic helpers
117
+ `verify_task/3`, `check_*`, and so on.
118
+ - `verify` is three-valued. `\+ verify(...)` does **not** mean "the model said
119
+ no"; inspect the returned `yes` / `no` / `unknown`.
120
+ - Keep `State` explicit: pass the exact text or object to judge. The judge sees
121
+ only that state plus the question and never reads DML memory.
122
+ - A `choose`/`rate` answer is constrained to your options/levels, so make them
123
+ exhaustive and mutually exclusive.
124
+ - Never present an `estimated` probability as `calibrated`. Require the
125
+ capability and fall back, or label the output as an estimate.
126
+ - Answers are memoized per run by backend, model, state, and questions, so
127
+ backtracking reuses an answer instead of re-querying the model.
128
+ - The deterministic linter warns when a skill has no `task()` call and suggests
129
+ `task()` for classification. That heuristic predates the judgment layer: a
130
+ skill whose core question is a bounded judgment can legitimately have no
131
+ `task/N`.
132
+
33
133
  ## Workflow
34
134
 
35
135
  1. **Ingest** the handbook to Markdown (PDF→`pdftotext`, DOCX/HTML→`pandoc`).
@@ -121,6 +221,13 @@ tool(user_feedback(Prompt, Response), "Ask the user one focused question and ret
121
221
  exec(ask_user(prompt: Prompt), Result),
122
222
  get_dict(user_response, Result, Response).
123
223
 
224
+ % --- bounded judgments (cheap: no memory, no tools) ------------------------
225
+ % Prefer a judgment over a task/N for a bounded question about explicit text.
226
+ % classify(Request, Kind) :-
227
+ % choose(Request, "Which case type is this?", [<kind_a>, <kind_b>, other], Kind).
228
+ % confirm_requirement(Report) :-
229
+ % holds(Report, "Does the report include the required referral advice?", 0.7).
230
+
124
231
  % --- deterministic helpers (mechanical only) -------------------------------
125
232
  <compute_or_check>(...). % arithmetic/counts; keep small
126
233
 
@@ -136,7 +243,7 @@ agent_main(Request) :-
136
243
  with_tools([<write tools>], (
137
244
  task("Produce the required effects for this case: {ConfirmedCase}.", string(Summary))
138
245
  )),
139
- <optional verification with LLM fallback>,
246
+ <optional judgment / prompt / deterministic verification with fallback>,
140
247
  answer(Final).
141
248
 
142
249
  agent_main(_) :-
@@ -146,49 +253,59 @@ agent_main(_) :-
146
253
  Notes:
147
254
 
148
255
  - `task/N` = agentic leaf (memory + DML tools). `prompt/N` = fresh-context
149
- review. `with_tools/2` scopes capability per phase.
256
+ generation/review. `choose`/`rate`/`verify`/`probability` = the judgment layer.
257
+ `with_tools/2` scopes capability per phase.
258
+ - Prefer a judgment over a `task/N` whenever the core question is a bounded
259
+ classification, rating, yes/no check, or probability gate.
150
260
  - Build `task/N` descriptions with `format/3` (not `{Var}` interpolation) when
151
261
  you embed dynamic values — avoids singleton-variable noise.
152
262
  - Mutable facts must be declared `:- dynamic` before `assertz`/`retract`.
153
263
 
154
- ## Verification (optional, LLM-first)
264
+ ## Verification (optional)
265
+
266
+ Only add checks when the procedure has observable post-conditions. Match the
267
+ check to the question: a bounded semantic check is a judgment, an open-ended
268
+ read is a model review, and a count is deterministic. Every check needs a
269
+ fallback path so it never hard-fails.
155
270
 
156
- Only add checks when the procedure has observable post-conditions. Prefer model
157
- review; use deterministic checks only for mechanical facts, and always give a
158
- deterministic check an LLM fallback.
271
+ - **Semantic gate (default for bounded checks).** "Does the output recommend
272
+ urgent referral when a danger sign is present?" is a `verify/3` question, or
273
+ `holds/2` when you only need the boolean. Use `holds/3` when the check is a
274
+ probability threshold.
275
+ - **Model review (`prompt/N`).** Tone, completeness, correctness of free text,
276
+ "does this read right", and open-ended rubric interpretation.
277
+ - **Deterministic (only when mechanical).** Counts, exact IDs, arithmetic. Keep
278
+ it tiny, and route failures to a judgment, a review, or the user.
159
279
 
160
280
  ```prolog
161
- % deterministic gate (mechanical only)
162
- holds(drafts_count) :- findall(_, draft(_,_,_,_), Ds), length(Ds, 2).
163
-
164
- verify_state(Failed, Report) :-
165
- findall(Name, (postcondition(Name), \+ holds(Name)), Failed),
166
- ( Failed = [] -> Report = "PASS" ; format(string(Report), "FAIL: ~w", [Failed]) ).
167
-
168
- % LLM fallback: never hard-fail on a deterministic check
169
- fallback_review(Failed, Verdict, Reason) :-
170
- format(string(Prompt),
171
- "These structural checks failed: ~w. Review the outcome and the requirement. Store 'acceptable' or 'needs-attention' in Verdict and a one-line reason in Reason.",
172
- [Failed]),
173
- prompt(Prompt, string(Verdict), string(Reason)).
281
+ % Semantic gate: a bounded yes/no question about explicit text.
282
+ referral_ok(Report) :-
283
+ holds(Report, "Does the report recommend urgent referral when a danger sign is present?").
284
+
285
+ % Deterministic gate (mechanical only).
286
+ drafts_complete :- findall(_, draft(_,_,_,_), Ds), length(Ds, 2).
287
+
288
+ verify_state(Report, Note) :-
289
+ ( referral_ok(Report), drafts_complete
290
+ -> Note = "verification passed"
291
+ ; % Fallback: never hard-fail; explain the failure and let pi/user decide.
292
+ format(string(Prompt),
293
+ "Review this outcome against the requirement and explain any gap: ~w. Store 'acceptable' or 'needs-attention' in Verdict and a one-line reason in Reason.",
294
+ [Report]),
295
+ prompt(Prompt, string(Verdict), string(Reason)),
296
+ format(string(Note), "fallback verdict ~w: ~w", [Verdict, Reason])
297
+ ).
174
298
  ```
175
299
 
176
300
  In `agent_main`:
177
301
 
178
302
  ```prolog
179
- verify_state(Failed, Report),
180
- ( Failed = [] -> V = "n/a", R = "deterministic checks passed"
181
- ; fallback_review(Failed, V, R)
182
- ),
183
- answer(... report Report + V/R ...).
303
+ verify_state(Report, Note),
304
+ answer(... report + Note ...).
184
305
  ```
185
306
 
186
- Choosing:
187
-
188
- - **Model review** (default): tone, completeness, correctness of free text,
189
- "does this read right", and anything the rubric phrases as a judgment.
190
- - **Deterministic** (only when mechanical): counts, exact IDs, arithmetic. Keep
191
- it tiny, and route failures to the model or the user instead of failing.
307
+ For a calibrated postcondition, wrap `holds(Report, Question, Threshold)` in
308
+ `require_judgment(calibrated, ...)` and keep the same fallback pattern.
192
309
 
193
310
  ## Dummy tools
194
311
 
@@ -243,10 +360,13 @@ For each generated skill:
243
360
  — runs clean; verification (if any) passes or the fallback explains.
244
361
  2. Inspect phase is read-only, action phase is write-only, and a forbidden
245
362
  action has no tool at all.
246
- 3. Deterministic code is limited to arithmetic/counts; every deterministic
247
- check has an LLM fallback branch.
248
- 4. The tool audit was done and its result is recorded in the header/INDEX.
249
- 5. Re-check the invalid-pattern table in `.pi/deepclause/AGENTS.md` (singleton
363
+ 3. Bounded classifications and checks use the judgment predicates; calibrated
364
+ probabilities are wrapped in `require_judgment(calibrated, ...)` with a
365
+ fallback clause, and `verify/3` results are read as `yes`/`no`/`unknown`.
366
+ 4. Deterministic code is limited to arithmetic/counts; every deterministic
367
+ check has a model/judgment fallback branch.
368
+ 5. The tool audit was done and its result is recorded in the header/INDEX.
369
+ 6. Re-check the invalid-pattern table in `.pi/deepclause/AGENTS.md` (singleton
250
370
  variables, `~` vs `{}` interpolation, `Result.field` vs `get_dict/3`, `->`
251
371
  committing over generators, `answer/1` last, `:- dynamic` before
252
372
  `assertz/retract`).
@@ -254,12 +374,20 @@ For each generated skill:
254
374
  `--context=isolated` keeps session text out of the run. The runtime still needs
255
375
  a model selected.
256
376
 
377
+ To exercise calibrated judgments for real, enable the Jev backend
378
+ (`/dc-judge enable`, `/dc-judge default jev`, or `--judge=jev`) and export the
379
+ backend's API key (default `TYPESAFE_API_KEY`) in the shell that launches pi.
380
+ Without it, the `llm` backend is uncalibrated and the
381
+ `require_judgment(calibrated, ...)` branch is skipped in favor of the fallback.
382
+
257
383
  ## Reference shape
258
384
 
259
385
  A typical procedure skill has: `agent_main(Request)` that parses the request
260
- with `task/N`, confirms with the user via a `user_feedback` tool loop, acts
386
+ with `task/N`, confirms with the user via a `user_feedback` tool loop, uses
387
+ bounded judgments (`choose`/`verify`/`holds`) for classification and gates, acts
261
388
  through write tools, and reviews with `prompt/N` — with small deterministic
262
389
  helpers for arithmetic and a fallback branch instead of hard failures.
263
390
 
264
391
  For DML mechanics, see the bundled example skills in a fresh workspace
265
- (`example.dml`, `deep_research.dml`) and `.pi/deepclause/AGENTS.md`.
392
+ (`example.dml`, `deep_research.dml`), `.pi/deepclause/DML_REFERENCE.md` (the
393
+ judgment predicates are documented there), and `.pi/deepclause/AGENTS.md`.