deepclause-pi 0.3.0 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +4 -4
- package/dist/config.d.ts +22 -0
- package/dist/config.js +60 -0
- package/dist/diagram/extract.d.ts +2 -0
- package/dist/diagram/extract.js +223 -3
- package/dist/diagram/grade.d.ts +1 -1
- package/dist/diagram/grade.js +4 -4
- package/dist/index.d.ts +10 -0
- package/dist/index.js +204 -86
- package/dist/runtime.d.ts +3 -1
- package/dist/runtime.js +23 -1
- package/docs/SPECKIT.md +10 -0
- package/package.json +2 -2
- package/skills/handbook-dml/SKILL.md +174 -46
- package/src/assets/AGENTS.md +43 -7
- package/src/assets/apply.dml +2 -2
- package/src/assets/specs.dml +30 -26
- package/src/config.ts +91 -0
- package/src/diagram/extract.ts +225 -3
- package/src/diagram/grade.ts +4 -4
- package/src/index.ts +224 -97
- package/src/runtime.ts +30 -2
|
@@ -1,14 +1,16 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: handbook-dml
|
|
3
|
-
description: Convert a long handbook/SOP into one DeepClause DML skill per workflow —
|
|
3
|
+
description: Convert a long handbook/SOP into one DeepClause DML skill per workflow — a `task/N` subagent for agentic work, the judgment predicates (`choose`/`rate`/`verify`/`probability`/`holds`/`judge`) for bounded classification and calibrated gates, and deterministic Prolog only for mechanical checks, always with fallbacks. Also teaches a tool audit, user confirmation, and how to write the repo-root AGENTS.md policy-routing table so pi calls the skills automatically via dc_run. Use when asked to turn a handbook or procedures manual into executable DML, update a handbook-derived skill, choose between a classifier, a probability gate, and a full agent task, or wire the policy router.
|
|
4
4
|
---
|
|
5
5
|
|
|
6
6
|
# Handbook → DML (procedures)
|
|
7
7
|
|
|
8
|
-
Turn a long handbook into **one DML skill per workflow/procedure**. Each skill
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
8
|
+
Turn a long handbook into **one DML skill per workflow/procedure**. Each skill
|
|
9
|
+
combines three primitives: agentic `task/N` leaves for work that must act or
|
|
10
|
+
reason across turns, bounded judgment predicates (`choose`/`rate`/`verify`/
|
|
11
|
+
`probability`/`holds`) for classification and calibrated gates, and Prolog only
|
|
12
|
+
for what is genuinely mechanical (arithmetic, counting, exact equality) or a
|
|
13
|
+
hard safety invariant.
|
|
12
14
|
|
|
13
15
|
Before writing DML, read `.pi/deepclause/AGENTS.md` and
|
|
14
16
|
`.pi/deepclause/DML_REFERENCE.md`. They are authoritative for syntax.
|
|
@@ -17,19 +19,117 @@ Before writing DML, read `.pi/deepclause/AGENTS.md` and
|
|
|
17
19
|
|
|
18
20
|
- **Pi authors; DML is the runtime artifact.** Decomposition and authoring happen
|
|
19
21
|
in a normal pi turn. There is no Markdown→DML compiler.
|
|
20
|
-
- **
|
|
21
|
-
|
|
22
|
+
- **Pick the cheapest primitive.** Bounded classifications, ratings, yes/no
|
|
23
|
+
checks, and probabilities use the judgment predicates (`choose/4`, `rate/4`,
|
|
24
|
+
`verify/3`, `probability/3`, `holds/2-3`, `judge/2`). Fresh-context generation
|
|
25
|
+
and critique use `prompt/N`. Multi-turn work with memory, tools, and typed
|
|
26
|
+
outputs uses `task/N`. Use Prolog only where a rule is mechanical or must
|
|
27
|
+
never be wrong. See **Choosing the reasoning primitive** below.
|
|
22
28
|
- **Input is a generic request.** `agent_main(Request)` takes free text; the
|
|
23
29
|
first step is an LLM `task/N` that parses it into a typed `object/1` case (or
|
|
24
30
|
the request is used directly for simple skills).
|
|
25
31
|
- **Forbidden actions become tool scoping.** "Never send" means *no send tool*.
|
|
26
32
|
"Read-only inspection" means the inspect phase gets read tools only.
|
|
27
|
-
- **Every deterministic rule gets
|
|
28
|
-
fails, branch to a `prompt/N`/`task/N
|
|
29
|
-
|
|
33
|
+
- **Every deterministic rule gets a fallback.** If a deterministic check
|
|
34
|
+
fails, branch to a judgment, a `prompt/N`/`task/N`, or the user — do not
|
|
35
|
+
hard-fail.
|
|
30
36
|
- **Ask, don't assume.** Confirm scope, the decomposition, and tool choices with
|
|
31
37
|
the user before authoring.
|
|
32
38
|
|
|
39
|
+
## Choosing the reasoning primitive
|
|
40
|
+
|
|
41
|
+
Before writing flow, ask what the **core question** of each step is. A full
|
|
42
|
+
`task/N` agent loop carries memory, DML tools, and a multi-turn reasoning loop;
|
|
43
|
+
it is the right answer only when the step must act or reason iteratively. A
|
|
44
|
+
bounded question over explicit text should use the judgment predicates instead:
|
|
45
|
+
they are cheaper, the answer is constrained to the labels/levels you supply,
|
|
46
|
+
they never touch DML memory, and they cannot call tools.
|
|
47
|
+
|
|
48
|
+
| Core question | Primitive | Notes |
|
|
49
|
+
| --- | --- | --- |
|
|
50
|
+
| "Which category/team/route?" from a small closed set | `choose/4` (or `judge/2` with `choose`) | The answer is always one of your option atoms |
|
|
51
|
+
| "How severe/frustrated/confident?" (ordered) | `rate/4` | Answer is constrained to your levels |
|
|
52
|
+
| "Is X true?" / "Does it ask for a refund?" | `verify/3`, or `holds/2` when only the boolean matters | Three-valued: `yes` / `no` / `unknown` |
|
|
53
|
+
| "How likely is X?" / "Is it over a threshold?" | `probability/3`, or `holds/3` for a threshold gate | Decision-grade only when the backend reports `calibrated` |
|
|
54
|
+
| Free-form generation, rewrite, or critique of supplied text | `prompt/N` | Fresh context, no tools, no memory |
|
|
55
|
+
| Multi-step job needing tools, memory, or iteration | `task/N` | Typed outputs; scope tools with `with_tools/2` |
|
|
56
|
+
|
|
57
|
+
**Simple classifier.** If the core question reduces to choosing among a fixed
|
|
58
|
+
set of labels, use `choose/4`. Do not spend a `task/N` on it, and do not let the
|
|
59
|
+
model invent a label: the judge is constrained to your options.
|
|
60
|
+
|
|
61
|
+
```prolog
|
|
62
|
+
route(Request, Team) :-
|
|
63
|
+
choose(Request, "Which team should handle this request?",
|
|
64
|
+
[billing-"Charges and refunds", orders-"Delivery and returns", account-"Login and security"],
|
|
65
|
+
Team).
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
**Calibrated probability.** If the decision depends on a probability, say so and
|
|
69
|
+
gate it. The default `llm` backend is *uncalibrated*: a number from it is an
|
|
70
|
+
estimate, not a calibrated probability. Wrap the judgment in
|
|
71
|
+
`require_judgment(calibrated, ...)` so the skill cannot silently run on a
|
|
72
|
+
backend that only estimates, and provide a fallback clause.
|
|
73
|
+
|
|
74
|
+
```prolog
|
|
75
|
+
risk_band(Text, Band) :-
|
|
76
|
+
require_judgment(calibrated,
|
|
77
|
+
probability(Text, "What is the probability this is high risk?", P)),
|
|
78
|
+
( P >= 0.8 -> Band = high
|
|
79
|
+
; P >= 0.4 -> Band = medium
|
|
80
|
+
; Band = low
|
|
81
|
+
).
|
|
82
|
+
|
|
83
|
+
risk_band(Text, Band) :-
|
|
84
|
+
% Uncalibrated fallback: a coarse classifier, never a fake probability.
|
|
85
|
+
choose(Text, "Is this high, medium, or low risk?", [high, medium, low], Band).
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
`holds(Text, Question, Threshold)` is the concise form when you only need the
|
|
89
|
+
threshold decision, and `holds(Text, Question)` is the concise form of a yes/no
|
|
90
|
+
check. `with_judgment(jev, Goal)` pins a scope to a specific backend (for
|
|
91
|
+
example a calibrated one). `require_judgment/2` fails **before** the judgment
|
|
92
|
+
runs, which is what makes the fallback clause above reachable; the runtime emits
|
|
93
|
+
a warning naming the missing capability.
|
|
94
|
+
|
|
95
|
+
**Full agent loop.** Use `task/N` (or `prompt/N`) when the step needs to read
|
|
96
|
+
accumulated memory, call a DML tool, run several model turns, or produce
|
|
97
|
+
free-form text. The judgment layer deliberately cannot do those things.
|
|
98
|
+
|
|
99
|
+
**Batch judgments.** When one step needs more than one or two judgments, send a
|
|
100
|
+
single `judge/2` batch rather than several one-offs:
|
|
101
|
+
|
|
102
|
+
```prolog
|
|
103
|
+
judge(Message, [
|
|
104
|
+
choose("Which team should handle this?", [billing, orders, account]) - Team,
|
|
105
|
+
rate("How frustrated is the customer?", [calm, frustrated, angry]) - Frustration,
|
|
106
|
+
verify("Does the message ask for a refund?") - Refund,
|
|
107
|
+
probability("Is this urgent?") - Urgency
|
|
108
|
+
]).
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
### Judgment rules
|
|
112
|
+
|
|
113
|
+
- `judge/2`, `choose/4`, `rate/4`, `verify/3`, `probability/3`, `holds/2`,
|
|
114
|
+
`holds/3`, `with_judgment/2`, and `require_judgment/2` are runtime special
|
|
115
|
+
predicates. Do not define your own predicates with those names — a common
|
|
116
|
+
collision is a hand-written `verify/3`. Name deterministic helpers
|
|
117
|
+
`verify_task/3`, `check_*`, and so on.
|
|
118
|
+
- `verify` is three-valued. `\+ verify(...)` does **not** mean "the model said
|
|
119
|
+
no"; inspect the returned `yes` / `no` / `unknown`.
|
|
120
|
+
- Keep `State` explicit: pass the exact text or object to judge. The judge sees
|
|
121
|
+
only that state plus the question and never reads DML memory.
|
|
122
|
+
- A `choose`/`rate` answer is constrained to your options/levels, so make them
|
|
123
|
+
exhaustive and mutually exclusive.
|
|
124
|
+
- Never present an `estimated` probability as `calibrated`. Require the
|
|
125
|
+
capability and fall back, or label the output as an estimate.
|
|
126
|
+
- Answers are memoized per run by backend, model, state, and questions, so
|
|
127
|
+
backtracking reuses an answer instead of re-querying the model.
|
|
128
|
+
- The deterministic linter warns when a skill has no `task()` call and suggests
|
|
129
|
+
`task()` for classification. That heuristic predates the judgment layer: a
|
|
130
|
+
skill whose core question is a bounded judgment can legitimately have no
|
|
131
|
+
`task/N`.
|
|
132
|
+
|
|
33
133
|
## Workflow
|
|
34
134
|
|
|
35
135
|
1. **Ingest** the handbook to Markdown (PDF→`pdftotext`, DOCX/HTML→`pandoc`).
|
|
@@ -121,6 +221,13 @@ tool(user_feedback(Prompt, Response), "Ask the user one focused question and ret
|
|
|
121
221
|
exec(ask_user(prompt: Prompt), Result),
|
|
122
222
|
get_dict(user_response, Result, Response).
|
|
123
223
|
|
|
224
|
+
% --- bounded judgments (cheap: no memory, no tools) ------------------------
|
|
225
|
+
% Prefer a judgment over a task/N for a bounded question about explicit text.
|
|
226
|
+
% classify(Request, Kind) :-
|
|
227
|
+
% choose(Request, "Which case type is this?", [<kind_a>, <kind_b>, other], Kind).
|
|
228
|
+
% confirm_requirement(Report) :-
|
|
229
|
+
% holds(Report, "Does the report include the required referral advice?", 0.7).
|
|
230
|
+
|
|
124
231
|
% --- deterministic helpers (mechanical only) -------------------------------
|
|
125
232
|
<compute_or_check>(...). % arithmetic/counts; keep small
|
|
126
233
|
|
|
@@ -136,7 +243,7 @@ agent_main(Request) :-
|
|
|
136
243
|
with_tools([<write tools>], (
|
|
137
244
|
task("Produce the required effects for this case: {ConfirmedCase}.", string(Summary))
|
|
138
245
|
)),
|
|
139
|
-
<optional verification with
|
|
246
|
+
<optional judgment / prompt / deterministic verification with fallback>,
|
|
140
247
|
answer(Final).
|
|
141
248
|
|
|
142
249
|
agent_main(_) :-
|
|
@@ -146,49 +253,59 @@ agent_main(_) :-
|
|
|
146
253
|
Notes:
|
|
147
254
|
|
|
148
255
|
- `task/N` = agentic leaf (memory + DML tools). `prompt/N` = fresh-context
|
|
149
|
-
review. `
|
|
256
|
+
generation/review. `choose`/`rate`/`verify`/`probability` = the judgment layer.
|
|
257
|
+
`with_tools/2` scopes capability per phase.
|
|
258
|
+
- Prefer a judgment over a `task/N` whenever the core question is a bounded
|
|
259
|
+
classification, rating, yes/no check, or probability gate.
|
|
150
260
|
- Build `task/N` descriptions with `format/3` (not `{Var}` interpolation) when
|
|
151
261
|
you embed dynamic values — avoids singleton-variable noise.
|
|
152
262
|
- Mutable facts must be declared `:- dynamic` before `assertz`/`retract`.
|
|
153
263
|
|
|
154
|
-
## Verification (optional
|
|
264
|
+
## Verification (optional)
|
|
265
|
+
|
|
266
|
+
Only add checks when the procedure has observable post-conditions. Match the
|
|
267
|
+
check to the question: a bounded semantic check is a judgment, an open-ended
|
|
268
|
+
read is a model review, and a count is deterministic. Every check needs a
|
|
269
|
+
fallback path so it never hard-fails.
|
|
155
270
|
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
271
|
+
- **Semantic gate (default for bounded checks).** "Does the output recommend
|
|
272
|
+
urgent referral when a danger sign is present?" is a `verify/3` question, or
|
|
273
|
+
`holds/2` when you only need the boolean. Use `holds/3` when the check is a
|
|
274
|
+
probability threshold.
|
|
275
|
+
- **Model review (`prompt/N`).** Tone, completeness, correctness of free text,
|
|
276
|
+
"does this read right", and open-ended rubric interpretation.
|
|
277
|
+
- **Deterministic (only when mechanical).** Counts, exact IDs, arithmetic. Keep
|
|
278
|
+
it tiny, and route failures to a judgment, a review, or the user.
|
|
159
279
|
|
|
160
280
|
```prolog
|
|
161
|
-
%
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
281
|
+
% Semantic gate: a bounded yes/no question about explicit text.
|
|
282
|
+
referral_ok(Report) :-
|
|
283
|
+
holds(Report, "Does the report recommend urgent referral when a danger sign is present?").
|
|
284
|
+
|
|
285
|
+
% Deterministic gate (mechanical only).
|
|
286
|
+
drafts_complete :- findall(_, draft(_,_,_,_), Ds), length(Ds, 2).
|
|
287
|
+
|
|
288
|
+
verify_state(Report, Note) :-
|
|
289
|
+
( referral_ok(Report), drafts_complete
|
|
290
|
+
-> Note = "verification passed"
|
|
291
|
+
; % Fallback: never hard-fail; explain the failure and let pi/user decide.
|
|
292
|
+
format(string(Prompt),
|
|
293
|
+
"Review this outcome against the requirement and explain any gap: ~w. Store 'acceptable' or 'needs-attention' in Verdict and a one-line reason in Reason.",
|
|
294
|
+
[Report]),
|
|
295
|
+
prompt(Prompt, string(Verdict), string(Reason)),
|
|
296
|
+
format(string(Note), "fallback verdict ~w: ~w", [Verdict, Reason])
|
|
297
|
+
).
|
|
174
298
|
```
|
|
175
299
|
|
|
176
300
|
In `agent_main`:
|
|
177
301
|
|
|
178
302
|
```prolog
|
|
179
|
-
verify_state(
|
|
180
|
-
(
|
|
181
|
-
; fallback_review(Failed, V, R)
|
|
182
|
-
),
|
|
183
|
-
answer(... report Report + V/R ...).
|
|
303
|
+
verify_state(Report, Note),
|
|
304
|
+
answer(... report + Note ...).
|
|
184
305
|
```
|
|
185
306
|
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
- **Model review** (default): tone, completeness, correctness of free text,
|
|
189
|
-
"does this read right", and anything the rubric phrases as a judgment.
|
|
190
|
-
- **Deterministic** (only when mechanical): counts, exact IDs, arithmetic. Keep
|
|
191
|
-
it tiny, and route failures to the model or the user instead of failing.
|
|
307
|
+
For a calibrated postcondition, wrap `holds(Report, Question, Threshold)` in
|
|
308
|
+
`require_judgment(calibrated, ...)` and keep the same fallback pattern.
|
|
192
309
|
|
|
193
310
|
## Dummy tools
|
|
194
311
|
|
|
@@ -243,10 +360,13 @@ For each generated skill:
|
|
|
243
360
|
— runs clean; verification (if any) passes or the fallback explains.
|
|
244
361
|
2. Inspect phase is read-only, action phase is write-only, and a forbidden
|
|
245
362
|
action has no tool at all.
|
|
246
|
-
3.
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
363
|
+
3. Bounded classifications and checks use the judgment predicates; calibrated
|
|
364
|
+
probabilities are wrapped in `require_judgment(calibrated, ...)` with a
|
|
365
|
+
fallback clause, and `verify/3` results are read as `yes`/`no`/`unknown`.
|
|
366
|
+
4. Deterministic code is limited to arithmetic/counts; every deterministic
|
|
367
|
+
check has a model/judgment fallback branch.
|
|
368
|
+
5. The tool audit was done and its result is recorded in the header/INDEX.
|
|
369
|
+
6. Re-check the invalid-pattern table in `.pi/deepclause/AGENTS.md` (singleton
|
|
250
370
|
variables, `~` vs `{}` interpolation, `Result.field` vs `get_dict/3`, `->`
|
|
251
371
|
committing over generators, `answer/1` last, `:- dynamic` before
|
|
252
372
|
`assertz/retract`).
|
|
@@ -254,12 +374,20 @@ For each generated skill:
|
|
|
254
374
|
`--context=isolated` keeps session text out of the run. The runtime still needs
|
|
255
375
|
a model selected.
|
|
256
376
|
|
|
377
|
+
To exercise calibrated judgments for real, enable the Jev backend
|
|
378
|
+
(`/dc-judge enable`, `/dc-judge default jev`, or `--judge=jev`) and export the
|
|
379
|
+
backend's API key (default `TYPESAFE_API_KEY`) in the shell that launches pi.
|
|
380
|
+
Without it, the `llm` backend is uncalibrated and the
|
|
381
|
+
`require_judgment(calibrated, ...)` branch is skipped in favor of the fallback.
|
|
382
|
+
|
|
257
383
|
## Reference shape
|
|
258
384
|
|
|
259
385
|
A typical procedure skill has: `agent_main(Request)` that parses the request
|
|
260
|
-
with `task/N`, confirms with the user via a `user_feedback` tool loop,
|
|
386
|
+
with `task/N`, confirms with the user via a `user_feedback` tool loop, uses
|
|
387
|
+
bounded judgments (`choose`/`verify`/`holds`) for classification and gates, acts
|
|
261
388
|
through write tools, and reviews with `prompt/N` — with small deterministic
|
|
262
389
|
helpers for arithmetic and a fallback branch instead of hard failures.
|
|
263
390
|
|
|
264
391
|
For DML mechanics, see the bundled example skills in a fresh workspace
|
|
265
|
-
(`example.dml`, `deep_research.dml`)
|
|
392
|
+
(`example.dml`, `deep_research.dml`), `.pi/deepclause/DML_REFERENCE.md` (the
|
|
393
|
+
judgment predicates are documented there), and `.pi/deepclause/AGENTS.md`.
|
package/src/assets/AGENTS.md
CHANGED
|
@@ -111,15 +111,51 @@ A task can bind up to four outputs. Keep each task focused even though the runti
|
|
|
111
111
|
|
|
112
112
|
### `prompt/N`: an isolated model call
|
|
113
113
|
|
|
114
|
-
Use `prompt/N` for a subtask that should not inherit accumulated conversation memory:
|
|
114
|
+
Use `prompt/N` for a subtask that should not inherit accumulated conversation memory: adversarial review, rewriting, or formatting based only on explicitly supplied text. For a bounded classification, prefer the judgment predicates below.
|
|
115
115
|
|
|
116
116
|
```prolog
|
|
117
|
-
prompt("
|
|
118
|
-
string(
|
|
117
|
+
prompt("Rewrite this summary in plain language for a patient: {Text}. Store only the rewrite in Plain.",
|
|
118
|
+
string(Plain)).
|
|
119
119
|
```
|
|
120
120
|
|
|
121
121
|
Fresh context is not a security boundary. Untrusted text can still contain hostile instructions; delimit it, state how it may be used, and request narrow structured output.
|
|
122
122
|
|
|
123
|
+
### Semantic judgments: bounded, typed questions
|
|
124
|
+
|
|
125
|
+
Use the judgment predicates when the core question is a **bounded, typed question about explicit state**. They are not agentic: they do not read or write DML memory, they cannot call tools, and their answers are constrained to the options or levels you supply. That makes them cheaper and more reliable than a `task/N` for classification, rating, verification, and probability gates.
|
|
126
|
+
|
|
127
|
+
| Core question | Predicate |
|
|
128
|
+
| --- | --- |
|
|
129
|
+
| "Which label/route?" from a closed set | `choose(State, Question, Options, Choice)` |
|
|
130
|
+
| "How severe/frustrated/confident?" on an ordered scale | `rate(State, Question, Levels, Level)` |
|
|
131
|
+
| "Is X true?" (`yes`/`no`/`unknown`) | `verify(State, Question, Truth)`; `holds(State, Question)` for a semidet check |
|
|
132
|
+
| "How likely is X?" / "Is it above a threshold?" | `probability(State, Question, P)`; `holds(State, Question, Threshold)` |
|
|
133
|
+
|
|
134
|
+
Batch several questions into one request with `judge/2`:
|
|
135
|
+
|
|
136
|
+
```prolog
|
|
137
|
+
judge(Message, [
|
|
138
|
+
choose("Which team should handle this?", [billing, orders, account]) - Team,
|
|
139
|
+
verify("Does the message ask for a refund?") - Refund
|
|
140
|
+
]).
|
|
141
|
+
```
|
|
142
|
+
|
|
143
|
+
Gate a calibrated probability and fall back when the backend only estimates:
|
|
144
|
+
|
|
145
|
+
```prolog
|
|
146
|
+
risk_band(Text, Band) :-
|
|
147
|
+
require_judgment(calibrated, probability(Text, "Risk of harm?", P)),
|
|
148
|
+
( P >= 0.8 -> Band = high ; Band = low ).
|
|
149
|
+
risk_band(Text, Band) :-
|
|
150
|
+
choose(Text, "Is the risk high or low?", [high, low], Band).
|
|
151
|
+
```
|
|
152
|
+
|
|
153
|
+
`with_judgment(Backend, Goal)` selects a backend per scope; `require_judgment/2` fails before the judgment runs when the backend lacks a capability such as `calibrated`, so the second clause is the fallback. Answers are memoized per run by backend, model, state, and questions, so backtracking is cheap.
|
|
154
|
+
|
|
155
|
+
Do not name your own predicates `choose/4`, `rate/4`, `verify/3`, `probability/3`, `holds/2`, `holds/3`, `judge/2`, `with_judgment/2`, or `require_judgment/2`; they are runtime special predicates. `verify` is three-valued — never treat `\+ verify(...)` as "the model said no".
|
|
156
|
+
|
|
157
|
+
See `DML_REFERENCE.md` for the full syntax, options, and capability list.
|
|
158
|
+
|
|
123
159
|
### `llm/2`: low-level completion
|
|
124
160
|
|
|
125
161
|
Prefer `task/N` and `prompt/N`. Use `get_memory/1` plus `llm/2` only when the program intentionally needs a raw completion over an explicit message list and does not need the task loop's result tools or DML tools.
|
|
@@ -357,7 +393,7 @@ Represent stable facts and rules as Prolog clauses; expose narrow query/update t
|
|
|
357
393
|
|
|
358
394
|
### 6. Independent reviewers
|
|
359
395
|
|
|
360
|
-
Use `task/N` to draft and `prompt/N` to review from fresh context, then apply deterministic acceptance criteria. Best for code review, risk assessment, and editorial checks.
|
|
396
|
+
Use `task/N` to draft and `prompt/N` to review from fresh context, then apply deterministic acceptance criteria. Best for code review, risk assessment, and editorial checks. When the review is a bounded checklist ("does it state the referral threshold?"), use `verify/3` or `holds/2` instead of a free-form reviewer.
|
|
361
397
|
|
|
362
398
|
### 7. Pure deterministic utility
|
|
363
399
|
|
|
@@ -388,10 +424,10 @@ Pi can turn any DML file into a self-contained, offline Mermaid viewer. Ask for
|
|
|
388
424
|
|
|
389
425
|
Pi calls the `dc_diagram` model tool with the DML path and a grade:
|
|
390
426
|
|
|
391
|
-
- **presentation** — about 8-12 nodes, plain language, headline numbers (slides and overviews).
|
|
392
|
-
- **specification** — function names, task/tool roles, post-conditions (engineers).
|
|
427
|
+
- **presentation** — about 8-12 nodes, plain language, headline numbers and the headline decision (slides and overviews).
|
|
428
|
+
- **specification** — function names, task/tool roles, post-conditions, plus the core decision logic: each decision predicate with its conditions, thresholds and outcomes, and the rule fact tables (engineers).
|
|
393
429
|
|
|
394
|
-
The tool extracts a deterministic Mermaid seed, has pi rewrite it in the chosen grade, validates the result, writes the viewer under `.pi/deepclause/diagrams/`, and opens it. The DML file may live anywhere (workspace-relative or absolute); only the generated viewer stays under `.pi/deepclause/`.
|
|
430
|
+
The tool extracts a deterministic Mermaid seed, enriches the specification seed with a `LOGIC` section (decision predicates, guards, thresholds, judgments) and a `RULES` section (fact tables), has pi rewrite it in the chosen grade, validates the result, writes the viewer under `.pi/deepclause/diagrams/`, and opens it. The DML file may live anywhere (workspace-relative or absolute); only the generated viewer stays under `.pi/deepclause/`.
|
|
395
431
|
|
|
396
432
|
Do not hand-write Mermaid for the user, and do not copy diagram tooling into the workspace. Regenerating a grade replaces only that grade's sidecar (`<name>.presentation.mmd` / `<name>.specification.mmd`).
|
|
397
433
|
|
package/src/assets/apply.dml
CHANGED
|
@@ -85,7 +85,7 @@ attempt(TasksPath, Statuses0, Change, Id, Props, N, Max, Feedback, Statuses) :-
|
|
|
85
85
|
format(string(Progress), "task ~w attempt ~w/~w", [Id, N, Max]),
|
|
86
86
|
output(Progress),
|
|
87
87
|
execute(Change, Id, Props, Feedback, Summary),
|
|
88
|
-
|
|
88
|
+
verify_task(Props, Summary, Verdict),
|
|
89
89
|
( Verdict = ok
|
|
90
90
|
-> sp_set_status(Statuses0, Id, done(N), Statuses),
|
|
91
91
|
sp_write_statuses(TasksPath, Statuses),
|
|
@@ -123,7 +123,7 @@ build_instruction(Id, Do, Expected, Feedback, Instruction) :-
|
|
|
123
123
|
"Task ~w. ~w~nExpected result: ~w~n~nThe previous attempt failed verification: ~w~nFix only what is needed; do not redo the whole task.",
|
|
124
124
|
[Id, Do, Expected, Feedback]).
|
|
125
125
|
|
|
126
|
-
|
|
126
|
+
verify_task(Props, Summary, Verdict) :-
|
|
127
127
|
catch(get_dict(checks, Props, Checks), _, Checks = []),
|
|
128
128
|
findall(Message, (member(Check, Checks), run_check(Check, Message), Message \= ok), Failures),
|
|
129
129
|
( Summary == ""
|
package/src/assets/specs.dml
CHANGED
|
@@ -64,9 +64,12 @@ sp_sections(Lines, Sections) :-
|
|
|
64
64
|
sp_skip_plain(Lines, Rest),
|
|
65
65
|
sp_sections_from(Rest, Sections).
|
|
66
66
|
|
|
67
|
-
sp_skip_plain([line(Line, Number)|Rest], [line(Line, Number)|Rest]) :- sp_heading(Line, Level, _), Level >= 1, Level =< 3, !.
|
|
68
|
-
sp_skip_plain([_|Rest], Out) :- sp_skip_plain(Rest, Out).
|
|
69
67
|
sp_skip_plain([], []).
|
|
68
|
+
sp_skip_plain([Item|Rest], Out) :-
|
|
69
|
+
( Item = line(Line, _), sp_heading(Line, Level, _), Level >= 1, Level =< 3
|
|
70
|
+
-> Out = [Item|Rest]
|
|
71
|
+
; sp_skip_plain(Rest, Out)
|
|
72
|
+
).
|
|
70
73
|
|
|
71
74
|
sp_sections_from([], []).
|
|
72
75
|
sp_sections_from([line(Line, N)|Rest], [section(Level, Text, N, Body)|More]) :-
|
|
@@ -77,12 +80,11 @@ sp_sections_from([line(Line, N)|Rest], [section(Level, Text, N, Body)|More]) :-
|
|
|
77
80
|
sp_sections_from(Tail, More).
|
|
78
81
|
|
|
79
82
|
sp_take_body([], [], []).
|
|
80
|
-
sp_take_body([
|
|
81
|
-
sp_heading(Line, Level, _),
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
sp_take_body([Item|Rest], [Item|Body], Tail) :- sp_take_body(Rest, Body, Tail).
|
|
83
|
+
sp_take_body([Item|Rest], Body, Tail) :-
|
|
84
|
+
( Item = line(Line, _), sp_heading(Line, Level, _), Level >= 1, Level =< 3
|
|
85
|
+
-> Body = [], Tail = [Item|Rest]
|
|
86
|
+
; Body = [Item|Body1], sp_take_body(Rest, Body1, Tail)
|
|
87
|
+
).
|
|
86
88
|
|
|
87
89
|
%% Split a requirement body (level-4 headings inline) into description + scenarios.
|
|
88
90
|
sp_split_scenarios(Body, DescLines, ScenarioSections) :-
|
|
@@ -90,11 +92,11 @@ sp_split_scenarios(Body, DescLines, ScenarioSections) :-
|
|
|
90
92
|
sp_scenario_sections(Rest, ScenarioSections).
|
|
91
93
|
|
|
92
94
|
sp_take_plain([], [], []).
|
|
93
|
-
sp_take_plain([
|
|
94
|
-
sp_heading(Line, Level, _),
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
95
|
+
sp_take_plain([Item|Rest], Body, Tail) :-
|
|
96
|
+
( Item = line(Line, _), sp_heading(Line, Level, _), Level >= 4
|
|
97
|
+
-> Body = [], Tail = [Item|Rest]
|
|
98
|
+
; Body = [Item|Body1], sp_take_plain(Rest, Body1, Tail)
|
|
99
|
+
).
|
|
98
100
|
|
|
99
101
|
sp_scenario_sections([], []).
|
|
100
102
|
sp_scenario_sections([line(Line, N)|Rest], [section(4, Text, N, Body)|More]) :-
|
|
@@ -103,11 +105,11 @@ sp_scenario_sections([line(Line, N)|Rest], [section(4, Text, N, Body)|More]) :-
|
|
|
103
105
|
sp_scenario_sections(Tail, More).
|
|
104
106
|
|
|
105
107
|
sp_take_until_level4([], [], []).
|
|
106
|
-
sp_take_until_level4([
|
|
107
|
-
sp_heading(Line, Level, _),
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
108
|
+
sp_take_until_level4([Item|Rest], Body, Tail) :-
|
|
109
|
+
( Item = line(Line, _), sp_heading(Line, Level, _), Level >= 4
|
|
110
|
+
-> Body = [], Tail = [Item|Rest]
|
|
111
|
+
; Body = [Item|Body1], sp_take_until_level4(Rest, Body1, Tail)
|
|
112
|
+
).
|
|
111
113
|
|
|
112
114
|
%% ---------------------------------------------------------------------------
|
|
113
115
|
%% Term extraction
|
|
@@ -164,10 +166,11 @@ sp_parse_delta(Text, Deltas) :-
|
|
|
164
166
|
sp_collect_ops(Rest, Deltas).
|
|
165
167
|
|
|
166
168
|
sp_drop_to_op([], []).
|
|
167
|
-
sp_drop_to_op([
|
|
168
|
-
sp_delta_header(Header, _)
|
|
169
|
-
|
|
170
|
-
sp_drop_to_op(
|
|
169
|
+
sp_drop_to_op([Item|Rest], Out) :-
|
|
170
|
+
( Item = section(2, Header, _, _), sp_delta_header(Header, _)
|
|
171
|
+
-> Out = [Item|Rest]
|
|
172
|
+
; sp_drop_to_op(Rest, Out)
|
|
173
|
+
).
|
|
171
174
|
|
|
172
175
|
sp_collect_ops([], []).
|
|
173
176
|
sp_collect_ops([section(2, Header, _, _)|Rest], Deltas) :-
|
|
@@ -184,10 +187,11 @@ sp_collect_ops([section(2, Header, _, _)|Rest], Deltas) :-
|
|
|
184
187
|
).
|
|
185
188
|
|
|
186
189
|
sp_take_until_op([], [], []).
|
|
187
|
-
sp_take_until_op([
|
|
188
|
-
sp_delta_header(Header, _)
|
|
189
|
-
|
|
190
|
-
|
|
190
|
+
sp_take_until_op([Item|Rest], Group, Tail) :-
|
|
191
|
+
( Item = section(2, Header, _, _), sp_delta_header(Header, _)
|
|
192
|
+
-> Group = [], Tail = [Item|Rest]
|
|
193
|
+
; Group = [Item|Group1], sp_take_until_op(Rest, Group1, Tail)
|
|
194
|
+
).
|
|
191
195
|
|
|
192
196
|
%% ---------------------------------------------------------------------------
|
|
193
197
|
%% Validation
|
package/src/config.ts
CHANGED
|
@@ -2,6 +2,21 @@ import { readFile, writeFile } from "node:fs/promises";
|
|
|
2
2
|
|
|
3
3
|
export type ContextMode = "turn" | "branch" | "isolated";
|
|
4
4
|
|
|
5
|
+
export interface JevJudgeConfig {
|
|
6
|
+
/** Whether the Jev (TypeSafe System One) backend may be used. */
|
|
7
|
+
enabled: boolean;
|
|
8
|
+
/** Model alias or pinned version passed to TypeSafe. */
|
|
9
|
+
model: string;
|
|
10
|
+
/** Environment variable holding the TypeSafe API key. */
|
|
11
|
+
apiKeyEnv: string;
|
|
12
|
+
}
|
|
13
|
+
|
|
14
|
+
export interface JudgmentConfig {
|
|
15
|
+
/** Default backend name: "llm", "jev", or a name registered by an extension. */
|
|
16
|
+
default: string;
|
|
17
|
+
jev: JevJudgeConfig;
|
|
18
|
+
}
|
|
19
|
+
|
|
5
20
|
export interface DeepClauseConfig {
|
|
6
21
|
version: 1;
|
|
7
22
|
contextMode: ContextMode;
|
|
@@ -10,6 +25,7 @@ export interface DeepClauseConfig {
|
|
|
10
25
|
maxTokens: number;
|
|
11
26
|
verbose: boolean;
|
|
12
27
|
modelToolEnabled: boolean;
|
|
28
|
+
judgment: JudgmentConfig;
|
|
13
29
|
}
|
|
14
30
|
|
|
15
31
|
export const DEFAULT_CONFIG: DeepClauseConfig = {
|
|
@@ -20,6 +36,10 @@ export const DEFAULT_CONFIG: DeepClauseConfig = {
|
|
|
20
36
|
maxTokens: 16_384,
|
|
21
37
|
verbose: false,
|
|
22
38
|
modelToolEnabled: false,
|
|
39
|
+
judgment: {
|
|
40
|
+
default: "llm",
|
|
41
|
+
jev: { enabled: false, model: "jev-latest", apiKeyEnv: "TYPESAFE_API_KEY" },
|
|
42
|
+
},
|
|
23
43
|
};
|
|
24
44
|
|
|
25
45
|
const isContextMode = (value: unknown): value is ContextMode =>
|
|
@@ -55,6 +75,42 @@ export async function loadConfig(path: string): Promise<DeepClauseConfig> {
|
|
|
55
75
|
maxTokens: positiveInteger("maxTokens", DEFAULT_CONFIG.maxTokens),
|
|
56
76
|
verbose: config.verbose === true,
|
|
57
77
|
modelToolEnabled: config.modelToolEnabled === true,
|
|
78
|
+
judgment: parseJudgmentConfig(config.judgment),
|
|
79
|
+
};
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
function parseJudgmentConfig(value: unknown): JudgmentConfig {
|
|
83
|
+
if (value === undefined) return DEFAULT_CONFIG.judgment;
|
|
84
|
+
if (!value || typeof value !== "object" || Array.isArray(value)) {
|
|
85
|
+
throw new Error("judgment must be a JSON object");
|
|
86
|
+
}
|
|
87
|
+
const judgment = value as Record<string, unknown>;
|
|
88
|
+
const defaultBackend = judgment.default ?? DEFAULT_CONFIG.judgment.default;
|
|
89
|
+
if (typeof defaultBackend !== "string" || !defaultBackend.trim()) {
|
|
90
|
+
throw new Error("judgment.default must be a non-empty backend name");
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
const jevValue = judgment.jev ?? {};
|
|
94
|
+
if (!jevValue || typeof jevValue !== "object" || Array.isArray(jevValue)) {
|
|
95
|
+
throw new Error("judgment.jev must be a JSON object");
|
|
96
|
+
}
|
|
97
|
+
const jev = jevValue as Record<string, unknown>;
|
|
98
|
+
const model = jev.model ?? DEFAULT_CONFIG.judgment.jev.model;
|
|
99
|
+
const apiKeyEnv = jev.apiKeyEnv ?? DEFAULT_CONFIG.judgment.jev.apiKeyEnv;
|
|
100
|
+
if (typeof model !== "string" || !model.trim()) {
|
|
101
|
+
throw new Error("judgment.jev.model must be a non-empty string");
|
|
102
|
+
}
|
|
103
|
+
if (typeof apiKeyEnv !== "string" || !apiKeyEnv.trim()) {
|
|
104
|
+
throw new Error("judgment.jev.apiKeyEnv must be a non-empty environment variable name");
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
return {
|
|
108
|
+
default: defaultBackend.trim(),
|
|
109
|
+
jev: {
|
|
110
|
+
enabled: jev.enabled === true,
|
|
111
|
+
model: model.trim(),
|
|
112
|
+
apiKeyEnv: apiKeyEnv.trim(),
|
|
113
|
+
},
|
|
58
114
|
};
|
|
59
115
|
}
|
|
60
116
|
|
|
@@ -72,3 +128,38 @@ export async function setModelToolEnabled(configPath: string, enabled: boolean):
|
|
|
72
128
|
await writeFile(configPath, `${JSON.stringify({ ...existing, modelToolEnabled: enabled }, null, 2)}\n`, "utf8");
|
|
73
129
|
return loadConfig(configPath);
|
|
74
130
|
}
|
|
131
|
+
|
|
132
|
+
export interface JudgeConfigPatch {
|
|
133
|
+
/** Default backend name ("llm" or "jev"). */
|
|
134
|
+
default?: string;
|
|
135
|
+
/** Partial update of the Jev backend settings. */
|
|
136
|
+
jev?: Partial<JevJudgeConfig>;
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
/** Merge a judgment patch into the workspace config, preserving every other field. */
|
|
140
|
+
export async function setJudgeConfig(
|
|
141
|
+
configPath: string,
|
|
142
|
+
patch: JudgeConfigPatch,
|
|
143
|
+
): Promise<DeepClauseConfig> {
|
|
144
|
+
let existing: Record<string, unknown> = {};
|
|
145
|
+
try {
|
|
146
|
+
const parsed: unknown = JSON.parse(await readFile(configPath, "utf8"));
|
|
147
|
+
if (parsed && typeof parsed === "object" && !Array.isArray(parsed)) existing = parsed as Record<string, unknown>;
|
|
148
|
+
} catch (error) {
|
|
149
|
+
if ((error as NodeJS.ErrnoException).code !== "ENOENT") {
|
|
150
|
+
throw new Error(`Invalid DeepClause config: ${error instanceof Error ? error.message : String(error)}`);
|
|
151
|
+
}
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
const record = (value: unknown): Record<string, unknown> =>
|
|
155
|
+
value && typeof value === "object" && !Array.isArray(value) ? (value as Record<string, unknown>) : {};
|
|
156
|
+
|
|
157
|
+
const nextJudgment = {
|
|
158
|
+
...record(existing.judgment),
|
|
159
|
+
...(patch.default !== undefined ? { default: patch.default } : {}),
|
|
160
|
+
jev: { ...record(record(existing.judgment).jev), ...(patch.jev ?? {}) },
|
|
161
|
+
};
|
|
162
|
+
|
|
163
|
+
await writeFile(configPath, `${JSON.stringify({ ...existing, judgment: nextJudgment }, null, 2)}\n`, "utf8");
|
|
164
|
+
return loadConfig(configPath);
|
|
165
|
+
}
|