deepclause-pi 0.3.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,14 +1,16 @@
1
1
  ---
2
2
  name: handbook-dml
3
- description: Convert a long handbook/SOP into one DeepClause DML skill per workflow — an LLM-first subagent (agentic task/N leaves, narrow tool/2 capabilities) that takes a generic natural-language request as input and uses deterministic Prolog only for mechanical checks, with LLM fallback. Also teaches a tool audit, user confirmation, and how to write the repo-root AGENTS.md policy-routing table so pi calls the skills automatically via dc_run. Use when asked to turn a handbook or procedures manual into executable DML, update a handbook-derived skill, or wire the policy router.
3
+ description: Convert a long handbook/SOP into one DeepClause DML skill per workflow — a `task/N` subagent for agentic work, the judgment predicates (`choose`/`rate`/`verify`/`probability`/`holds`/`judge`) for bounded classification and calibrated gates, and deterministic Prolog only for mechanical checks, always with fallbacks. Also teaches a tool audit, user confirmation, and how to write the repo-root AGENTS.md policy-routing table so pi calls the skills automatically via dc_run. Use when asked to turn a handbook or procedures manual into executable DML, update a handbook-derived skill, choose between a classifier, a probability gate, and a full agent task, or wire the policy router.
4
4
  ---
5
5
 
6
6
  # Handbook → DML (procedures)
7
7
 
8
- Turn a long handbook into **one DML skill per workflow/procedure**. Each skill is
9
- an **LLM-first subagent**: agentic `task/N` leaves do the reading, reasoning,
10
- and acting, while Prolog handles only what is genuinely mechanical (arithmetic,
11
- counting, exact equality) or a hard safety invariant.
8
+ Turn a long handbook into **one DML skill per workflow/procedure**. Each skill
9
+ combines three primitives: agentic `task/N` leaves for work that must act or
10
+ reason across turns, bounded judgment predicates (`choose`/`rate`/`verify`/
11
+ `probability`/`holds`) for classification and calibrated gates, and Prolog only
12
+ for what is genuinely mechanical (arithmetic, counting, exact equality) or a
13
+ hard safety invariant.
12
14
 
13
15
  Before writing DML, read `.pi/deepclause/AGENTS.md` and
14
16
  `.pi/deepclause/DML_REFERENCE.md`. They are authoritative for syntax.
@@ -17,19 +19,117 @@ Before writing DML, read `.pi/deepclause/AGENTS.md` and
17
19
 
18
20
  - **Pi authors; DML is the runtime artifact.** Decomposition and authoring happen
19
21
  in a normal pi turn. There is no Markdown→DML compiler.
20
- - **LLM-first.** Rules and flow are `task/N` / `prompt/N` by default. Use Prolog
21
- only where a rule is mechanical or must never be wrong.
22
+ - **Pick the cheapest primitive.** Bounded classifications, ratings, yes/no
23
+ checks, and probabilities use the judgment predicates (`choose/4`, `rate/4`,
24
+ `verify/3`, `probability/3`, `holds/2-3`, `judge/2`). Fresh-context generation
25
+ and critique use `prompt/N`. Multi-turn work with memory, tools, and typed
26
+ outputs uses `task/N`. Use Prolog only where a rule is mechanical or must
27
+ never be wrong. See **Choosing the reasoning primitive** below.
22
28
  - **Input is a generic request.** `agent_main(Request)` takes free text; the
23
29
  first step is an LLM `task/N` that parses it into a typed `object/1` case (or
24
30
  the request is used directly for simple skills).
25
31
  - **Forbidden actions become tool scoping.** "Never send" means *no send tool*.
26
32
  "Read-only inspection" means the inspect phase gets read tools only.
27
- - **Every deterministic rule gets an LLM fallback.** If a deterministic check
28
- fails, branch to a `prompt/N`/`task/N` that reviews, repairs, or asks the user
29
- — do not hard-fail.
33
+ - **Every deterministic rule gets a fallback.** If a deterministic check
34
+ fails, branch to a judgment, a `prompt/N`/`task/N`, or the user — do not
35
+ hard-fail.
30
36
  - **Ask, don't assume.** Confirm scope, the decomposition, and tool choices with
31
37
  the user before authoring.
32
38
 
39
+ ## Choosing the reasoning primitive
40
+
41
+ Before writing flow, ask what the **core question** of each step is. A full
42
+ `task/N` agent loop carries memory, DML tools, and a multi-turn reasoning loop;
43
+ it is the right answer only when the step must act or reason iteratively. A
44
+ bounded question over explicit text should use the judgment predicates instead:
45
+ they are cheaper, the answer is constrained to the labels/levels you supply,
46
+ they never touch DML memory, and they cannot call tools.
47
+
48
+ | Core question | Primitive | Notes |
49
+ | --- | --- | --- |
50
+ | "Which category/team/route?" from a small closed set | `choose/4` (or `judge/2` with `choose`) | The answer is always one of your option atoms |
51
+ | "How severe/frustrated/confident?" (ordered) | `rate/4` | Answer is constrained to your levels |
52
+ | "Is X true?" / "Does it ask for a refund?" | `verify/3`, or `holds/2` when only the boolean matters | Three-valued: `yes` / `no` / `unknown` |
53
+ | "How likely is X?" / "Is it over a threshold?" | `probability/3`, or `holds/3` for a threshold gate | Decision-grade only when the backend reports `calibrated` |
54
+ | Free-form generation, rewrite, or critique of supplied text | `prompt/N` | Fresh context, no tools, no memory |
55
+ | Multi-step job needing tools, memory, or iteration | `task/N` | Typed outputs; scope tools with `with_tools/2` |
56
+
57
+ **Simple classifier.** If the core question reduces to choosing among a fixed
58
+ set of labels, use `choose/4`. Do not spend a `task/N` on it, and do not let the
59
+ model invent a label: the judge is constrained to your options.
60
+
61
+ ```prolog
62
+ route(Request, Team) :-
63
+ choose(Request, "Which team should handle this request?",
64
+ [billing-"Charges and refunds", orders-"Delivery and returns", account-"Login and security"],
65
+ Team).
66
+ ```
67
+
68
+ **Calibrated probability.** If the decision depends on a probability, say so and
69
+ gate it. The default `llm` backend is *uncalibrated*: a number from it is an
70
+ estimate, not a calibrated probability. Wrap the judgment in
71
+ `require_judgment(calibrated, ...)` so the skill cannot silently run on a
72
+ backend that only estimates, and provide a fallback clause.
73
+
74
+ ```prolog
75
+ risk_band(Text, Band) :-
76
+ require_judgment(calibrated,
77
+ probability(Text, "What is the probability this is high risk?", P)),
78
+ ( P >= 0.8 -> Band = high
79
+ ; P >= 0.4 -> Band = medium
80
+ ; Band = low
81
+ ).
82
+
83
+ risk_band(Text, Band) :-
84
+ % Uncalibrated fallback: a coarse classifier, never a fake probability.
85
+ choose(Text, "Is this high, medium, or low risk?", [high, medium, low], Band).
86
+ ```
87
+
88
+ `holds(Text, Question, Threshold)` is the concise form when you only need the
89
+ threshold decision, and `holds(Text, Question)` is the concise form of a yes/no
90
+ check. `with_judgment(jev, Goal)` pins a scope to a specific backend (for
91
+ example a calibrated one). `require_judgment/2` fails **before** the judgment
92
+ runs, which is what makes the fallback clause above reachable; the runtime emits
93
+ a warning naming the missing capability.
94
+
95
+ **Full agent loop.** Use `task/N` (or `prompt/N`) when the step needs to read
96
+ accumulated memory, call a DML tool, run several model turns, or produce
97
+ free-form text. The judgment layer deliberately cannot do those things.
98
+
99
+ **Batch judgments.** When one step needs more than one or two judgments, send a
100
+ single `judge/2` batch rather than several one-offs:
101
+
102
+ ```prolog
103
+ judge(Message, [
104
+ choose("Which team should handle this?", [billing, orders, account]) - Team,
105
+ rate("How frustrated is the customer?", [calm, frustrated, angry]) - Frustration,
106
+ verify("Does the message ask for a refund?") - Refund,
107
+ probability("Is this urgent?") - Urgency
108
+ ]).
109
+ ```
110
+
111
+ ### Judgment rules
112
+
113
+ - `judge/2`, `choose/4`, `rate/4`, `verify/3`, `probability/3`, `holds/2`,
114
+ `holds/3`, `with_judgment/2`, and `require_judgment/2` are runtime special
115
+ predicates. Do not define your own predicates with those names — a common
116
+ collision is a hand-written `verify/3`. Name deterministic helpers
117
+ `verify_task/3`, `check_*`, and so on.
118
+ - `verify` is three-valued. `\+ verify(...)` does **not** mean "the model said
119
+ no"; inspect the returned `yes` / `no` / `unknown`.
120
+ - Keep `State` explicit: pass the exact text or object to judge. The judge sees
121
+ only that state plus the question and never reads DML memory.
122
+ - A `choose`/`rate` answer is constrained to your options/levels, so make them
123
+ exhaustive and mutually exclusive.
124
+ - Never present an `estimated` probability as `calibrated`. Require the
125
+ capability and fall back, or label the output as an estimate.
126
+ - Answers are memoized per run by backend, model, state, and questions, so
127
+ backtracking reuses an answer instead of re-querying the model.
128
+ - The deterministic linter warns when a skill has no `task()` call and suggests
129
+ `task()` for classification. That heuristic predates the judgment layer: a
130
+ skill whose core question is a bounded judgment can legitimately have no
131
+ `task/N`.
132
+
33
133
  ## Workflow
34
134
 
35
135
  1. **Ingest** the handbook to Markdown (PDF→`pdftotext`, DOCX/HTML→`pandoc`).
@@ -121,6 +221,13 @@ tool(user_feedback(Prompt, Response), "Ask the user one focused question and ret
121
221
  exec(ask_user(prompt: Prompt), Result),
122
222
  get_dict(user_response, Result, Response).
123
223
 
224
+ % --- bounded judgments (cheap: no memory, no tools) ------------------------
225
+ % Prefer a judgment over a task/N for a bounded question about explicit text.
226
+ % classify(Request, Kind) :-
227
+ % choose(Request, "Which case type is this?", [<kind_a>, <kind_b>, other], Kind).
228
+ % confirm_requirement(Report) :-
229
+ % holds(Report, "Does the report include the required referral advice?", 0.7).
230
+
124
231
  % --- deterministic helpers (mechanical only) -------------------------------
125
232
  <compute_or_check>(...). % arithmetic/counts; keep small
126
233
 
@@ -136,7 +243,7 @@ agent_main(Request) :-
136
243
  with_tools([<write tools>], (
137
244
  task("Produce the required effects for this case: {ConfirmedCase}.", string(Summary))
138
245
  )),
139
- <optional verification with LLM fallback>,
246
+ <optional judgment / prompt / deterministic verification with fallback>,
140
247
  answer(Final).
141
248
 
142
249
  agent_main(_) :-
@@ -146,49 +253,59 @@ agent_main(_) :-
146
253
  Notes:
147
254
 
148
255
  - `task/N` = agentic leaf (memory + DML tools). `prompt/N` = fresh-context
149
- review. `with_tools/2` scopes capability per phase.
256
+ generation/review. `choose`/`rate`/`verify`/`probability` = the judgment layer.
257
+ `with_tools/2` scopes capability per phase.
258
+ - Prefer a judgment over a `task/N` whenever the core question is a bounded
259
+ classification, rating, yes/no check, or probability gate.
150
260
  - Build `task/N` descriptions with `format/3` (not `{Var}` interpolation) when
151
261
  you embed dynamic values — avoids singleton-variable noise.
152
262
  - Mutable facts must be declared `:- dynamic` before `assertz`/`retract`.
153
263
 
154
- ## Verification (optional, LLM-first)
264
+ ## Verification (optional)
265
+
266
+ Only add checks when the procedure has observable post-conditions. Match the
267
+ check to the question: a bounded semantic check is a judgment, an open-ended
268
+ read is a model review, and a count is deterministic. Every check needs a
269
+ fallback path so it never hard-fails.
155
270
 
156
- Only add checks when the procedure has observable post-conditions. Prefer model
157
- review; use deterministic checks only for mechanical facts, and always give a
158
- deterministic check an LLM fallback.
271
+ - **Semantic gate (default for bounded checks).** "Does the output recommend
272
+ urgent referral when a danger sign is present?" is a `verify/3` question, or
273
+ `holds/2` when you only need the boolean. Use `holds/3` when the check is a
274
+ probability threshold.
275
+ - **Model review (`prompt/N`).** Tone, completeness, correctness of free text,
276
+ "does this read right", and open-ended rubric interpretation.
277
+ - **Deterministic (only when mechanical).** Counts, exact IDs, arithmetic. Keep
278
+ it tiny, and route failures to a judgment, a review, or the user.
159
279
 
160
280
  ```prolog
161
- % deterministic gate (mechanical only)
162
- holds(drafts_count) :- findall(_, draft(_,_,_,_), Ds), length(Ds, 2).
163
-
164
- verify_state(Failed, Report) :-
165
- findall(Name, (postcondition(Name), \+ holds(Name)), Failed),
166
- ( Failed = [] -> Report = "PASS" ; format(string(Report), "FAIL: ~w", [Failed]) ).
167
-
168
- % LLM fallback: never hard-fail on a deterministic check
169
- fallback_review(Failed, Verdict, Reason) :-
170
- format(string(Prompt),
171
- "These structural checks failed: ~w. Review the outcome and the requirement. Store 'acceptable' or 'needs-attention' in Verdict and a one-line reason in Reason.",
172
- [Failed]),
173
- prompt(Prompt, string(Verdict), string(Reason)).
281
+ % Semantic gate: a bounded yes/no question about explicit text.
282
+ referral_ok(Report) :-
283
+ holds(Report, "Does the report recommend urgent referral when a danger sign is present?").
284
+
285
+ % Deterministic gate (mechanical only).
286
+ drafts_complete :- findall(_, draft(_,_,_,_), Ds), length(Ds, 2).
287
+
288
+ verify_state(Report, Note) :-
289
+ ( referral_ok(Report), drafts_complete
290
+ -> Note = "verification passed"
291
+ ; % Fallback: never hard-fail; explain the failure and let pi/user decide.
292
+ format(string(Prompt),
293
+ "Review this outcome against the requirement and explain any gap: ~w. Store 'acceptable' or 'needs-attention' in Verdict and a one-line reason in Reason.",
294
+ [Report]),
295
+ prompt(Prompt, string(Verdict), string(Reason)),
296
+ format(string(Note), "fallback verdict ~w: ~w", [Verdict, Reason])
297
+ ).
174
298
  ```
175
299
 
176
300
  In `agent_main`:
177
301
 
178
302
  ```prolog
179
- verify_state(Failed, Report),
180
- ( Failed = [] -> V = "n/a", R = "deterministic checks passed"
181
- ; fallback_review(Failed, V, R)
182
- ),
183
- answer(... report Report + V/R ...).
303
+ verify_state(Report, Note),
304
+ answer(... report + Note ...).
184
305
  ```
185
306
 
186
- Choosing:
187
-
188
- - **Model review** (default): tone, completeness, correctness of free text,
189
- "does this read right", and anything the rubric phrases as a judgment.
190
- - **Deterministic** (only when mechanical): counts, exact IDs, arithmetic. Keep
191
- it tiny, and route failures to the model or the user instead of failing.
307
+ For a calibrated postcondition, wrap `holds(Report, Question, Threshold)` in
308
+ `require_judgment(calibrated, ...)` and keep the same fallback pattern.
192
309
 
193
310
  ## Dummy tools
194
311
 
@@ -243,10 +360,13 @@ For each generated skill:
243
360
  — runs clean; verification (if any) passes or the fallback explains.
244
361
  2. Inspect phase is read-only, action phase is write-only, and a forbidden
245
362
  action has no tool at all.
246
- 3. Deterministic code is limited to arithmetic/counts; every deterministic
247
- check has an LLM fallback branch.
248
- 4. The tool audit was done and its result is recorded in the header/INDEX.
249
- 5. Re-check the invalid-pattern table in `.pi/deepclause/AGENTS.md` (singleton
363
+ 3. Bounded classifications and checks use the judgment predicates; calibrated
364
+ probabilities are wrapped in `require_judgment(calibrated, ...)` with a
365
+ fallback clause, and `verify/3` results are read as `yes`/`no`/`unknown`.
366
+ 4. Deterministic code is limited to arithmetic/counts; every deterministic
367
+ check has a model/judgment fallback branch.
368
+ 5. The tool audit was done and its result is recorded in the header/INDEX.
369
+ 6. Re-check the invalid-pattern table in `.pi/deepclause/AGENTS.md` (singleton
250
370
  variables, `~` vs `{}` interpolation, `Result.field` vs `get_dict/3`, `->`
251
371
  committing over generators, `answer/1` last, `:- dynamic` before
252
372
  `assertz/retract`).
@@ -254,12 +374,20 @@ For each generated skill:
254
374
  `--context=isolated` keeps session text out of the run. The runtime still needs
255
375
  a model selected.
256
376
 
377
+ To exercise calibrated judgments for real, enable the Jev backend
378
+ (`/dc-judge enable`, `/dc-judge default jev`, or `--judge=jev`) and export the
379
+ backend's API key (default `TYPESAFE_API_KEY`) in the shell that launches pi.
380
+ Without it, the `llm` backend is uncalibrated and the
381
+ `require_judgment(calibrated, ...)` branch is skipped in favor of the fallback.
382
+
257
383
  ## Reference shape
258
384
 
259
385
  A typical procedure skill has: `agent_main(Request)` that parses the request
260
- with `task/N`, confirms with the user via a `user_feedback` tool loop, acts
386
+ with `task/N`, confirms with the user via a `user_feedback` tool loop, uses
387
+ bounded judgments (`choose`/`verify`/`holds`) for classification and gates, acts
261
388
  through write tools, and reviews with `prompt/N` — with small deterministic
262
389
  helpers for arithmetic and a fallback branch instead of hard failures.
263
390
 
264
391
  For DML mechanics, see the bundled example skills in a fresh workspace
265
- (`example.dml`, `deep_research.dml`) and `.pi/deepclause/AGENTS.md`.
392
+ (`example.dml`, `deep_research.dml`), `.pi/deepclause/DML_REFERENCE.md` (the
393
+ judgment predicates are documented there), and `.pi/deepclause/AGENTS.md`.
@@ -111,15 +111,51 @@ A task can bind up to four outputs. Keep each task focused even though the runti
111
111
 
112
112
  ### `prompt/N`: an isolated model call
113
113
 
114
- Use `prompt/N` for a subtask that should not inherit accumulated conversation memory: independent classification, adversarial review, or formatting based only on explicitly supplied text.
114
+ Use `prompt/N` for a subtask that should not inherit accumulated conversation memory: adversarial review, rewriting, or formatting based only on explicitly supplied text. For a bounded classification, prefer the judgment predicates below.
115
115
 
116
116
  ```prolog
117
- prompt("Classify this text as low, medium, or high risk: {Text}. Store the label in Risk.",
118
- string(Risk)).
117
+ prompt("Rewrite this summary in plain language for a patient: {Text}. Store only the rewrite in Plain.",
118
+ string(Plain)).
119
119
  ```
120
120
 
121
121
  Fresh context is not a security boundary. Untrusted text can still contain hostile instructions; delimit it, state how it may be used, and request narrow structured output.
122
122
 
123
+ ### Semantic judgments: bounded, typed questions
124
+
125
+ Use the judgment predicates when the core question is a **bounded, typed question about explicit state**. They are not agentic: they do not read or write DML memory, they cannot call tools, and their answers are constrained to the options or levels you supply. That makes them cheaper and more reliable than a `task/N` for classification, rating, verification, and probability gates.
126
+
127
+ | Core question | Predicate |
128
+ | --- | --- |
129
+ | "Which label/route?" from a closed set | `choose(State, Question, Options, Choice)` |
130
+ | "How severe/frustrated/confident?" on an ordered scale | `rate(State, Question, Levels, Level)` |
131
+ | "Is X true?" (`yes`/`no`/`unknown`) | `verify(State, Question, Truth)`; `holds(State, Question)` for a semidet check |
132
+ | "How likely is X?" / "Is it above a threshold?" | `probability(State, Question, P)`; `holds(State, Question, Threshold)` |
133
+
134
+ Batch several questions into one request with `judge/2`:
135
+
136
+ ```prolog
137
+ judge(Message, [
138
+ choose("Which team should handle this?", [billing, orders, account]) - Team,
139
+ verify("Does the message ask for a refund?") - Refund
140
+ ]).
141
+ ```
142
+
143
+ Gate a calibrated probability and fall back when the backend only estimates:
144
+
145
+ ```prolog
146
+ risk_band(Text, Band) :-
147
+ require_judgment(calibrated, probability(Text, "Risk of harm?", P)),
148
+ ( P >= 0.8 -> Band = high ; Band = low ).
149
+ risk_band(Text, Band) :-
150
+ choose(Text, "Is the risk high or low?", [high, low], Band).
151
+ ```
152
+
153
+ `with_judgment(Backend, Goal)` selects a backend per scope; `require_judgment/2` fails before the judgment runs when the backend lacks a capability such as `calibrated`, so the second clause is the fallback. Answers are memoized per run by backend, model, state, and questions, so backtracking is cheap.
154
+
155
+ Do not name your own predicates `choose/4`, `rate/4`, `verify/3`, `probability/3`, `holds/2`, `holds/3`, `judge/2`, `with_judgment/2`, or `require_judgment/2`; they are runtime special predicates. `verify` is three-valued — never treat `\+ verify(...)` as "the model said no".
156
+
157
+ See `DML_REFERENCE.md` for the full syntax, options, and capability list.
158
+
123
159
  ### `llm/2`: low-level completion
124
160
 
125
161
  Prefer `task/N` and `prompt/N`. Use `get_memory/1` plus `llm/2` only when the program intentionally needs a raw completion over an explicit message list and does not need the task loop's result tools or DML tools.
@@ -357,7 +393,7 @@ Represent stable facts and rules as Prolog clauses; expose narrow query/update t
357
393
 
358
394
  ### 6. Independent reviewers
359
395
 
360
- Use `task/N` to draft and `prompt/N` to review from fresh context, then apply deterministic acceptance criteria. Best for code review, risk assessment, and editorial checks.
396
+ Use `task/N` to draft and `prompt/N` to review from fresh context, then apply deterministic acceptance criteria. Best for code review, risk assessment, and editorial checks. When the review is a bounded checklist ("does it state the referral threshold?"), use `verify/3` or `holds/2` instead of a free-form reviewer.
361
397
 
362
398
  ### 7. Pure deterministic utility
363
399
 
@@ -388,10 +424,10 @@ Pi can turn any DML file into a self-contained, offline Mermaid viewer. Ask for
388
424
 
389
425
  Pi calls the `dc_diagram` model tool with the DML path and a grade:
390
426
 
391
- - **presentation** — about 8-12 nodes, plain language, headline numbers (slides and overviews).
392
- - **specification** — function names, task/tool roles, post-conditions (engineers).
427
+ - **presentation** — about 8-12 nodes, plain language, headline numbers and the headline decision (slides and overviews).
428
+ - **specification** — function names, task/tool roles, post-conditions, plus the core decision logic: each decision predicate with its conditions, thresholds and outcomes, and the rule fact tables (engineers).
393
429
 
394
- The tool extracts a deterministic Mermaid seed, has pi rewrite it in the chosen grade, validates the result, writes the viewer under `.pi/deepclause/diagrams/`, and opens it. The DML file may live anywhere (workspace-relative or absolute); only the generated viewer stays under `.pi/deepclause/`.
430
+ The tool extracts a deterministic Mermaid seed, enriches the specification seed with a `LOGIC` section (decision predicates, guards, thresholds, judgments) and a `RULES` section (fact tables), has pi rewrite it in the chosen grade, validates the result, writes the viewer under `.pi/deepclause/diagrams/`, and opens it. The DML file may live anywhere (workspace-relative or absolute); only the generated viewer stays under `.pi/deepclause/`.
395
431
 
396
432
  Do not hand-write Mermaid for the user, and do not copy diagram tooling into the workspace. Regenerating a grade replaces only that grade's sidecar (`<name>.presentation.mmd` / `<name>.specification.mmd`).
397
433
 
@@ -85,7 +85,7 @@ attempt(TasksPath, Statuses0, Change, Id, Props, N, Max, Feedback, Statuses) :-
85
85
  format(string(Progress), "task ~w attempt ~w/~w", [Id, N, Max]),
86
86
  output(Progress),
87
87
  execute(Change, Id, Props, Feedback, Summary),
88
- verify(Props, Summary, Verdict),
88
+ verify_task(Props, Summary, Verdict),
89
89
  ( Verdict = ok
90
90
  -> sp_set_status(Statuses0, Id, done(N), Statuses),
91
91
  sp_write_statuses(TasksPath, Statuses),
@@ -123,7 +123,7 @@ build_instruction(Id, Do, Expected, Feedback, Instruction) :-
123
123
  "Task ~w. ~w~nExpected result: ~w~n~nThe previous attempt failed verification: ~w~nFix only what is needed; do not redo the whole task.",
124
124
  [Id, Do, Expected, Feedback]).
125
125
 
126
- verify(Props, Summary, Verdict) :-
126
+ verify_task(Props, Summary, Verdict) :-
127
127
  catch(get_dict(checks, Props, Checks), _, Checks = []),
128
128
  findall(Message, (member(Check, Checks), run_check(Check, Message), Message \= ok), Failures),
129
129
  ( Summary == ""
@@ -64,9 +64,12 @@ sp_sections(Lines, Sections) :-
64
64
  sp_skip_plain(Lines, Rest),
65
65
  sp_sections_from(Rest, Sections).
66
66
 
67
- sp_skip_plain([line(Line, Number)|Rest], [line(Line, Number)|Rest]) :- sp_heading(Line, Level, _), Level >= 1, Level =< 3, !.
68
- sp_skip_plain([_|Rest], Out) :- sp_skip_plain(Rest, Out).
69
67
  sp_skip_plain([], []).
68
+ sp_skip_plain([Item|Rest], Out) :-
69
+ ( Item = line(Line, _), sp_heading(Line, Level, _), Level >= 1, Level =< 3
70
+ -> Out = [Item|Rest]
71
+ ; sp_skip_plain(Rest, Out)
72
+ ).
70
73
 
71
74
  sp_sections_from([], []).
72
75
  sp_sections_from([line(Line, N)|Rest], [section(Level, Text, N, Body)|More]) :-
@@ -77,12 +80,11 @@ sp_sections_from([line(Line, N)|Rest], [section(Level, Text, N, Body)|More]) :-
77
80
  sp_sections_from(Tail, More).
78
81
 
79
82
  sp_take_body([], [], []).
80
- sp_take_body([line(Line, Number)|Rest], [], [line(Line, Number)|Rest]) :-
81
- sp_heading(Line, Level, _),
82
- Level >= 1,
83
- Level =< 3,
84
- !.
85
- sp_take_body([Item|Rest], [Item|Body], Tail) :- sp_take_body(Rest, Body, Tail).
83
+ sp_take_body([Item|Rest], Body, Tail) :-
84
+ ( Item = line(Line, _), sp_heading(Line, Level, _), Level >= 1, Level =< 3
85
+ -> Body = [], Tail = [Item|Rest]
86
+ ; Body = [Item|Body1], sp_take_body(Rest, Body1, Tail)
87
+ ).
86
88
 
87
89
  %% Split a requirement body (level-4 headings inline) into description + scenarios.
88
90
  sp_split_scenarios(Body, DescLines, ScenarioSections) :-
@@ -90,11 +92,11 @@ sp_split_scenarios(Body, DescLines, ScenarioSections) :-
90
92
  sp_scenario_sections(Rest, ScenarioSections).
91
93
 
92
94
  sp_take_plain([], [], []).
93
- sp_take_plain([line(Line, Number)|Rest], [], [line(Line, Number)|Rest]) :-
94
- sp_heading(Line, Level, _),
95
- Level >= 4,
96
- !.
97
- sp_take_plain([Item|Rest], [Item|Body], Tail) :- sp_take_plain(Rest, Body, Tail).
95
+ sp_take_plain([Item|Rest], Body, Tail) :-
96
+ ( Item = line(Line, _), sp_heading(Line, Level, _), Level >= 4
97
+ -> Body = [], Tail = [Item|Rest]
98
+ ; Body = [Item|Body1], sp_take_plain(Rest, Body1, Tail)
99
+ ).
98
100
 
99
101
  sp_scenario_sections([], []).
100
102
  sp_scenario_sections([line(Line, N)|Rest], [section(4, Text, N, Body)|More]) :-
@@ -103,11 +105,11 @@ sp_scenario_sections([line(Line, N)|Rest], [section(4, Text, N, Body)|More]) :-
103
105
  sp_scenario_sections(Tail, More).
104
106
 
105
107
  sp_take_until_level4([], [], []).
106
- sp_take_until_level4([line(Line, Number)|Rest], [], [line(Line, Number)|Rest]) :-
107
- sp_heading(Line, Level, _),
108
- Level >= 4,
109
- !.
110
- sp_take_until_level4([Item|Rest], [Item|Body], Tail) :- sp_take_until_level4(Rest, Body, Tail).
108
+ sp_take_until_level4([Item|Rest], Body, Tail) :-
109
+ ( Item = line(Line, _), sp_heading(Line, Level, _), Level >= 4
110
+ -> Body = [], Tail = [Item|Rest]
111
+ ; Body = [Item|Body1], sp_take_until_level4(Rest, Body1, Tail)
112
+ ).
111
113
 
112
114
  %% ---------------------------------------------------------------------------
113
115
  %% Term extraction
@@ -164,10 +166,11 @@ sp_parse_delta(Text, Deltas) :-
164
166
  sp_collect_ops(Rest, Deltas).
165
167
 
166
168
  sp_drop_to_op([], []).
167
- sp_drop_to_op([section(2, Header, Number, Body)|Rest], [section(2, Header, Number, Body)|Rest]) :-
168
- sp_delta_header(Header, _),
169
- !.
170
- sp_drop_to_op([_|Rest], Out) :- sp_drop_to_op(Rest, Out).
169
+ sp_drop_to_op([Item|Rest], Out) :-
170
+ ( Item = section(2, Header, _, _), sp_delta_header(Header, _)
171
+ -> Out = [Item|Rest]
172
+ ; sp_drop_to_op(Rest, Out)
173
+ ).
171
174
 
172
175
  sp_collect_ops([], []).
173
176
  sp_collect_ops([section(2, Header, _, _)|Rest], Deltas) :-
@@ -184,10 +187,11 @@ sp_collect_ops([section(2, Header, _, _)|Rest], Deltas) :-
184
187
  ).
185
188
 
186
189
  sp_take_until_op([], [], []).
187
- sp_take_until_op([section(2, Header, Number, Body)|Rest], [], [section(2, Header, Number, Body)|Rest]) :-
188
- sp_delta_header(Header, _),
189
- !.
190
- sp_take_until_op([Item|Rest], [Item|Group], Tail) :- sp_take_until_op(Rest, Group, Tail).
190
+ sp_take_until_op([Item|Rest], Group, Tail) :-
191
+ ( Item = section(2, Header, _, _), sp_delta_header(Header, _)
192
+ -> Group = [], Tail = [Item|Rest]
193
+ ; Group = [Item|Group1], sp_take_until_op(Rest, Group1, Tail)
194
+ ).
191
195
 
192
196
  %% ---------------------------------------------------------------------------
193
197
  %% Validation
package/src/config.ts CHANGED
@@ -2,6 +2,21 @@ import { readFile, writeFile } from "node:fs/promises";
2
2
 
3
3
  export type ContextMode = "turn" | "branch" | "isolated";
4
4
 
5
+ export interface JevJudgeConfig {
6
+ /** Whether the Jev (TypeSafe System One) backend may be used. */
7
+ enabled: boolean;
8
+ /** Model alias or pinned version passed to TypeSafe. */
9
+ model: string;
10
+ /** Environment variable holding the TypeSafe API key. */
11
+ apiKeyEnv: string;
12
+ }
13
+
14
+ export interface JudgmentConfig {
15
+ /** Default backend name: "llm", "jev", or a name registered by an extension. */
16
+ default: string;
17
+ jev: JevJudgeConfig;
18
+ }
19
+
5
20
  export interface DeepClauseConfig {
6
21
  version: 1;
7
22
  contextMode: ContextMode;
@@ -10,6 +25,7 @@ export interface DeepClauseConfig {
10
25
  maxTokens: number;
11
26
  verbose: boolean;
12
27
  modelToolEnabled: boolean;
28
+ judgment: JudgmentConfig;
13
29
  }
14
30
 
15
31
  export const DEFAULT_CONFIG: DeepClauseConfig = {
@@ -20,6 +36,10 @@ export const DEFAULT_CONFIG: DeepClauseConfig = {
20
36
  maxTokens: 16_384,
21
37
  verbose: false,
22
38
  modelToolEnabled: false,
39
+ judgment: {
40
+ default: "llm",
41
+ jev: { enabled: false, model: "jev-latest", apiKeyEnv: "TYPESAFE_API_KEY" },
42
+ },
23
43
  };
24
44
 
25
45
  const isContextMode = (value: unknown): value is ContextMode =>
@@ -55,6 +75,42 @@ export async function loadConfig(path: string): Promise<DeepClauseConfig> {
55
75
  maxTokens: positiveInteger("maxTokens", DEFAULT_CONFIG.maxTokens),
56
76
  verbose: config.verbose === true,
57
77
  modelToolEnabled: config.modelToolEnabled === true,
78
+ judgment: parseJudgmentConfig(config.judgment),
79
+ };
80
+ }
81
+
82
+ function parseJudgmentConfig(value: unknown): JudgmentConfig {
83
+ if (value === undefined) return DEFAULT_CONFIG.judgment;
84
+ if (!value || typeof value !== "object" || Array.isArray(value)) {
85
+ throw new Error("judgment must be a JSON object");
86
+ }
87
+ const judgment = value as Record<string, unknown>;
88
+ const defaultBackend = judgment.default ?? DEFAULT_CONFIG.judgment.default;
89
+ if (typeof defaultBackend !== "string" || !defaultBackend.trim()) {
90
+ throw new Error("judgment.default must be a non-empty backend name");
91
+ }
92
+
93
+ const jevValue = judgment.jev ?? {};
94
+ if (!jevValue || typeof jevValue !== "object" || Array.isArray(jevValue)) {
95
+ throw new Error("judgment.jev must be a JSON object");
96
+ }
97
+ const jev = jevValue as Record<string, unknown>;
98
+ const model = jev.model ?? DEFAULT_CONFIG.judgment.jev.model;
99
+ const apiKeyEnv = jev.apiKeyEnv ?? DEFAULT_CONFIG.judgment.jev.apiKeyEnv;
100
+ if (typeof model !== "string" || !model.trim()) {
101
+ throw new Error("judgment.jev.model must be a non-empty string");
102
+ }
103
+ if (typeof apiKeyEnv !== "string" || !apiKeyEnv.trim()) {
104
+ throw new Error("judgment.jev.apiKeyEnv must be a non-empty environment variable name");
105
+ }
106
+
107
+ return {
108
+ default: defaultBackend.trim(),
109
+ jev: {
110
+ enabled: jev.enabled === true,
111
+ model: model.trim(),
112
+ apiKeyEnv: apiKeyEnv.trim(),
113
+ },
58
114
  };
59
115
  }
60
116
 
@@ -72,3 +128,38 @@ export async function setModelToolEnabled(configPath: string, enabled: boolean):
72
128
  await writeFile(configPath, `${JSON.stringify({ ...existing, modelToolEnabled: enabled }, null, 2)}\n`, "utf8");
73
129
  return loadConfig(configPath);
74
130
  }
131
+
132
+ export interface JudgeConfigPatch {
133
+ /** Default backend name ("llm" or "jev"). */
134
+ default?: string;
135
+ /** Partial update of the Jev backend settings. */
136
+ jev?: Partial<JevJudgeConfig>;
137
+ }
138
+
139
+ /** Merge a judgment patch into the workspace config, preserving every other field. */
140
+ export async function setJudgeConfig(
141
+ configPath: string,
142
+ patch: JudgeConfigPatch,
143
+ ): Promise<DeepClauseConfig> {
144
+ let existing: Record<string, unknown> = {};
145
+ try {
146
+ const parsed: unknown = JSON.parse(await readFile(configPath, "utf8"));
147
+ if (parsed && typeof parsed === "object" && !Array.isArray(parsed)) existing = parsed as Record<string, unknown>;
148
+ } catch (error) {
149
+ if ((error as NodeJS.ErrnoException).code !== "ENOENT") {
150
+ throw new Error(`Invalid DeepClause config: ${error instanceof Error ? error.message : String(error)}`);
151
+ }
152
+ }
153
+
154
+ const record = (value: unknown): Record<string, unknown> =>
155
+ value && typeof value === "object" && !Array.isArray(value) ? (value as Record<string, unknown>) : {};
156
+
157
+ const nextJudgment = {
158
+ ...record(existing.judgment),
159
+ ...(patch.default !== undefined ? { default: patch.default } : {}),
160
+ jev: { ...record(record(existing.judgment).jev), ...(patch.jev ?? {}) },
161
+ };
162
+
163
+ await writeFile(configPath, `${JSON.stringify({ ...existing, judgment: nextJudgment }, null, 2)}\n`, "utf8");
164
+ return loadConfig(configPath);
165
+ }