clearotron 0.3.2 → 0.3.3-beta.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CONTRIBUTING.md +12 -0
- package/INSTALL.md +8 -0
- package/bin/onboard.mjs +29 -7
- package/bin/start.mjs +1 -1
- package/build-info.json +2 -2
- package/demo/MANIFEST.json +27 -0
- package/docs/INTAKE.md +8 -0
- package/docs/architecture/04-configuration-reference.md +1 -1
- package/driver/CHANGELOG.md +37 -0
- package/driver/clearance-variants-record.mjs +12 -1
- package/driver/common-law-coverage-status.mjs +113 -0
- package/driver/contract-audit.mjs +1 -1
- package/driver/contract-e3-backlog.mjs +37 -37
- package/driver/contract-vocabulary.mjs +8 -8
- package/driver/coverage-form-io.mjs +3 -1
- package/driver/coverage-form.mjs +38 -11
- package/driver/coverage-ledger.mjs +37 -7
- package/driver/coverage-union.mjs +2 -2
- package/driver/crowd-context.mjs +19 -6
- package/driver/drainer-identity.mjs +1 -1
- package/driver/engine/mcp/clarivate-server.mjs +4 -2
- package/driver/engine/mcp/corsearch-server.mjs +3 -1
- package/driver/engine/mcp/coverage-server.mjs +1 -1
- package/driver/engine/mcp/dispositions-server.mjs +47 -5
- package/driver/engine/mcp/euipo-server.mjs +2 -0
- package/driver/engine/mcp/free-tier-server.mjs +2 -0
- package/driver/engine/mcp/gather-config.mjs +8 -2
- package/driver/engine/mcp/probe-server.mjs +28 -0
- package/driver/engine/mcp/proposal-fields.mjs +45 -0
- package/driver/engine/mcp/recording-server.mjs +30 -0
- package/driver/engine/mcp/signa-server.mjs +2 -0
- package/driver/engine/mcp/supplemental.mjs +89 -12
- package/driver/engine/mcp/unit-note-server.mjs +50 -0
- package/driver/engine/mcp/uspto-local-server.mjs +2 -0
- package/driver/engine/openai-agent.mjs +7 -0
- package/driver/engine/probe.mjs +67 -14
- package/driver/engine/tool-refusal.mjs +16 -0
- package/driver/enqueue-schema.mjs +2 -2
- package/driver/envelope-settle.mjs +82 -13
- package/driver/findings-model.mjs +4 -4
- package/driver/gateway.mjs +18 -2
- package/driver/manager-groups-verdict.mjs +1 -1
- package/driver/matter-frame-record.mjs +24 -7
- package/driver/named-band.mjs +1 -1
- package/driver/package.json +1 -1
- package/driver/partial-payload-baseline.json +12 -3
- package/driver/pipeline.mjs +141 -37
- package/driver/publish/index.mjs +40 -25
- package/driver/publish/xlsx.mjs +26 -4
- package/driver/queue-markers.mjs +44 -0
- package/driver/queue-watch-verdict.mjs +2 -2
- package/driver/register-availability.mjs +2 -2
- package/driver/register-plan.mjs +313 -21
- package/driver/roster-verdict.mjs +1 -1
- package/driver/runner.mjs +26 -2
- package/driver/skills/clearance-common-law/SKILL.md +2 -0
- package/driver/skills/clearance-register/SKILL.md +44 -3
- package/driver/skills/clearance-register/digest.md +5 -5
- package/driver/skills/clearance-register/providers/clarivate.md +1 -1
- package/driver/skills/clearance-register/unit.md +39 -0
- package/driver/skills/clearance-variants/SKILL.md +1 -1
- package/driver/skills/matter-frame/SKILL.md +4 -2
- package/driver/stages.mjs +12 -5
- package/driver/status-snapshot.mjs +1 -1
- package/driver/suite-census.json +236 -14
- package/driver/synthesis-record.mjs +80 -2
- package/driver/unit-file-drift.mjs +3 -3
- package/driver/unit-inventory.mjs +2 -2
- package/driver/unit-state-verdict.mjs +1 -1
- package/driver/updater-identity.mjs +2 -3
- package/driver/variant-manifest-model.mjs +11 -1
- package/driver/verify.mjs +5 -5
- package/driver/withheld-families.mjs +104 -0
- package/mcp-server/CHANGELOG.md +4 -0
- package/mcp-server/lib/brief.mjs +5 -7
- package/mcp-server/lib/runs.mjs +1 -1
- package/mcp-server/package.json +1 -1
- package/mcp-server/server.mjs +3 -2
- package/package.json +1 -1
- package/portal-ui/dist/assets/{index-DMthc7PQ.js → index-GBbbyQxc.js} +22 -4
- package/portal-ui/dist/index.html +1 -1
- package/portal-ui/package.json +1 -1
- package/providers/_shared/count.mjs +2 -2
- package/providers/_shared/enumerate.mjs +15 -2
- package/providers/_shared/execute-plan.mjs +19 -1
- package/providers/_shared/plan-guards.mjs +40 -0
- package/providers/clarivate/src/capabilities.js +15 -5
- package/providers/clarivate/src/core.js +41 -5
- package/providers/corsearch/src/capabilities.js +4 -0
- package/providers/oauth-mcp-bridge/CHANGELOG.md +4 -0
- package/providers/oauth-mcp-bridge/package.json +1 -1
- package/providers/signa/src/capabilities.js +22 -8
- package/providers/signa/src/core.js +12 -1
- package/scripts/demo-evidence.mjs +114 -0
- package/scripts/engine-probe.mjs +6 -5
- package/scripts/env-audit.mjs +1 -1
- package/scripts/freeze-example-run.mjs +3 -3
- package/scripts/live-surface-check.mjs +26 -19
- package/scripts/mint-suite-census.mjs +66 -0
- package/scripts/package-size-budget.mjs +117 -0
- package/scripts/register-plan-shape.mjs +259 -0
- package/scripts/release-note-required.mjs +38 -1
- package/scripts/settings-render-check.mjs +36 -0
- package/scripts/travelling-predicates.mjs +1 -1
- package/shared/identifier-scan.mjs +22 -5
|
@@ -40,6 +40,7 @@
|
|
|
40
40
|
// for it and this tool never guesses one.
|
|
41
41
|
import { serve } from "./stdio-server.mjs";
|
|
42
42
|
import { recordUnitNote } from "../../register-unit-record.mjs";
|
|
43
|
+
import { recordWithheldFamilies, WITHHELD_REASON_MAX } from "../../withheld-families.mjs";
|
|
43
44
|
|
|
44
45
|
async function record_unit_note(params) {
|
|
45
46
|
const runDir = String(process.env.CLEAROTRON_BAND_RUN_DIR ?? "");
|
|
@@ -66,6 +67,38 @@ async function record_unit_note(params) {
|
|
|
66
67
|
}
|
|
67
68
|
}
|
|
68
69
|
|
|
70
|
+
// ── THE FAMILIES THE READING TURN LEAVES UNASKED (withheld-families.mjs) ───────────────────────────
|
|
71
|
+
//
|
|
72
|
+
// Same server, same binding: the axis is the one the driver fanned this seat out for, and a payload
|
|
73
|
+
// naming another is refused. What it records is a judgment only this turn can make — which waiting
|
|
74
|
+
// families were not worth asking, and why — so it lives beside the turn's note rather than on the
|
|
75
|
+
// digest's coverage tool, which never saw the families.
|
|
76
|
+
async function record_withheld_families(params) {
|
|
77
|
+
const runDir = String(process.env.CLEAROTRON_BAND_RUN_DIR ?? "");
|
|
78
|
+
if (!runDir) return { isError: true, text: "ERROR: this server was started without a run — the driver wires it per run; there is no parameter for it and this tool never guesses one." };
|
|
79
|
+
const bound = String(process.env.CLEAROTRON_RECORD_AXIS ?? "").trim();
|
|
80
|
+
if (!bound) return { isError: true, text: "ERROR: this server was started without a bound axis — the driver binds the axis it fanned out for." };
|
|
81
|
+
const named = String(params?.axis ?? "").trim();
|
|
82
|
+
if (named && named !== bound)
|
|
83
|
+
return { isError: true, text: `ERROR: unit_axis_not_yours:${named} — you are the seat for axis "${bound}". Send "${bound}" or omit the field.` };
|
|
84
|
+
try {
|
|
85
|
+
const r = recordWithheldFamilies(runDir, { axis: bound, families: params?.families });
|
|
86
|
+
if (r.refused) return { isError: true, text: `REFUSED: ${r.refused}` };
|
|
87
|
+
if (r.write_failed) return { isError: true, text: `ERROR: the driver could not store this record (${r.write_failed}). This is a driver fault — do not re-type it.` };
|
|
88
|
+
const refused = r.rejected.map((x) => `- ${x.qid || "(no qid)"}: ${x.issue}`);
|
|
89
|
+
const left = r.still_to_judge;
|
|
90
|
+
return { isError: !r.recorded.length && refused.length > 0, text: [
|
|
91
|
+
`Recorded ${r.recorded.length} famil${r.recorded.length === 1 ? "y" : "ies"} as withheld-by-judgment on axis "${r.axis}".`,
|
|
92
|
+
...(refused.length ? ["Refused:", ...refused] : []),
|
|
93
|
+
left.length
|
|
94
|
+
? `Still to judge on this axis — waiting, not asked, no reason recorded (${left.length}): ${left.slice(0, 40).join(", ")}${left.length > 40 ? ", …" : ""}`
|
|
95
|
+
: "Every waiting family on this axis is now asked or recorded.",
|
|
96
|
+
].join("\n") };
|
|
97
|
+
} catch (e) {
|
|
98
|
+
return { isError: true, text: `ERROR: the driver could not record this call (${String(e?.message ?? e).slice(0, 200)}). This is a driver fault — do not re-type it.` };
|
|
99
|
+
}
|
|
100
|
+
}
|
|
101
|
+
|
|
69
102
|
serve({
|
|
70
103
|
name: "unit-note", version: "0.1.0",
|
|
71
104
|
tools: [{
|
|
@@ -84,5 +117,22 @@ serve({
|
|
|
84
117
|
} },
|
|
85
118
|
annotations: { readOnlyHint: false, destructiveHint: false, idempotentHint: true, openWorldHint: false },
|
|
86
119
|
handler: record_unit_note,
|
|
120
|
+
}, {
|
|
121
|
+
name: "record_withheld_families",
|
|
122
|
+
description:
|
|
123
|
+
"Record the WAITING families on this axis that you decided NOT to ask, each with your reason. A family you " +
|
|
124
|
+
"do not ask was never searched: it is recorded withheld-by-judgment, and the reason goes into the run's " +
|
|
125
|
+
"record and the audit workbook, never into the report. Every waiting family must end this run either asked " +
|
|
126
|
+
"or recorded here — one nobody judged holds up delivery. One reason may cover several families. The answer " +
|
|
127
|
+
"lists the waiting families on this axis that are still neither asked nor recorded.",
|
|
128
|
+
inputSchema: { type: "object", required: ["families"], properties: {
|
|
129
|
+
families: { type: "array", items: { type: "object", required: ["qids", "reason"], properties: {
|
|
130
|
+
qids: { type: "array", items: { type: "string" }, description: "The qids of waiting families on this axis, exactly as the dispatch lists them." },
|
|
131
|
+
reason: { type: "string", description: `Why these were not asked, in a lawyer's words (at most ${WITHHELD_REASON_MAX} characters). The audit workbook prints it.` },
|
|
132
|
+
} } },
|
|
133
|
+
axis: { type: "string", description: "Optional, and checked rather than trusted: the driver binds the axis it dispatched you for." },
|
|
134
|
+
} },
|
|
135
|
+
annotations: { readOnlyHint: false, destructiveHint: false, idempotentHint: true, openWorldHint: false },
|
|
136
|
+
handler: record_withheld_families,
|
|
87
137
|
}],
|
|
88
138
|
});
|
|
@@ -36,6 +36,7 @@ import {
|
|
|
36
36
|
CAPABILITIES, DEFAULT_DB_ENV, doSearch, doRecordFetch, doBatchScreen, doEnumerate, doExecutePlan,
|
|
37
37
|
} from "../../../providers/uspto-local/src/core.js";
|
|
38
38
|
import { proposeSupplemental } from "./supplemental.mjs";
|
|
39
|
+
import { narrowingFields } from "./proposal-fields.mjs";
|
|
39
40
|
|
|
40
41
|
const DB_PATH = process.env[DEFAULT_DB_ENV] || "";
|
|
41
42
|
// The auth object IS the index path — see the core's header. Passed as an object rather than a bare
|
|
@@ -139,6 +140,7 @@ serve({
|
|
|
139
140
|
romanization: { type: "string", description: "The Latin-script form of a NON-LATIN term — plain ASCII letters/digits, syllable-separated by single spaces, no tone marks or diacritics. On THIS source it does not rescue the slice: nativeScriptIndex is undeclared, so a non-Latin term defers and is disclosed rather than being answered by its romanisation." },
|
|
140
141
|
owner: { type: "string", description: "OPTIONAL owner scope field on a MARK-TEXT proposal: the query is the owner×term intersection (the owner's filings within the term band). Not allowed on predicate:owner (there the owner name IS the term)." },
|
|
141
142
|
nice_classes: { type: "array", items: {} },
|
|
143
|
+
...narrowingFields(), // the narrowing fields every register serves (proposal-fields.mjs)
|
|
142
144
|
rationale: { type: "string" },
|
|
143
145
|
term_literal: { type: "boolean", description: "TRUE only when the term genuinely IS the mark verbatim (a multi-word slogan mark, a mark carrying an anchored star) — it bypasses the term-shape lint. Never use it to push a label through." },
|
|
144
146
|
} } },
|
|
@@ -23,6 +23,7 @@
|
|
|
23
23
|
// `--dangerously-bypass-approvals-and-sandbox` — see buildCodexArgs below for why.
|
|
24
24
|
|
|
25
25
|
import { mkdtempSync, writeFileSync, copyFileSync, existsSync, rmSync, readFileSync, readdirSync, statSync, symlinkSync, lstatSync, realpathSync } from "node:fs";
|
|
26
|
+
import { everyToolCallRefused } from "./tool-refusal.mjs";
|
|
26
27
|
import { writeSecretFile } from "../../shared/secret-file.mjs"; // the rotated login goes back the way every credential is written
|
|
27
28
|
import { tmpdir, homedir } from "node:os";
|
|
28
29
|
import { join } from "node:path";
|
|
@@ -531,6 +532,12 @@ function settleTuple({ r, ev, resumeRef }) {
|
|
|
531
532
|
// The engine's sign-in could not be refreshed: what the operator runs to fix it. The gateway names the
|
|
532
533
|
// stage's failure with it.
|
|
533
534
|
signedOut: signedOut ? "codex sign-in expired — run `codex login`, then start the search again" : undefined,
|
|
535
|
+
// Every tool call refused and none completed: codex's own sandbox on this host, and what fixes it. The
|
|
536
|
+
// turn is still `ok` here; when the stage then fails, the gateway names its failure with this, as it
|
|
537
|
+
// names a sign-in, instead of the missing file the refusal left behind.
|
|
538
|
+
toolsRefused: everyToolCallRefused(mcpToolGauge(ev))
|
|
539
|
+
? "codex refused every tool call this search needs — set CLEAROTRON_CODEX_SANDBOX_BYPASS=1 in this install's environment file, or use the Anthropic engine, then start the search again"
|
|
540
|
+
: undefined,
|
|
534
541
|
rateLimitBasis: rateLimited ? "text-match" : undefined,
|
|
535
542
|
// resetsAtBasis (2026-08-20): same honesty as rateLimitBasis one line up, for the reset
|
|
536
543
|
// CLOCK rather than the classification. codex states its reset as human prose with NO timezone
|
package/driver/engine/probe.mjs
CHANGED
|
@@ -15,11 +15,20 @@
|
|
|
15
15
|
//
|
|
16
16
|
// THE CHEAPEST TURN THAT PROVES THE WHOLE PATH
|
|
17
17
|
//
|
|
18
|
-
// One `haiku`-tier turn at `low` effort
|
|
19
|
-
//
|
|
20
|
-
//
|
|
21
|
-
//
|
|
22
|
-
// stage
|
|
18
|
+
// One `haiku`-tier turn at `low` effort, asked to call one tool, with no skills dir and no run dir. The
|
|
19
|
+
// tool is `ping` on engine/mcp/probe-server.mjs, handed over exactly as a stage hands over its own.
|
|
20
|
+
//
|
|
21
|
+
// WHY IT CALLS A TOOL. It used to call none, and a turn with no tools proves nothing about the tools every
|
|
22
|
+
// search stage is given. On some hosts codex's own sandbox refuses every tool call while the turn reports
|
|
23
|
+
// success; the probe passed there, and every search then failed after real spend. So the probe now passes
|
|
24
|
+
// only when the reply carries the word `ping` returned, a random word minted for this turn and given to
|
|
25
|
+
// that server alone, so the model cannot supply it. Every call refused is a configuration fault, refused at
|
|
26
|
+
// the door; no call and no word shows nothing either way, and warns.
|
|
27
|
+
//
|
|
28
|
+
// It exercises every link a stage uses: binary → spawn → billing mode → credential → model access → a tool
|
|
29
|
+
// call through the stage's own tool path → a completed turn parsed by the adapter's own settle path. And
|
|
30
|
+
// it is far lighter than the thing it protects: one register-sweep stage prompt inlines 150 KB of plan and
|
|
31
|
+
// runs for minutes.
|
|
23
32
|
//
|
|
24
33
|
// WHAT IT DOES NOT PROVE, said out loud. It exercises the CHEAP tier. Both tiers ride one credential on
|
|
25
34
|
// a subscription, so AUTH is proven for all of them; a per-tier model entitlement or a per-tier quota is
|
|
@@ -56,9 +65,28 @@
|
|
|
56
65
|
|
|
57
66
|
import { ENGINE_BINARIES, DEFAULT_ENGINE_ID, engineAdapterSpecifier, resolveEngineProgram } from "../driver.config.mjs";
|
|
58
67
|
import { resolveAuthMode, CLOUD_SETTINGS, CLOUD_CREDENTIAL_CHECK } from "./auth.mjs";
|
|
68
|
+
import { everyToolCallRefused } from "./tool-refusal.mjs";
|
|
69
|
+
import { randomBytes } from "node:crypto";
|
|
70
|
+
import { fileURLToPath } from "node:url";
|
|
71
|
+
|
|
72
|
+
/** One tool call and one word back: short enough to be free in practice, and it needs a working tool path. */
|
|
73
|
+
export const PROBE_PROMPT = "Call the ping tool once, then reply with exactly the word it returned.";
|
|
74
|
+
/** The server and tool the probe hands the engine, named as a stage names its own (`mcp__<server>__<tool>`). */
|
|
75
|
+
export const PROBE_SERVER = "probe";
|
|
76
|
+
export const PROBE_TOOL = "ping";
|
|
77
|
+
const PROBE_SERVER_PATH = fileURLToPath(new URL("./mcp/probe-server.mjs", import.meta.url));
|
|
78
|
+
|
|
79
|
+
/** The tool config for one probe turn, in the shape both adapters take from a stage. */
|
|
80
|
+
export function probeToolConfig(sentinel) {
|
|
81
|
+
return {
|
|
82
|
+
mcpConfig: JSON.stringify({ mcpServers: { [PROBE_SERVER]: {
|
|
83
|
+
command: process.execPath, args: [PROBE_SERVER_PATH, String(sentinel)], env: {} } } }),
|
|
84
|
+
allowedTools: `mcp__${PROBE_SERVER}__${PROBE_TOOL}`,
|
|
85
|
+
};
|
|
86
|
+
}
|
|
59
87
|
|
|
60
|
-
/**
|
|
61
|
-
export const
|
|
88
|
+
/** A word no model would produce unprompted, fresh per turn. */
|
|
89
|
+
export const mintProbeSentinel = () => `probe-${randomBytes(4).toString("hex")}`;
|
|
62
90
|
/** The CHEAP rung of the driver's tier vocabulary on BOTH adapters (engine/CONTRACT.md §3). */
|
|
63
91
|
export const PROBE_MODEL = "haiku";
|
|
64
92
|
/** The floor rung of both EFFORT tables. There is nothing below `low` on anthropic. */
|
|
@@ -159,7 +187,7 @@ const capitalised = (s) => s.charAt(0).toUpperCase() + s.slice(1);
|
|
|
159
187
|
* (`{ source, path }`); both only shape the advice, never the mode, so the run door refuses exactly what it
|
|
160
188
|
* refused before. Absent, the advice is the subscription's, naming the bare program word.
|
|
161
189
|
*/
|
|
162
|
-
export function classifyProbe({ engine, tuple = null, error = null, timeoutSec = PROBE_TIMEOUT_SEC, auth = null, program = null } = {}) {
|
|
190
|
+
export function classifyProbe({ engine, tuple = null, error = null, timeoutSec = PROBE_TIMEOUT_SEC, auth = null, program = null, expect = null } = {}) {
|
|
163
191
|
const id = String(engine ?? "").trim().toLowerCase();
|
|
164
192
|
const v = (mode, basis, headline, fix, extra = {}) =>
|
|
165
193
|
({ ok: false, engine: id, mode, basis, headline, fix, detail: null, ...extra });
|
|
@@ -197,6 +225,21 @@ export function classifyProbe({ engine, tuple = null, error = null, timeoutSec =
|
|
|
197
225
|
// which pipe this happened to look at.
|
|
198
226
|
const detail = tail(tuple.stderr) ?? tail(tuple.stdout);
|
|
199
227
|
|
|
228
|
+
// THE TOOL, ON A TURN THAT COMPLETED. `expect` is the word the probe's tool returns, and only a probe that
|
|
229
|
+
// handed the engine a tool passes one; without it a completed turn proves what it always proved. With
|
|
230
|
+
// it, three outcomes, never two. Every call refused is this machine's configuration, read off the
|
|
231
|
+
// adapter's own gauge, and the run door refuses on it. The word in the reply is the pass. Neither is a
|
|
232
|
+
// turn that shows nothing about the tools either way: not a pass, and not this box's fault to refuse on.
|
|
233
|
+
if (tuple.code === 0 && expect != null) {
|
|
234
|
+
if (everyToolCallRefused(tuple))
|
|
235
|
+
return v("tools-refused", "tool-gauge", `${id} refused every tool call it was given`, toolsRefusedFix(id),
|
|
236
|
+
{ detail: tail(tuple.mcpToolCallRefusals?.[0]?.message) });
|
|
237
|
+
if (!String(tuple.stdout ?? "").includes(String(expect)))
|
|
238
|
+
return v("tools-unproven", "no-tool-answer", `${id} did not return the word its probe tool gives`,
|
|
239
|
+
"Nothing here shows that the tools a search needs work on this machine. Run this again; if it repeats, a search is likely to fail the same way.",
|
|
240
|
+
{ detail: tail(tuple.stdout) });
|
|
241
|
+
}
|
|
242
|
+
|
|
200
243
|
// A completed turn names what served it, the model and the provider as the program reported them, and
|
|
201
244
|
// null where it named neither, so a proof says which model and whose account it proved.
|
|
202
245
|
if (tuple.code === 0) return { ok: true, engine: id, mode: "ok", basis: "completed-turn", headline: `${id} completed a turn`, fix: null, detail: null,
|
|
@@ -234,7 +277,7 @@ export function classifyProbe({ engine, tuple = null, error = null, timeoutSec =
|
|
|
234
277
|
return v("tier-unavailable", "text-match", `${id} cannot reach the model it was asked for`, tierFix(id, text), { detail });
|
|
235
278
|
|
|
236
279
|
if (tuple.killed || s.stalled || s.hardWall)
|
|
237
|
-
return v("timed-out", "watchdog", `${id} started but did not finish
|
|
280
|
+
return v("timed-out", "watchdog", `${id} started but did not finish its probe turn in ${timeoutSec}s`,
|
|
238
281
|
"The binary runs and the turn produces nothing. Run the CLI by hand once and see what it is waiting for — an unanswered login prompt and a wedged MCP server both look like this.", { detail });
|
|
239
282
|
|
|
240
283
|
// The anthropic adapter's own diagnosis: the CLI exited without emitting a single stream event, i.e.
|
|
@@ -250,6 +293,14 @@ export function classifyProbe({ engine, tuple = null, error = null, timeoutSec =
|
|
|
250
293
|
"The engine's stderr below is the whole story; a turn that starts and fails is not a configuration this check can name.", { detail });
|
|
251
294
|
}
|
|
252
295
|
|
|
296
|
+
/** What fixes a host that refuses every tool call. On codex it is its own sandbox, and the setting is named. */
|
|
297
|
+
function toolsRefusedFix(engine) {
|
|
298
|
+
if (engine === "openai-agent")
|
|
299
|
+
return "Every search stage calls tools, so no search can finish here. codex's own sandbox refuses them on this machine: "
|
|
300
|
+
+ "set CLEAROTRON_CODEX_SANDBOX_BYPASS=1 in this install's environment file, or use the Anthropic engine, then run this again.";
|
|
301
|
+
return "Every search stage calls tools, so no search can finish here. The engine's stderr below is the place to start.";
|
|
302
|
+
}
|
|
303
|
+
|
|
253
304
|
/** The one place the tier doctrine is POINTED AT rather than re-authored. */
|
|
254
305
|
function tierFix(engine, msg) {
|
|
255
306
|
if (engine === "openai-agent")
|
|
@@ -290,7 +341,7 @@ export function probeFailureText(verdict) {
|
|
|
290
341
|
// classify is by definition one the door cannot claim to understand — and the cost of being wrong runs
|
|
291
342
|
// the other way here: refusing wrongly kills a run that would have worked, proceeding wrongly costs the
|
|
292
343
|
// stages before a failure the engine was going to produce anyway.
|
|
293
|
-
const CONFIGURATION_MODES = new Set(["unknown-engine", "auth-misconfigured", "signed-out", "tier-unavailable", "cannot-spawn"]);
|
|
344
|
+
const CONFIGURATION_MODES = new Set(["unknown-engine", "auth-misconfigured", "signed-out", "tier-unavailable", "cannot-spawn", "tools-refused"]);
|
|
294
345
|
|
|
295
346
|
// …and a mode alone is not enough, because `basis` says HOW WELL the mode is known and the ladder already
|
|
296
347
|
// makes that distinction for its own reasons. `startup-class` is an INFERENCE FROM SILENCE — the CLI died
|
|
@@ -299,7 +350,8 @@ const CONFIGURATION_MODES = new Set(["unknown-engine", "auth-misconfigured", "si
|
|
|
299
350
|
// hosted runner where a hermetic PATH left `env` unable to resolve node, and the probe "correctly reported
|
|
300
351
|
// a startup-class engine death" on a machine whose engine was fine. Refusing a production run on that
|
|
301
352
|
// inference manufactures the outage it was written to prevent, so it warns instead.
|
|
302
|
-
|
|
353
|
+
// `tool-gauge` is named, not inferred: the adapter counted each refused call off the engine's own stream.
|
|
354
|
+
const NAMED_BASES = new Set(["config", "text-match", "spawn-error", "tool-gauge"]);
|
|
303
355
|
|
|
304
356
|
/**
|
|
305
357
|
* "ok" | "configuration" | "weather" — PURE, and the whole of the door's judgment.
|
|
@@ -465,8 +517,9 @@ export async function probeEngineTurn({
|
|
|
465
517
|
// the environment back whatever it does. A resolver that cannot answer leaves the bare word.
|
|
466
518
|
if (knownProgram?.path) program = { source: knownProgram.source ?? null, path: knownProgram.path };
|
|
467
519
|
else try { const r = resolveEngineProgram(id); program = r.resolved ? { source: r.source, path: r.resolved } : null; } catch { /* the bare word */ }
|
|
468
|
-
const
|
|
469
|
-
|
|
520
|
+
const sentinel = mintProbeSentinel();
|
|
521
|
+
const tuple = await turn({ message: PROBE_PROMPT, model: PROBE_MODEL, thinking: PROBE_THINKING, timeoutSec, stallSec, ...probeToolConfig(sentinel) });
|
|
522
|
+
return classifyProbe({ engine: id, tuple, timeoutSec, auth, program, expect: sentinel });
|
|
470
523
|
} catch (e) {
|
|
471
524
|
return classifyProbe({ engine: id, error: e, timeoutSec, auth, program });
|
|
472
525
|
} finally {
|
|
@@ -479,7 +532,7 @@ export async function probeEngineTurn({
|
|
|
479
532
|
*
|
|
480
533
|
* The choice, stated so nobody has to re-decide it. The register-credential precedent refuses because a
|
|
481
534
|
* run whose work is impossible must say so before it costs anything, and the reasoning transfers exactly:
|
|
482
|
-
* every one of the fourteen stages spawns the engine, so an engine that cannot complete
|
|
535
|
+
* every one of the fourteen stages spawns the engine, so an engine that cannot complete its probe turn
|
|
483
536
|
* cannot complete any of them. A warning would let the run build its directory, freeze its profile and
|
|
484
537
|
* write its status sidecar, then die at stage one leaving a resumable-looking husk and a failure wearing
|
|
485
538
|
* the shape of a model fault — which is the precise outcome preflightEngineBinary exists to prevent, and
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
// SPDX-License-Identifier: AGPL-3.0-only
|
|
2
|
+
// Copyright 2026 Cordillera Sàrl. Additional terms under section 7 of the AGPL-3.0 apply — see ADDITIONAL-TERMS.md
|
|
3
|
+
// engine/tool-refusal.mjs — one definition of "this turn's every tool call was refused", for the two
|
|
4
|
+
// places that act on it: the gateway, which stops a stage's retries and names the failure, and the engine
|
|
5
|
+
// probe, which refuses a host where it happens before a search is paid for. Two copies would drift.
|
|
6
|
+
|
|
7
|
+
/**
|
|
8
|
+
* Every tool call the turn made was refused, and none completed. Read off the adapter's own gauge
|
|
9
|
+
* (`mcpToolCalls` / `mcpToolCallsRefused`), which counts a refusal only when the call never reached its
|
|
10
|
+
* server — a tool that ran and errored is not one. One refused call beside a completed one is a model
|
|
11
|
+
* asking for something it may not have, not a host that refuses tools, so it does not count. An engine
|
|
12
|
+
* that keeps no gauge reports nothing, and nothing is not a refusal. Pure.
|
|
13
|
+
*/
|
|
14
|
+
export function everyToolCallRefused(turn) {
|
|
15
|
+
return Number(turn?.mcpToolCallsRefused ?? 0) > 0 && Number(turn?.mcpToolCalls ?? 0) === 0;
|
|
16
|
+
}
|
|
@@ -28,7 +28,7 @@ import { productName, productSpec, checkProductScope, checkNativeLanguage, unkno
|
|
|
28
28
|
import { partitionTerritories } from "./territory-tiers.mjs";
|
|
29
29
|
// — the sidecar field map, from the module that OWNS it. The check below asks whether the prose a
|
|
30
30
|
// manifest declared actually arrived, and a local copy of that list is how the two would drift apart.
|
|
31
|
-
import { PROSE_PARTS } from "./queue-markers.mjs";
|
|
31
|
+
import { PROSE_PARTS, goodsOf } from "./queue-markers.mjs";
|
|
32
32
|
|
|
33
33
|
// ── per-run scope limits ──────────────────────────────────────────────────────────────────────────────
|
|
34
34
|
// Caps, not policy. They exist so one malformed request cannot mint an unbounded search: every extra
|
|
@@ -422,7 +422,7 @@ export function validateJob(job, { atClaim = false } = {}) {
|
|
|
422
422
|
}
|
|
423
423
|
// §B2: classes OR a goods description — either suffices; both absent ⇒ the subject can't be scoped.
|
|
424
424
|
const hasClasses = requestNamesClasses(job);
|
|
425
|
-
const hasGoods =
|
|
425
|
+
const hasGoods = goodsOf(job) !== null; // — the spellings live in queue-markers.mjs, with the fold that keeps every reader on one answer
|
|
426
426
|
if (!hasClasses && !hasGoods) {
|
|
427
427
|
// — THE GATE HAS TO ASK WHAT THE RUN WOULD ASK, not what the request typed.
|
|
428
428
|
//
|
|
@@ -36,6 +36,7 @@
|
|
|
36
36
|
|
|
37
37
|
import { existsSync, readFileSync, writeFileSync, renameSync } from "node:fs";
|
|
38
38
|
import { isCapabilityGapReason } from "./coverage-ledger.mjs";
|
|
39
|
+
import { fullyDeferredAxes } from "./register-plan.mjs";
|
|
39
40
|
|
|
40
41
|
export const SETTLE_SCHEMA_VERSION = 1;
|
|
41
42
|
|
|
@@ -52,22 +53,28 @@ export const SETTLE_SCHEMA_VERSION = 1;
|
|
|
52
53
|
* receipt is final. So `suspect` catches a MIS-STAMPED deferral — an executor bug flagging a transient as
|
|
53
54
|
* deterministic — and exists so that shape gets one attempt instead of silently becoming permanent.
|
|
54
55
|
*
|
|
55
|
-
*
|
|
56
|
+
* STICKY — a qid this run has already accepted as a capability gap, under any plan version, is accepted
|
|
57
|
+
* again whatever its current reason text says (stickyGaps below). PURE.
|
|
56
58
|
*/
|
|
57
|
-
export function partitionReceiptDeferrals(plan, receipt) {
|
|
59
|
+
export function partitionReceiptDeferrals(plan, receipt, { sticky = new Set() } = {}) {
|
|
58
60
|
const axisOf = new Map((plan?.entries ?? []).map((e) => [String(e?.qid ?? ""), String(e?.axis ?? "").toLowerCase()]));
|
|
59
61
|
// An axis the frozen plan says is deferred end to end is accepted whatever its per-qid reason text says:
|
|
60
62
|
// there is no slice of it left for a dispatch to reach.
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
63
|
+
//
|
|
64
|
+
// THE PLAN SAYS SO, NOT THE COVERAGE SKELETON. This read the skeleton's `deferred` state, which
|
|
65
|
+
// deriveCoverageSkeleton sets for an axis carrying ANY deferred qid and nothing missing, and at the
|
|
66
|
+
// fan-in nothing is missing. So every deferral on every axis was accepted: a provider hard error was
|
|
67
|
+
// logged "ACCEPTED as provider capability gaps … never retried", skipped the one bounded attempt the
|
|
68
|
+
// hard-error path is designed to get, and was then re-opened by the envelope, which reads the reason.
|
|
69
|
+
// fullyDeferredAxes asks the plan: every entry on the axis `unsupported`.
|
|
70
|
+
const fullyDeferred = new Set(fullyDeferredAxes(plan).map((a) => String(a.axis ?? "").toLowerCase()));
|
|
64
71
|
const accepted = [], suspect = [];
|
|
65
72
|
for (const d of receipt?.deferred ?? []) {
|
|
66
73
|
const qid = String(d?.qid ?? "");
|
|
67
74
|
const axis = axisOf.get(qid) ?? "";
|
|
68
75
|
const reason = String(d?.reason ?? "");
|
|
69
76
|
const row = { qid, axis, reason };
|
|
70
|
-
(isCapabilityGapReason(reason) || fullyDeferred.has(axis) ? accepted : suspect).push(row);
|
|
77
|
+
(isCapabilityGapReason(reason) || fullyDeferred.has(axis) || sticky.has(qid) ? accepted : suspect).push(row);
|
|
71
78
|
}
|
|
72
79
|
return { accepted, suspect };
|
|
73
80
|
}
|
|
@@ -129,7 +136,7 @@ const shallowRow = (d) => ({ qid: String(d?.qid ?? ""), reason: String(d?.reason
|
|
|
129
136
|
* the one case the file exists to make durable. `supersede()` builds the entry; it is deliberately a
|
|
130
137
|
* summary rather than a full copy, because the qid-level truth is always re-derivable from the receipt.
|
|
131
138
|
*/
|
|
132
|
-
export function buildDecisionDoc({ planVersion, deferredTotal, accepted, closed, closeFailed, decidedAt, history = [] }) {
|
|
139
|
+
export function buildDecisionDoc({ planVersion, deferredTotal, accepted, closed, closeFailed, decidedAt, history = [], stickyGaps = [], settleTried = [] }) {
|
|
133
140
|
return {
|
|
134
141
|
schema_version: SETTLE_SCHEMA_VERSION,
|
|
135
142
|
plan_version: planVersion ?? null,
|
|
@@ -138,10 +145,60 @@ export function buildDecisionDoc({ planVersion, deferredTotal, accepted, closed,
|
|
|
138
145
|
accepted: accepted ?? [],
|
|
139
146
|
closed: closed ?? [],
|
|
140
147
|
close_failed: closeFailed ?? [],
|
|
148
|
+
sticky_gaps: stickyGaps ?? [],
|
|
149
|
+
settle_tried: settleTried ?? [],
|
|
141
150
|
history: history ?? [],
|
|
142
151
|
};
|
|
143
152
|
}
|
|
144
153
|
|
|
154
|
+
// ── A CAPABILITY GAP IS DECIDED ONCE PER RUN, NOT ONCE PER PLAN VERSION ────────────────────────────────
|
|
155
|
+
//
|
|
156
|
+
// The live `accepted[]` is rebuilt from the current receipt on every settle, and a settle happens again
|
|
157
|
+
// whenever the plan version moves — every supplemental fold bumps it. So an acceptance lasted exactly as
|
|
158
|
+
// long as the version it was made under. On a production run on 2026-09-22 one slice the provider refused
|
|
159
|
+
// was accepted ("never retried"), re-opened, re-proposed, re-executed and accepted again, several times
|
|
160
|
+
// in one run, each pass paying the provider for an answer that could not change.
|
|
161
|
+
//
|
|
162
|
+
// `sticky_gaps` is the run's memory of them: every qid accepted with a reason that states a capability gap
|
|
163
|
+
// (isCapabilityGapReason), with the reason and the plan version it was first accepted under. It is carried
|
|
164
|
+
// forward on every settle and never dropped: not when the plan version moves, not when the receipt stops
|
|
165
|
+
// listing the qid. Only a REASON-MATCHED gap enters it. An axis-level acceptance (the skeleton branch in
|
|
166
|
+
// partitionReceiptDeferrals) and every suspect or close_failed row stay out, so a deferral a later attempt
|
|
167
|
+
// could close keeps today's behaviour.
|
|
168
|
+
export function stickyGapsAfter(prior, accepted, planVersion) {
|
|
169
|
+
const out = new Map((prior?.sticky_gaps ?? []).map((g) => [String(g?.qid ?? ""), g]));
|
|
170
|
+
for (const a of accepted ?? []) {
|
|
171
|
+
const qid = String(a?.qid ?? "");
|
|
172
|
+
if (!qid || out.has(qid) || !isCapabilityGapReason(a?.reason)) continue;
|
|
173
|
+
out.set(qid, { qid, axis: String(a?.axis ?? ""), reason: String(a?.reason ?? "").slice(0, 400), since_plan_version: planVersion ?? null });
|
|
174
|
+
}
|
|
175
|
+
out.delete("");
|
|
176
|
+
return [...out.values()];
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
// ── A SUSPECT DEFERRAL GETS ITS ONE ATTEMPT ONCE PER RUN ─────────────────────────────────────────────
|
|
180
|
+
//
|
|
181
|
+
// The settle step spends one executor attempt on a suspect deferral: a hard error the run could not
|
|
182
|
+
// classify as permanent, on the chance the provider has recovered. Every supplemental fold bumps the plan
|
|
183
|
+
// version and re-settles, and the repair ledger's budget is keyed on the plan version, so "one attempt"
|
|
184
|
+
// was one per plan version. `settle_tried` records each qid the settle step has sent, carried forward like
|
|
185
|
+
// `sticky_gaps`; a suspect already in it is recorded close_failed without another call.
|
|
186
|
+
export function settleTriedAfter(prior, dispatched, planVersion) {
|
|
187
|
+
const out = new Map((prior?.settle_tried ?? []).map((t) => [String(t?.qid ?? ""), t]));
|
|
188
|
+
for (const d of dispatched ?? []) {
|
|
189
|
+
const qid = String(d?.qid ?? "");
|
|
190
|
+
if (!qid || out.has(qid)) continue;
|
|
191
|
+
out.set(qid, { qid, axis: String(d?.axis ?? ""), plan_version: planVersion ?? null, outcome: String(d?.outcome ?? "").slice(0, 160) });
|
|
192
|
+
}
|
|
193
|
+
out.delete("");
|
|
194
|
+
return [...out.values()];
|
|
195
|
+
}
|
|
196
|
+
|
|
197
|
+
/** The run's sticky capability gaps, by qid → row. Absent or unreadable decision ⇒ none. */
|
|
198
|
+
export function readStickyGaps(P) {
|
|
199
|
+
return new Map((readEnvelopeDecision(P)?.sticky_gaps ?? []).map((g) => [String(g?.qid ?? ""), g]).filter(([q]) => q));
|
|
200
|
+
}
|
|
201
|
+
|
|
145
202
|
/** One history entry per superseded decision. Returns the prior history with the old decision appended. */
|
|
146
203
|
export function supersede(prior, source) {
|
|
147
204
|
if (!prior) return [];
|
|
@@ -167,30 +224,42 @@ export function supersede(prior, source) {
|
|
|
167
224
|
*/
|
|
168
225
|
export async function settleReceipt({ P, plan, receipt, dispatch = null, rejoin = null, now = null }) {
|
|
169
226
|
const deferred = receipt?.deferred ?? [];
|
|
170
|
-
|
|
227
|
+
const prior = readEnvelopeDecision(P);
|
|
228
|
+
const sticky = new Set((prior?.sticky_gaps ?? []).map((g) => String(g?.qid ?? "")));
|
|
229
|
+
let { accepted, suspect } = partitionReceiptDeferrals(plan, receipt, { sticky });
|
|
171
230
|
const closed = [], closeFailed = [];
|
|
231
|
+
const tried = new Map((prior?.settle_tried ?? []).map((t) => [String(t?.qid ?? ""), t]));
|
|
232
|
+
for (const s of suspect.filter((x) => tried.has(x.qid)))
|
|
233
|
+
closeFailed.push({ qid: s.qid, axis: s.axis, outcome: `already tried once this run (${String(tried.get(s.qid)?.outcome ?? "").slice(0, 100)}) — not sent again` });
|
|
234
|
+
suspect = suspect.filter((x) => !tried.has(x.qid));
|
|
235
|
+
const dispatched = [];
|
|
172
236
|
if (suspect.length && dispatch && rejoin) {
|
|
173
237
|
const byAxis = new Map();
|
|
174
238
|
for (const s of suspect) { if (!byAxis.has(s.axis)) byAxis.set(s.axis, []); byAxis.get(s.axis).push(s.qid); }
|
|
175
239
|
for (const [axis, qids] of byAxis) { if (axis) await dispatch(axis, qids); }
|
|
176
240
|
const after = await rejoin();
|
|
177
241
|
const stillDeferred = new Map((after?.deferred ?? []).map((d) => [String(d.qid), String(d.reason ?? "")]));
|
|
178
|
-
const re = partitionReceiptDeferrals(plan, after);
|
|
242
|
+
const re = partitionReceiptDeferrals(plan, after, { sticky });
|
|
179
243
|
accepted = re.accepted;
|
|
180
244
|
for (const s of suspect) {
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
245
|
+
// Only a qid that was actually sent (dispatch skips one with no axis) counts as tried.
|
|
246
|
+
if (!stillDeferred.has(s.qid)) { closed.push({ qid: s.qid, axis: s.axis, outcome: "ok" }); if (s.axis) dispatched.push({ ...s, outcome: "ok" }); }
|
|
247
|
+
else if (!re.accepted.some((a) => a.qid === s.qid)) {
|
|
248
|
+
const outcome = `still deferred: ${stillDeferred.get(s.qid).slice(0, 140)}`;
|
|
249
|
+
closeFailed.push({ qid: s.qid, axis: s.axis, outcome });
|
|
250
|
+
if (s.axis) dispatched.push({ ...s, outcome });
|
|
251
|
+
}
|
|
184
252
|
}
|
|
185
253
|
} else if (suspect.length) {
|
|
186
254
|
for (const s of suspect) closeFailed.push({ qid: s.qid, axis: s.axis, outcome: "no executor lane at this seam" });
|
|
187
255
|
}
|
|
188
|
-
const prior = readEnvelopeDecision(P);
|
|
189
256
|
return writeEnvelopeDecision(P, buildDecisionDoc({
|
|
190
257
|
planVersion: receipt?.plan_version, deferredTotal: deferred.length,
|
|
191
258
|
accepted: accepted.map((a) => ({ ...a, decision: "accepted-capability-gap" })),
|
|
192
259
|
closed, closeFailed, decidedAt: now ?? new Date().toISOString(),
|
|
193
260
|
history: supersede(prior, prior ? "re-settled" : null),
|
|
261
|
+
stickyGaps: stickyGapsAfter(prior, accepted, receipt?.plan_version),
|
|
262
|
+
settleTried: settleTriedAfter(prior, dispatched, receipt?.plan_version),
|
|
194
263
|
}));
|
|
195
264
|
}
|
|
196
265
|
|
|
@@ -335,14 +335,14 @@ export function isUnconditionalProceed(rec) {
|
|
|
335
335
|
|
|
336
336
|
/**
|
|
337
337
|
* Deterministically BIND a recommendation line to the verdict: on a CONDITIONAL verdict an
|
|
338
|
-
* unconditional "proceed" is rewritten to carry the conditions
|
|
339
|
-
*
|
|
338
|
+
* unconditional "proceed" is rewritten to carry the conditions. BLOCKING is the reviewer's sign-off on
|
|
339
|
+
* the draft, not a hold: a BLOCKING run delivers, so its recommendation stands as the report wrote it.
|
|
340
340
|
* Code derives the bound from the model's OWN verdict + reasons; it never invents a verdict. PURE.
|
|
341
341
|
*/
|
|
342
342
|
export function bindRecommendation(rec, verdict, reasons = [], { maxReasons, maxLen } = {}) {
|
|
343
343
|
const r = String(rec ?? "").trim();
|
|
344
344
|
const v = String(verdict || "").toUpperCase();
|
|
345
|
-
|
|
345
|
+
// BLOCKING returns the recommendation as written: nothing holds delivery on it, so "on hold" was untrue (ruled 2026-09-22).
|
|
346
346
|
if (v !== "CONDITIONAL" || !r || !isUnconditionalProceed(r)) return r;
|
|
347
347
|
const conds = (reasons ?? []).filter(Boolean);
|
|
348
348
|
if (!conds.length) return r;
|
|
@@ -600,7 +600,7 @@ export function riskStatement({ tier, verdict, reasons, basis, clauses } = {}) {
|
|
|
600
600
|
// CLEAR, so long as it is clear on register findings ALONE).
|
|
601
601
|
const registerOnly = basis === "register-only";
|
|
602
602
|
const basisNote = registerOnly ? " Register findings only — no common-law or marketplace search was run." : "";
|
|
603
|
-
if (v === "BLOCKING") return `
|
|
603
|
+
if (v === "BLOCKING") return `${t}${basisNote ? `.${basisNote}` : ""}`; // the reviewer's sign-off is not the clearance's answer and nothing is held on it: the band stands alone, as below
|
|
604
604
|
if (v === "CONDITIONAL") {
|
|
605
605
|
// A clause stored as explicit null is a condition ruled to the run record alone: it is not the lede,
|
|
606
606
|
// it is not counted, and its reason is never the fallback text. `undefined` (legacy/short) is not null.
|
package/driver/gateway.mjs
CHANGED
|
@@ -225,6 +225,8 @@ export function toolWrittenArtifact(p) {
|
|
|
225
225
|
?? null;
|
|
226
226
|
}
|
|
227
227
|
import { buildGatherMcpConfig, allowedToolsFor, toolGroupsForStage, recordAxisFor, seatWritesForGroups } from "./engine/mcp/gather-config.mjs";
|
|
228
|
+
import { everyToolCallRefused } from "./engine/tool-refusal.mjs"; // one definition, shared with the engine probe
|
|
229
|
+
export { everyToolCallRefused };
|
|
228
230
|
// The profiles STORE root, for the write boundary only — read from the module that owns it, never
|
|
229
231
|
// re-derived from CLEAROTRON_CUSTOMERS_DIR here. Acyclic: profiles.mjs imports node builtins + config.
|
|
230
232
|
import { unitRefusalsFor } from "./register-unit-record.mjs";
|
|
@@ -1528,6 +1530,11 @@ async function runStageLadder(name, opts, stageCodexHome = null) {
|
|
|
1528
1530
|
// appear: a postponed run resumes on its own, and a signed-out one cannot. The sentence rides after
|
|
1529
1531
|
// the colon, as `model_mismatch:` carries its detail, so no run-level classifier needs a new token.
|
|
1530
1532
|
else if (turn.signals?.signedOut && fail) fail = `engine_signed_out: ${turn.signals.signedOut}`;
|
|
1533
|
+
// AND ONE WHOSE TOOLS WERE ALL REFUSED IS NAMED, not left as the missing file the refusal caused. The
|
|
1534
|
+
// turn reported success and the stage failed for want of what its tools would have produced; a
|
|
1535
|
+
// `missing_file` reads as a model that did not write, and sends the reader to the wrong place. The
|
|
1536
|
+
// sentence says what fixes this host. After the sign-in, which is the more basic fault when both appear.
|
|
1537
|
+
else if (turn.signals?.toolsRefused && fail) fail = `engine_tools_refused: ${turn.signals.toolsRefused}`;
|
|
1531
1538
|
// A6 (addendum 2026-07-30): stop_reason max_tokens with ZERO usable output is a DETECTED FAULT with a
|
|
1532
1539
|
// name — never a silent paid retry. The turn ran to its output-token ceiling and the artifact never
|
|
1533
1540
|
// landed (a content fail on a "successful" turn), or the turn itself died at the ceiling (transport
|
|
@@ -2027,6 +2034,15 @@ async function runStageLadder(name, opts, stageCodexHome = null) {
|
|
|
2027
2034
|
note(`[${name}] the engine's sign-in could not be refreshed — breaking the ladder (a retry re-sends the same credential)`);
|
|
2028
2035
|
break;
|
|
2029
2036
|
}
|
|
2037
|
+
// EVERY TOOL CALL THE TURN MADE WAS REFUSED. On some hosts codex's own sandbox refuses the stage's
|
|
2038
|
+
// tool servers while the turn reports success, and the stage then fails for want of what the tools
|
|
2039
|
+
// would have produced. The next attempt spawns the same sandbox and is refused the same way, so a
|
|
2040
|
+
// retry re-buys the refusal: measured at three paid attempts per stage on a production run before the
|
|
2041
|
+
// identical-failure break stopped it. Break on the first. A turn where any call completed is not this.
|
|
2042
|
+
if (everyToolCallRefused(turn)) {
|
|
2043
|
+
note(`[${name}] every tool call this turn made was refused (${turn.mcpToolCallsRefused}) — breaking the ladder (a retry is refused the same way)`);
|
|
2044
|
+
break;
|
|
2045
|
+
}
|
|
2030
2046
|
// D3: overload (529 / status_overloaded) — stop the ladder here: an in-ladder re-attempt hammers an
|
|
2031
2047
|
// API that just said it is overloaded, seconds apart, on the SAME model. Breaking hands the failure
|
|
2032
2048
|
// to the existing machinery: the chain cascades models (FALLBACK_ELIGIBLE) and an exhausted chain
|
|
@@ -2816,7 +2832,7 @@ export function correctionHint(lastFail, { gridLedgerName = "common-law-grid.jso
|
|
|
2816
2832
|
`The driver computed every obligation and every identifier in it — the coverage unit, the query id, the hit ` +
|
|
2817
2833
|
`count, the unaccounted classes and terms, each deferred slice's own receipt reason. Record the named row(s) ` +
|
|
2818
2834
|
`through the \`record_coverage\` tool — {"row_id","status","reason"} per row, never by writing or editing any ` +
|
|
2819
|
-
`file: "status" EXACTLY one bare token of confirmed-clean / coverage-limited / deferred, "reason" the sentence ` +
|
|
2835
|
+
`file: "status" EXACTLY one bare token of confirmed-clean / coverage-limited / deferred / withheld-by-judgment, "reason" the sentence ` +
|
|
2820
2836
|
`the lawyer reads (qualifiers go in the reason, never in the status). ` +
|
|
2821
2837
|
`A row marked "open" cannot be confirmed-clean, and its own "open_because" says which of the two kinds ` +
|
|
2822
2838
|
`it is. A NEVER-SEARCHED slice — the active register provider cannot express it, so nothing can make it run — is ` +
|
|
@@ -2983,7 +2999,7 @@ export function correctionHint(lastFail, { gridLedgerName = "common-law-grid.jso
|
|
|
2983
2999
|
// run the driver stamped as form-required, and the stamp is conditional. On an unstamped run
|
|
2984
3000
|
// validators.registerFindings demands the table exactly as it did before and emits this label,
|
|
2985
3001
|
// so dropping the arm left the one lane that can still fire it with a generic hint.
|
|
2986
|
-
hint = "the file has a findings heading plus a Coverage ledger with a status row (confirmed-clean / coverage-limited / deferred)";
|
|
3002
|
+
hint = "the file has a findings heading plus a Coverage ledger with a status row (confirmed-clean / coverage-limited / deferred)" + (/common-law-findings/.test(lastFail) ? ", or each ledger row's status is recorded by calling `record_coverage_status` with `grid_spec_path`, the same spec path the grid tool was given" : "");
|
|
2987
3003
|
} else if (/negative-results|coverage-ledger|audit-trail|findings-heading/.test(lastFail)) {
|
|
2988
3004
|
hint = "the findings file carries ALL required sections: a findings heading, the Negative results matrix " +
|
|
2989
3005
|
"(every variant × platform row), the Coverage ledger with a status row, and the Audit trail call log";
|
|
@@ -47,7 +47,7 @@ export function managerGroupsVerdict({ idGroups, managerGroups, user, uid, why =
|
|
|
47
47
|
// was not established", and saying so is the whole point — the deployment that HAD this fault also
|
|
48
48
|
// had a check that reported nothing wrong.
|
|
49
49
|
if (!Array.isArray(idGroups) || !idGroups.length) {
|
|
50
|
-
return { state: "skip", message: `could not read the groups of ${user}${why ? ` — ${why}` : ""}. Not checked, not passed.` };
|
|
50
|
+
return { state: "skip", blocked: true, message: `could not read the groups of ${user}${why ? ` — ${why}` : ""}. Not checked, not passed.` };
|
|
51
51
|
}
|
|
52
52
|
if (!Array.isArray(managerGroups)) {
|
|
53
53
|
return { state: "skip", message: `no readable systemd --user manager for ${user}${why ? ` — ${why}` : ""}. `
|
|
@@ -195,22 +195,39 @@ export const refuseUndeclared = (params) => refuseUndeclaredShared(params, DECLA
|
|
|
195
195
|
* The Nice classes the frame judged necessary beyond the instructed ones, as strings. IMPURE (reads
|
|
196
196
|
* the run's own accepted call).
|
|
197
197
|
*
|
|
198
|
-
* THE PLAN COMPILE
|
|
199
|
-
*
|
|
200
|
-
*
|
|
201
|
-
*
|
|
202
|
-
* the run this came from the delivered report said two such classes were "covered for the name and open
|
|
203
|
-
* for its variants" while the reviewing lawyer's scope included one of them throughout.
|
|
198
|
+
* THE PLAN COMPILE DOES NOT READ THIS. It takes `frameIdentifiedClassRows` below, where each added
|
|
199
|
+
* class keeps its reason and costs one identical-mark question rather than riding every entry (decision
|
|
200
|
+
* 18). The one caller left is the house-element check, which unions this list with the instructed
|
|
201
|
+
* classes because the client's own filings are best looked for across the widest class set.
|
|
204
202
|
*
|
|
205
203
|
* EMPTY IS THE ORDINARY ANSWER and must stay cheap: no frame yet, a legacy or replayed run whose
|
|
206
204
|
* accepted call predates the field, a frame that identified nothing — all of them return `[]`, the
|
|
207
|
-
* union is a no-op, and the
|
|
205
|
+
* union is a no-op, and the check looks in the instructed classes alone.
|
|
208
206
|
*/
|
|
209
207
|
export function frameIdentifiedClasses(runDir) {
|
|
210
208
|
const rows = lastAcceptedMatterFrame(runDir)?.identified_classes;
|
|
211
209
|
return (Array.isArray(rows) ? rows : []).map((r) => String(r?.class ?? "").trim()).filter(Boolean);
|
|
212
210
|
}
|
|
213
211
|
|
|
212
|
+
/**
|
|
213
|
+
* The same rows WITH THEIR REASONS, for the register plan (decision 18). IMPURE, like its sibling.
|
|
214
|
+
*
|
|
215
|
+
* The reason is why the class is here and it is half the row: the bound on this widening is that a
|
|
216
|
+
* class is added only for goods the CLIENT'S OWN business plainly reaches, and a class carrying no
|
|
217
|
+
* stated reason cannot be checked against that by anyone. The plan compiler drops such a row rather
|
|
218
|
+
* than searching it, so the reason is load-bearing and not documentation.
|
|
219
|
+
*
|
|
220
|
+
* Kept beside `frameIdentifiedClasses` rather than replacing it: the other caller verifies a proposed
|
|
221
|
+
* house element against an owner-scoped lookup, where the widest class set is the right one to look in
|
|
222
|
+
* and a reason would mean nothing.
|
|
223
|
+
*/
|
|
224
|
+
export function frameIdentifiedClassRows(runDir) {
|
|
225
|
+
const rows = lastAcceptedMatterFrame(runDir)?.identified_classes;
|
|
226
|
+
return (Array.isArray(rows) ? rows : [])
|
|
227
|
+
.map((r) => ({ class: String(r?.class ?? "").trim(), reason: String(r?.reason ?? "").trim() }))
|
|
228
|
+
.filter((r) => r.class);
|
|
229
|
+
}
|
|
230
|
+
|
|
214
231
|
/**
|
|
215
232
|
* The forms of the name this run's client ratified, as strings. IMPURE (reads the run's own accepted
|
|
216
233
|
* call). Empty or one form is the ordinary answer.
|
package/driver/named-band.mjs
CHANGED
|
@@ -137,7 +137,7 @@ export function parseNamedBand(raw) {
|
|
|
137
137
|
// byte-identical to a slice the plan deliberately counted without fetching. Measured on a real
|
|
138
138
|
// run: four capability-gap blocks carried `error:true, deferred:true` into this function and
|
|
139
139
|
// reached record-carry.json with both fields gone and a sentence claiming the run "has a hit
|
|
140
|
-
// COUNT for this slice". register-plan.mjs
|
|
140
|
+
// COUNT for this slice". register-plan.mjs validatePlanFeasibility already enforces the same rule one layer up
|
|
141
141
|
// ("a transient must not ship indistinguishable from a sanctioned descriptor") — it reads the
|
|
142
142
|
// RAW blocks, which is why it could. Every consumer that reads THIS projection could not.
|
|
143
143
|
// Conditional like the four keys above, so old bands carry neither key and nothing shifts.
|