ruvnet-brain 4.3.21 → 4.3.25
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +5 -5
- package/bin/install.mjs +275 -60
- package/console/app.js +141 -9
- package/console/index.html +51 -24
- package/console/scope.css +137 -0
- package/console/scope.html +144 -0
- package/console/scope.js +209 -0
- package/console/tips.html +1 -0
- package/kb/corpus-release-identity.mjs +239 -0
- package/kb/update-storage-transaction.mjs +20 -3
- package/package.json +9 -2
- package/plugin/.claude-plugin/plugin.json +2 -2
- package/plugin/.codex-plugin/plugin.json +1 -1
- package/plugin/commands/checkpoint.md +61 -0
- package/plugin/hooks/codex-hooks.json +64 -1
- package/plugin/hooks/hook-contracts.json +299 -6
- package/plugin/hooks/hooks.json +81 -1
- package/plugin/mcp/server.mjs +23 -0
- package/plugin/scripts/advocacy-catalog.mjs +245 -0
- package/plugin/scripts/advocacy-route.mjs +460 -0
- package/plugin/scripts/continuation-gate.mjs +25 -2
- package/plugin/scripts/continuation-objective.mjs +7 -1
- package/plugin/scripts/continuity-hook-policy.mjs +190 -15
- package/plugin/scripts/coverage-integrity.mjs +7 -0
- package/plugin/scripts/gates.mjs +113 -10
- package/plugin/scripts/grounding-turn-gate.mjs +167 -0
- package/plugin/scripts/grounding-turn-mark.mjs +91 -0
- package/plugin/scripts/hook-shim.mjs +14 -0
- package/plugin/scripts/nightly-scheduler.mjs +37 -4
- package/plugin/scripts/project-progression-checkpoint.mjs +145 -0
- package/plugin/scripts/project-progression-contract.mjs +16 -0
- package/plugin/scripts/project-progression-hook.mjs +3 -0
- package/plugin/scripts/project-progression-producer.mjs +252 -0
- package/plugin/scripts/project-progression-reader.mjs +271 -0
- package/plugin/scripts/project-progression-session-start.mjs +93 -16
- package/plugin/scripts/project-progression-sources.mjs +220 -0
- package/plugin/scripts/project-progression-store.mjs +106 -13
- package/plugin/scripts/ruvnet-gate1-pattern.mjs +29 -0
- package/plugin/scripts/session-snapshot-hook.mjs +115 -7
- package/plugin/scripts/session-start-budget.mjs +59 -0
- package/plugin/scripts/session-start-core.mjs +234 -457
- package/plugin/scripts/session-start-fsutil.mjs +61 -0
- package/plugin/scripts/session-start-health.mjs +64 -0
- package/plugin/scripts/session-start-hook-description.mjs +45 -0
- package/plugin/scripts/session-start-issue-alert.mjs +77 -0
- package/plugin/scripts/session-start-repo-identity.mjs +54 -0
- package/plugin/scripts/session-start-signals.mjs +73 -0
- package/plugin/scripts/session-start-trace.mjs +86 -0
- package/plugin/scripts/session-start-update-plane.mjs +104 -0
- package/plugin/scripts/unprompted-runtime.mjs +32 -2
- package/plugin/skills/ruvnet-brain/PLAYBOOK.md +26 -2
- package/plugin/skills/ruvnet-brain/SKILL.md +67 -2
- package/scripts/adr-072-completion.mjs +1 -1
- package/scripts/agentdb-fleet-doctor.mjs +5 -1
- package/scripts/approved-runtime.mjs +197 -0
- package/scripts/brain-novice-50.mjs +16 -1
- package/scripts/brain-score.mjs +23 -5
- package/scripts/build-bundle.mjs +971 -530
- package/scripts/build-concepts.mjs +36 -116
- package/scripts/console-engine.test.mjs +8 -7
- package/scripts/console-runtime-identity.mjs +4 -0
- package/scripts/corpus-aggregates.mjs +94 -77
- package/scripts/corpus-candidate.mjs +475 -222
- package/scripts/corpus-next-seed.mjs +225 -0
- package/scripts/corpus-promotion.mjs +58 -0
- package/scripts/corpus-reconcile.mjs +411 -105
- package/scripts/doc-currency.mjs +16 -1
- package/scripts/dual-host-deliberation.mjs +25 -2
- package/scripts/dual-host-suggest.mjs +17 -1
- package/scripts/falsify.mjs +13 -3
- package/scripts/gist-receipts.mjs +482 -87
- package/scripts/github-health-watch.mjs +12 -2
- package/scripts/handoff-asset.mjs +34 -0
- package/scripts/hook-retirement-check.mjs +8 -1
- package/scripts/host-registry.mjs +1 -1
- package/scripts/ingest-gists.mjs +74 -101
- package/scripts/job-heartbeat.sh +77 -14
- package/scripts/learning-replay-execution.mjs +10 -4
- package/scripts/nightly-gists.sh +27 -13
- package/scripts/nightly-two-run-proof.mjs +1 -1
- package/scripts/nightly-watchdog.mjs +61 -4
- package/scripts/onboarding-console.mjs +319 -27
- package/scripts/oracle/produce-questions.mjs +293 -0
- package/scripts/oracle/producer-hosts.mjs +235 -0
- package/scripts/oracle/repo-recall.mjs +448 -0
- package/scripts/oracle/retrieval-accuracy.mjs +818 -0
- package/scripts/oracle/source-tree.mjs +165 -0
- package/scripts/oracle/source-units.mjs +391 -0
- package/scripts/oracle/spike-run.mjs +98 -0
- package/scripts/oracle/unit-inventory.mjs +141 -0
- package/scripts/oracle/unit-sampling.mjs +128 -0
- package/scripts/oracle/validate-labels.mjs +250 -0
- package/scripts/private-overlay.mjs +248 -0
- package/scripts/product-integrity-contract.mjs +1 -1
- package/scripts/proxy/claude-proxied.sh +6 -0
- package/scripts/proxy/proxy-revert.sh +5 -0
- package/scripts/proxy/proxy-up.sh +6 -0
- package/scripts/proxy/proxy-verify.mjs +4 -0
- package/scripts/public-inputs.mjs +409 -0
- package/scripts/public-verification-inputs.mjs +112 -26
- package/scripts/public-verification-lane.mjs +1 -1
- package/scripts/published-surface-probe.mjs +34 -4
- package/scripts/qe/card-lane-gate.mjs +16 -1
- package/scripts/qe/session-start-gate.mjs +16 -1
- package/scripts/rebuild-gists-from-receipts.mjs +58 -78
- package/scripts/record-lesson.mjs +4 -1
- package/scripts/rehearse-corpus-pipeline.mjs +994 -0
- package/scripts/release-abort-stale.mjs +5 -1
- package/scripts/release-authority.mjs +104 -12
- package/scripts/release-channel-kind.mjs +86 -0
- package/scripts/release-convergence-watchdog.mjs +7 -2
- package/scripts/release-projection.mjs +177 -72
- package/scripts/release-transaction-provider.mjs +23 -6
- package/scripts/release.mjs +252 -17
- package/scripts/retrieval-canary.mjs +87 -0
- package/scripts/rvf-index-audit.mjs +573 -13
- package/scripts/rvf-wire.mjs +269 -0
- package/scripts/seal-gist-receipt.mjs +65 -0
- package/scripts/selfcheck.mjs +42 -21
- package/scripts/source-coverage.mjs +253 -24
- package/scripts/status-honesty.mjs +25 -0
- package/scripts/sync-census.mjs +0 -0
- package/scripts/sync-version.mjs +2 -0
- package/scripts/trismart.mjs +42 -0
- package/scripts/updater-manifest.mjs +162 -0
- package/scripts/verify-channels.mjs +17 -5
- package/scripts/wired-check.mjs +48 -10
- package/tri-smart-skill/QUICKSTART.md +37 -0
- package/tri-smart-skill/README.md +92 -0
- package/tri-smart-skill/install.cmd +14 -0
- package/tri-smart-skill/install.command +13 -0
- package/tri-smart-skill/install.mjs +51 -0
- package/tri-smart-skill/install.sh +9 -0
- package/tri-smart-skill/tri-smart/SKILL.md +90 -0
- package/tri-smart-skill/tri-smart/evals/evals.json +25 -0
- package/tri-smart-skill/tri-smart/references/protocol.md +25 -0
- package/tri-smart-skill/tri-smart/references/provider-cli.md +18 -0
- package/tri-smart-skill/tri-smart/scripts/review.mjs +154 -0
- package/tri-smart-skill/tri-smart/scripts/setup.mjs +97 -0
- package/tri-smart-skill/tri-smart/scripts/verify-access.mjs +107 -0
- package/scripts/corpus-seed-publish.mjs +0 -110
|
@@ -0,0 +1,293 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* scripts/oracle/produce-questions.mjs — Step 14 (ADR-086 C3) SUBSCRIPTION-ONLY semantic label producer.
|
|
4
|
+
*
|
|
5
|
+
* Two independent passes over the units selected by source-units.mjs, both on subscription-billed
|
|
6
|
+
* native hosts and NEVER on a provider API key:
|
|
7
|
+
*
|
|
8
|
+
* 1. the GENERATOR sees the exact upstream bytes of the units in a batch and must emit, per unit, one
|
|
9
|
+
* direct question, one meaning-preserving paraphrase, and the VERBATIM supporting span (plus its
|
|
10
|
+
* line range inside the unit) as strict structured output;
|
|
11
|
+
* 2. the JUDGE — a DIFFERENT vendor — sees ONLY question + candidate span (never the unit, path or
|
|
12
|
+
* repo) and answers whether the span alone answers each question, and separately whether the
|
|
13
|
+
* direct question and its paraphrase mean the same thing.
|
|
14
|
+
*
|
|
15
|
+
* Roles default to claude generates / codex judges (the configuration the Step 14 spike measured).
|
|
16
|
+
* Path B — codex generates / claude judges — was approved by an Astra-only Dual deliberation on
|
|
17
|
+
* 2026-09-14 and must qualify on its own pilot; host details live in ./producer-hosts.mjs.
|
|
18
|
+
*
|
|
19
|
+
* Fence (owner mandate): the child environment is scripts/subscription-hosts.mjs#subscriptionOnlyEnv,
|
|
20
|
+
* which deletes every API_BILLING_ENV name; this module never reads those names itself (the unit test
|
|
21
|
+
* greps this file for them). `--bare` is never used — see producer-hosts.mjs.
|
|
22
|
+
*
|
|
23
|
+
* The producer sees upstream bytes only. Nothing here reads a candidate corpus, RVF, passage sidecar
|
|
24
|
+
* or retrieval output — by construction, not by promise (inputs: inventory JSON + snapshot dir).
|
|
25
|
+
*/
|
|
26
|
+
import crypto from 'node:crypto';
|
|
27
|
+
import fs from 'node:fs';
|
|
28
|
+
import os from 'node:os';
|
|
29
|
+
import path from 'node:path';
|
|
30
|
+
import { fileURLToPath } from 'node:url';
|
|
31
|
+
import { API_BILLING_ENV, subscriptionOnlyEnv } from '../subscription-hosts.mjs';
|
|
32
|
+
import { gitBlobSha, sha256Hex, unitText } from './source-units.mjs';
|
|
33
|
+
import {
|
|
34
|
+
DEFAULT_ROLES, LABEL_SCHEMA, PRODUCER_MODELS, VERDICT_SCHEMA, generatorPrompt, hostAdapters, isQuotaRefusal,
|
|
35
|
+
judgePrompt, spawnHost,
|
|
36
|
+
} from './producer-hosts.mjs';
|
|
37
|
+
|
|
38
|
+
export {
|
|
39
|
+
CLAUDE_SYSTEM_PROMPT, CLAUDE_TIMEOUT_MS, CODEX_TIMEOUT_MS, DEFAULT_ROLES, JUDGE_SYSTEM_PROMPT, LABEL_SCHEMA,
|
|
40
|
+
PRODUCER_MODELS, VERDICT_SCHEMA, claudeArgs, claudePrompt, codexArgs, codexPrompt, hostAdapters, isQuotaRefusal,
|
|
41
|
+
parseClaudeEnvelope, parseClaudeVerdicts, parseCodexJsonl, parseCodexLabels, spawnHost,
|
|
42
|
+
} from './producer-hosts.mjs';
|
|
43
|
+
|
|
44
|
+
export const PRODUCER_VERSION = 'oracle-producer/2';
|
|
45
|
+
export const DEFAULT_BATCH = 20;
|
|
46
|
+
export const MAX_UNIT_CHARS = 6000;
|
|
47
|
+
|
|
48
|
+
export function batchUnits(units, size = DEFAULT_BATCH) {
|
|
49
|
+
const out = [];
|
|
50
|
+
for (let i = 0; i < units.length; i += size) out.push(units.slice(i, i + size));
|
|
51
|
+
return out;
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
/** Re-read every selected unit from disk and refuse drift: blob and unit hashes must match the inventory. */
|
|
55
|
+
export function loadUnitTexts(snapshotDir, units) {
|
|
56
|
+
const cache = new Map();
|
|
57
|
+
return units.map((unit) => {
|
|
58
|
+
if (!cache.has(unit.path)) {
|
|
59
|
+
const buf = fs.readFileSync(path.join(snapshotDir, unit.path));
|
|
60
|
+
cache.set(unit.path, { blobSha: gitBlobSha(buf), lines: buf.toString('utf8').split('\n') });
|
|
61
|
+
}
|
|
62
|
+
const file = cache.get(unit.path);
|
|
63
|
+
if (file.blobSha !== unit.blobSha) throw new Error(`blob drift: ${unit.path} is ${file.blobSha}, inventory says ${unit.blobSha}`);
|
|
64
|
+
const text = unitText(file.lines, unit.startLine, unit.endLine);
|
|
65
|
+
if (sha256Hex(Buffer.from(text, 'utf8')) !== unit.bytesSha256) throw new Error(`unit drift: ${unit.unitId} ${unit.path}:${unit.startLine}-${unit.endLine}`);
|
|
66
|
+
const truncated = text.length > MAX_UNIT_CHARS;
|
|
67
|
+
return { unit, text: truncated ? text.slice(0, MAX_UNIT_CHARS) : text, truncated };
|
|
68
|
+
});
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
function billingNamesPresent(env) { return API_BILLING_ENV.filter((name) => name in env); }
|
|
72
|
+
|
|
73
|
+
/**
|
|
74
|
+
* A checkpoint is reusable only under the SAME producer configuration. blobSha alone is not enough
|
|
75
|
+
* (Dual): the key binds source, rules, roles, models, effort, batch size and the exact prompt text.
|
|
76
|
+
*/
|
|
77
|
+
export function checkpointKey({ inventory, roles, adapters, effort, batchSize }) {
|
|
78
|
+
return sha256Hex(Buffer.from(JSON.stringify({
|
|
79
|
+
producerVersion: PRODUCER_VERSION, repo: inventory.repo, commit: inventory.commit, rulesVersion: inventory.rulesVersion,
|
|
80
|
+
roles, generatorModel: adapters.generator.model, judgeModel: adapters.judge.model, effort, batchSize,
|
|
81
|
+
generatorPromptSha256: sha256Hex(Buffer.from(generatorPrompt([]), 'utf8')),
|
|
82
|
+
judgePromptSha256: sha256Hex(Buffer.from(judgePrompt([{ id: 'x', questionA: 'a', questionB: 'b' }]), 'utf8')),
|
|
83
|
+
}), 'utf8'));
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
export async function produceQuestions({
|
|
87
|
+
inventory, snapshotDir, batchSize = DEFAULT_BATCH, maxClaudeCalls = 8, maxCodexCalls = 8, maxGeneratorCalls, maxJudgeCalls,
|
|
88
|
+
effort = 'medium', models = PRODUCER_MODELS, roles = DEFAULT_ROLES, allowSameVendor = false, retries = 0,
|
|
89
|
+
checkpointFile = null, spawnImpl = spawnHost, workDir = fs.mkdtempSync(path.join(os.tmpdir(), 'oracle-producer-')),
|
|
90
|
+
log = () => {},
|
|
91
|
+
}) {
|
|
92
|
+
const env = subscriptionOnlyEnv();
|
|
93
|
+
const envAudit = { parentHadBillingKeys: billingNamesPresent(process.env), childHadBillingKeys: billingNamesPresent(env) };
|
|
94
|
+
if (envAudit.childHadBillingKeys.length) throw new Error(`fence violated: child env still carries ${envAudit.childHadBillingKeys.join(',')}`);
|
|
95
|
+
fs.mkdirSync(workDir, { recursive: true }); // an explicitly-supplied workDir may not exist yet
|
|
96
|
+
const schemaFiles = { labels: path.join(workDir, 'label-schema.json'), verdicts: path.join(workDir, 'verdict-schema.json') };
|
|
97
|
+
fs.writeFileSync(schemaFiles.labels, JSON.stringify(LABEL_SCHEMA));
|
|
98
|
+
fs.writeFileSync(schemaFiles.verdicts, JSON.stringify(VERDICT_SCHEMA));
|
|
99
|
+
const codexCwd = path.join(workDir, 'codex-empty-cwd');
|
|
100
|
+
fs.mkdirSync(codexCwd, { recursive: true });
|
|
101
|
+
const adapters = hostAdapters({ roles, models, effort, schemaFiles, workDir, codexCwd, allowSameVendor });
|
|
102
|
+
const { generator, judge } = adapters;
|
|
103
|
+
const budgetFor = (host, explicit) => explicit ?? (host === 'claude' ? maxClaudeCalls : maxCodexCalls);
|
|
104
|
+
const generatorBudget = budgetFor(generator.host, maxGeneratorCalls);
|
|
105
|
+
const judgeBudget = budgetFor(judge.host, maxJudgeCalls);
|
|
106
|
+
|
|
107
|
+
const key = checkpointKey({ inventory, roles, adapters, effort, batchSize });
|
|
108
|
+
const reused = new Map();
|
|
109
|
+
if (checkpointFile && fs.existsSync(checkpointFile)) {
|
|
110
|
+
const saved = JSON.parse(fs.readFileSync(checkpointFile, 'utf8'));
|
|
111
|
+
if (saved.key === key) {
|
|
112
|
+
for (const l of saved.labels || []) if (!l.producerError) reused.set(l.unitId, l);
|
|
113
|
+
}
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
const loaded = loadUnitTexts(snapshotDir, inventory.selected);
|
|
117
|
+
const byUnit = new Map();
|
|
118
|
+
for (const { unit } of loaded) {
|
|
119
|
+
const prior = reused.get(unit.unitId);
|
|
120
|
+
if (prior && prior.path === unit.path && prior.blobSha === unit.blobSha && prior.bytesSha256 === unit.bytesSha256) byUnit.set(unit.unitId, prior);
|
|
121
|
+
}
|
|
122
|
+
const reusedUnits = byUnit.size;
|
|
123
|
+
const calls = [];
|
|
124
|
+
let suspended = null;
|
|
125
|
+
const ordered = () => loaded.map(({ unit }) => byUnit.get(unit.unitId)).filter(Boolean);
|
|
126
|
+
const saveCheckpoint = () => {
|
|
127
|
+
if (checkpointFile) fs.writeFileSync(checkpointFile, `${JSON.stringify({ key, producerVersion: PRODUCER_VERSION, labels: ordered(), suspended })}\n`);
|
|
128
|
+
};
|
|
129
|
+
|
|
130
|
+
const runCall = async (adapter, batchIndex, items, input) => {
|
|
131
|
+
for (let attempt = 0; ; attempt += 1) {
|
|
132
|
+
const result = await spawnImpl(adapter.binary, adapter.args, { cwd: adapter.cwd, env, timeoutMs: adapter.timeoutMs }, input);
|
|
133
|
+
const parsed = result.status === 0 && !result.timedOut
|
|
134
|
+
? adapter.parse(result.stdout)
|
|
135
|
+
: { error: result.timedOut ? 'timeout' : `exit ${result.status}: ${String(result.stderr || '').slice(0, 300)}` };
|
|
136
|
+
const quotaRefusal = Boolean(parsed.error) && isQuotaRefusal(`${result.stderr}\n${parsed.error}`);
|
|
137
|
+
calls.push(callRecord(adapter, batchIndex, items, result, parsed, attempt, quotaRefusal));
|
|
138
|
+
if (quotaRefusal) {
|
|
139
|
+
// Suspend: no further calls to ANY host. Remaining work is recorded as unproduced, never dropped.
|
|
140
|
+
suspended = { host: adapter.host, stage: adapter.stage, batchIndex, reason: String(result.stderr || parsed.error).slice(-300) };
|
|
141
|
+
return parsed;
|
|
142
|
+
}
|
|
143
|
+
if (!parsed.error || attempt >= retries) return parsed;
|
|
144
|
+
log(`[producer] ${adapter.host} ${adapter.stage} batch ${batchIndex + 1} retry ${attempt + 1}/${retries} after: ${parsed.error}`);
|
|
145
|
+
}
|
|
146
|
+
};
|
|
147
|
+
|
|
148
|
+
const pending = loaded.filter(({ unit }) => !byUnit.has(unit.unitId));
|
|
149
|
+
for (const [index, batch] of batchUnits(pending, batchSize).entries()) {
|
|
150
|
+
const stop = suspended ? `suspended: ${suspended.host} refused for capacity`
|
|
151
|
+
: index >= generatorBudget ? `${generator.host} call budget exhausted` : null;
|
|
152
|
+
if (stop) { for (const { unit } of batch) byUnit.set(unit.unitId, baseLabel(unit, { producerError: stop })); continue; }
|
|
153
|
+
log(`[producer] ${generator.host} generator batch ${index + 1} (${batch.length} units)`);
|
|
154
|
+
const parsed = await runCall(generator, index, batch.length, generator.prompt(batch));
|
|
155
|
+
if (parsed.error) {
|
|
156
|
+
for (const { unit } of batch) byUnit.set(unit.unitId, baseLabel(unit, { producerError: parsed.error }));
|
|
157
|
+
} else {
|
|
158
|
+
const returned = new Map(parsed.structured.labels.map((l) => [l.unitId, l]));
|
|
159
|
+
for (const { unit, truncated } of batch) {
|
|
160
|
+
const l = returned.get(unit.unitId);
|
|
161
|
+
byUnit.set(unit.unitId, l ? baseLabel(unit, {
|
|
162
|
+
direct: l.direct, paraphrase: l.paraphrase, span: l.span, spanStartLine: l.spanStartLine, spanEndLine: l.spanEndLine,
|
|
163
|
+
skip: l.skip === true, skipReason: l.skip ? l.reason : '', truncatedForProducer: truncated,
|
|
164
|
+
}) : baseLabel(unit, { producerError: `unit missing from ${generator.host} output` }));
|
|
165
|
+
}
|
|
166
|
+
}
|
|
167
|
+
saveCheckpoint();
|
|
168
|
+
}
|
|
169
|
+
|
|
170
|
+
const unjudged = ordered().filter((l) => !l.producerError && !l.skip && !l.judge);
|
|
171
|
+
for (const [index, batch] of batchUnits(unjudged, batchSize).entries()) {
|
|
172
|
+
const stop = suspended ? 'suspended' : index >= judgeBudget ? `${judge.host} call budget exhausted` : null;
|
|
173
|
+
if (stop) { for (const l of batch) setVerdicts(l, judge, { error: stop }); continue; }
|
|
174
|
+
const items = batch.flatMap((l) => [
|
|
175
|
+
{ id: `${l.unitId}:d`, question: l.direct, span: l.span },
|
|
176
|
+
{ id: `${l.unitId}:p`, question: l.paraphrase, span: l.span },
|
|
177
|
+
{ id: `${l.unitId}:e`, questionA: l.direct, questionB: l.paraphrase },
|
|
178
|
+
]);
|
|
179
|
+
log(`[producer] ${judge.host} judge batch ${index + 1} (${items.length} items)`);
|
|
180
|
+
const parsed = await runCall(judge, index, items.length, judge.prompt(items));
|
|
181
|
+
const verdicts = new Map((parsed.structured?.verdicts || []).map((v) => [v.id, v]));
|
|
182
|
+
for (const l of batch) setVerdicts(l, judge, parsed.error ? { error: parsed.error } : { verdicts });
|
|
183
|
+
saveCheckpoint();
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
const shown = (adapter) => ({
|
|
187
|
+
host: adapter.host, binary: adapter.binary, requestedModel: adapter.model, effort,
|
|
188
|
+
args: adapter.args.filter((a) => !a.startsWith('{')).map((a) => (a.startsWith(workDir) ? '<schema>' : a)),
|
|
189
|
+
});
|
|
190
|
+
return {
|
|
191
|
+
schemaVersion: 1, kind: 'oracle-labels', repo: inventory.repo, commit: inventory.commit, rulesVersion: inventory.rulesVersion,
|
|
192
|
+
producerVersion: PRODUCER_VERSION, roles,
|
|
193
|
+
producer: {
|
|
194
|
+
generator: shown(generator), judge: shown(judge),
|
|
195
|
+
claude: { binary: 'claude', requestedModel: models.claude, effort },
|
|
196
|
+
codex: { binary: 'codex', requestedModel: models.codex, effort },
|
|
197
|
+
batchSize, retries,
|
|
198
|
+
note: 'total_cost_usd is the host\'s at-list-price estimate; both hosts were verified subscription-authenticated (claude.ai/max, ChatGPT) and no provider key was present in the child env.',
|
|
199
|
+
},
|
|
200
|
+
checkpoint: checkpointFile ? { file: checkpointFile, key, reusedUnits } : null,
|
|
201
|
+
suspended, envAudit, calls, labels: ordered(),
|
|
202
|
+
};
|
|
203
|
+
}
|
|
204
|
+
|
|
205
|
+
/** Role-neutral verdicts on `judge`; the historical `codex` shape is kept exactly when codex judged. */
|
|
206
|
+
function setVerdicts(label, judge, { error, verdicts }) {
|
|
207
|
+
const pick = (suffix) => {
|
|
208
|
+
if (error) return { error };
|
|
209
|
+
const v = verdicts.get(`${label.unitId}:${suffix}`);
|
|
210
|
+
return v ? { answers: v.answers, reason: v.reason } : { error: 'missing verdict' };
|
|
211
|
+
};
|
|
212
|
+
label.judge = { host: judge.host, model: judge.model, direct: pick('d'), paraphrase: pick('p'), equivalent: pick('e') };
|
|
213
|
+
if (judge.host === 'codex') label.codex = error ? { error } : { direct: label.judge.direct, paraphrase: label.judge.paraphrase };
|
|
214
|
+
}
|
|
215
|
+
|
|
216
|
+
function baseLabel(unit, extra) {
|
|
217
|
+
return {
|
|
218
|
+
unitId: unit.unitId, path: unit.path, blobSha: unit.blobSha, startLine: unit.startLine, endLine: unit.endLine,
|
|
219
|
+
bytesSha256: unit.bytesSha256, kind: unit.kind, language: unit.language, ...extra,
|
|
220
|
+
};
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
function callRecord(adapter, batchIndex, items, result, parsed, attempt, quotaRefusal) {
|
|
224
|
+
const rec = {
|
|
225
|
+
host: adapter.host, stage: adapter.stage, batchIndex, attempt, items, requestedModel: adapter.model,
|
|
226
|
+
status: result.status, timedOut: result.timedOut, durationMs: result.durationMs, ok: !parsed.error, error: parsed.error, quotaRefusal,
|
|
227
|
+
};
|
|
228
|
+
if (adapter.host === 'claude' && parsed.envelope) {
|
|
229
|
+
const e = parsed.envelope;
|
|
230
|
+
rec.modelUsage = Object.keys(e.modelUsage || {});
|
|
231
|
+
rec.numTurns = e.num_turns;
|
|
232
|
+
rec.reportedCostEstimateUsd = e.total_cost_usd;
|
|
233
|
+
rec.usage = e.usage && { input: e.usage.input_tokens, cacheCreate: e.usage.cache_creation_input_tokens, cacheRead: e.usage.cache_read_input_tokens, output: e.usage.output_tokens };
|
|
234
|
+
}
|
|
235
|
+
if (adapter.host === 'codex') rec.hostErrors = parsed.errors || [];
|
|
236
|
+
rec.returnedLabels = parsed.structured?.labels?.length;
|
|
237
|
+
rec.returnedVerdicts = parsed.structured?.verdicts?.length;
|
|
238
|
+
if (parsed.error) rec.stderrTail = String(result.stderr || '').slice(-400);
|
|
239
|
+
return rec;
|
|
240
|
+
}
|
|
241
|
+
|
|
242
|
+
function arg(argv, flag) { const i = argv.indexOf(flag); return i >= 0 ? argv[i + 1] : undefined; }
|
|
243
|
+
const optionalNumber = (value) => (value == null ? undefined : Number(value));
|
|
244
|
+
|
|
245
|
+
export async function main(argv = process.argv.slice(2)) {
|
|
246
|
+
const inventoryFile = arg(argv, '--inventory');
|
|
247
|
+
const snapshotDir = arg(argv, '--dir');
|
|
248
|
+
const out = arg(argv, '--out');
|
|
249
|
+
if (!inventoryFile || !snapshotDir || !out) {
|
|
250
|
+
process.stderr.write('Usage: produce-questions.mjs --inventory <inventory.json> --dir <snapshot> --out <labels.json> '
|
|
251
|
+
+ '[--generator claude|codex] [--judge codex|claude] [--claude-model id] [--codex-model id] [--batch 20] '
|
|
252
|
+
+ '[--max-generator-calls n] [--max-judge-calls n] [--max-claude-calls 8] [--max-codex-calls 8] [--retries 0] '
|
|
253
|
+
+ '[--checkpoint <file>] [--effort medium]\n');
|
|
254
|
+
return 64;
|
|
255
|
+
}
|
|
256
|
+
const inventory = JSON.parse(fs.readFileSync(inventoryFile, 'utf8'));
|
|
257
|
+
const labels = await produceQuestions({
|
|
258
|
+
inventory, snapshotDir: path.resolve(snapshotDir),
|
|
259
|
+
roles: { generator: arg(argv, '--generator') || DEFAULT_ROLES.generator, judge: arg(argv, '--judge') || DEFAULT_ROLES.judge },
|
|
260
|
+
models: { claude: arg(argv, '--claude-model') || PRODUCER_MODELS.claude, codex: arg(argv, '--codex-model') || PRODUCER_MODELS.codex },
|
|
261
|
+
batchSize: Number(arg(argv, '--batch') || DEFAULT_BATCH),
|
|
262
|
+
maxClaudeCalls: Number(arg(argv, '--max-claude-calls') || 8),
|
|
263
|
+
maxCodexCalls: Number(arg(argv, '--max-codex-calls') || 8),
|
|
264
|
+
maxGeneratorCalls: optionalNumber(arg(argv, '--max-generator-calls')),
|
|
265
|
+
maxJudgeCalls: optionalNumber(arg(argv, '--max-judge-calls')),
|
|
266
|
+
retries: Number(arg(argv, '--retries') || 0),
|
|
267
|
+
checkpointFile: arg(argv, '--checkpoint') || null,
|
|
268
|
+
effort: arg(argv, '--effort') || 'medium',
|
|
269
|
+
log: (line) => process.stderr.write(`${line}\n`),
|
|
270
|
+
});
|
|
271
|
+
fs.writeFileSync(out, `${JSON.stringify(labels, null, 2)}\n`);
|
|
272
|
+
const okCalls = labels.calls.filter((c) => c.ok).length;
|
|
273
|
+
process.stderr.write(`[producer] ${labels.repo}: ${labels.labels.length} labels, ${labels.calls.length} calls (${okCalls} ok)`
|
|
274
|
+
+ `${labels.suspended ? ` — SUSPENDED on ${labels.suspended.host} capacity refusal` : ''}\n`);
|
|
275
|
+
return labels.suspended ? 75 : 0;
|
|
276
|
+
}
|
|
277
|
+
|
|
278
|
+
// Entry-point guard. Compares REALPATHS on both sides: path.resolve() normalizes a path but does
|
|
279
|
+
// NOT follow symlinks, while import.meta.url IS symlink-resolved by Node. Through a symlink (npm bin
|
|
280
|
+
// shims, wrapper scripts, and every os.tmpdir() path on macOS) the two sides disagree, so main()
|
|
281
|
+
// never runs -- and because nothing throws, the process exits 0. A silent exit 0 is indistinguishable
|
|
282
|
+
// from "ran, found nothing", which is how prepareCorpusCandidate once reported SUCCESS with no
|
|
283
|
+
// archive on disk. Reproduced live 2026-07-27; pinned by tests/unit/entrypoint-symlink.test.mjs.
|
|
284
|
+
function isDirectInvocation() {
|
|
285
|
+
try {
|
|
286
|
+
if (!process.argv[1]) return false;
|
|
287
|
+
return fs.realpathSync(process.argv[1]) === fs.realpathSync(fileURLToPath(import.meta.url));
|
|
288
|
+
} catch {
|
|
289
|
+
return false;
|
|
290
|
+
}
|
|
291
|
+
}
|
|
292
|
+
|
|
293
|
+
if (isDirectInvocation()) process.exitCode = await main();
|
|
@@ -0,0 +1,235 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* scripts/oracle/producer-hosts.mjs — the two subscription-billed HOSTS the Step 14 oracle producer
|
|
3
|
+
* drives, and the ROLE ADAPTERS that let either host generate while the other judges.
|
|
4
|
+
*
|
|
5
|
+
* WHY ROLES ARE NOW SWAPPABLE (2026-09-14). The Fable weekly quota is exhausted, and a plain `claude -p`
|
|
6
|
+
* generation pass is the expensive direction. An Astra-only Dual deliberation (verifier
|
|
7
|
+
* ACCEPT_WITH_CORRECTIONS) approved PATH B — `codex` (gpt-6-astra) GENERATES and `claude` JUDGES —
|
|
8
|
+
* because ADR-086's C3 definition requires source-grounded generation plus INDEPENDENT validation, not
|
|
9
|
+
* Claude as the generator specifically. The same Dual rejected same-vendor production (both roles on
|
|
10
|
+
* one host) for acceptance, so hostAdapters() refuses it unless a caller explicitly opts in for a
|
|
11
|
+
* diagnostic run. The spike's measured quality does not transfer to B; B must qualify on its own pilot.
|
|
12
|
+
*
|
|
13
|
+
* Fence (owner mandate): children get scripts/subscription-hosts.mjs#subscriptionOnlyEnv, which strips
|
|
14
|
+
* every API_BILLING_ENV name. `--bare` is never used — its help text makes auth "strictly
|
|
15
|
+
* ANTHROPIC_API_KEY", the opposite of the fence. Claude context is minimised with --system-prompt,
|
|
16
|
+
* --tools "", --strict-mcp-config, --disable-slash-commands and --setting-sources "" instead: measured
|
|
17
|
+
* 2026-09-14, a one-word `claude -p` loaded 303,673 cache-creation tokens without those flags and 2,334
|
|
18
|
+
* with them.
|
|
19
|
+
*/
|
|
20
|
+
import { spawn } from 'node:child_process';
|
|
21
|
+
|
|
22
|
+
// Pinned on 2026-09-13 against the native hosts (same ids scripts/dual-host-deliberation.mjs pins).
|
|
23
|
+
export const PRODUCER_MODELS = Object.freeze({ claude: 'claude-fable-5-1', codex: 'gpt-6-astra' });
|
|
24
|
+
export const PRODUCER_HOSTS = Object.freeze(['claude', 'codex']);
|
|
25
|
+
export const DEFAULT_ROLES = Object.freeze({ generator: 'claude', judge: 'codex' });
|
|
26
|
+
export const CLAUDE_TIMEOUT_MS = 900_000;
|
|
27
|
+
export const CODEX_TIMEOUT_MS = 600_000;
|
|
28
|
+
|
|
29
|
+
const labelItem = {
|
|
30
|
+
type: 'object', additionalProperties: false,
|
|
31
|
+
properties: {
|
|
32
|
+
unitId: { type: 'string' }, direct: { type: 'string' }, paraphrase: { type: 'string' }, span: { type: 'string' },
|
|
33
|
+
spanStartLine: { type: 'integer' }, spanEndLine: { type: 'integer' }, skip: { type: 'boolean' }, reason: { type: 'string' },
|
|
34
|
+
},
|
|
35
|
+
required: ['unitId', 'direct', 'paraphrase', 'span', 'spanStartLine', 'spanEndLine', 'skip', 'reason'],
|
|
36
|
+
};
|
|
37
|
+
export const LABEL_SCHEMA = Object.freeze({
|
|
38
|
+
type: 'object', additionalProperties: false, properties: { labels: { type: 'array', items: labelItem } }, required: ['labels'],
|
|
39
|
+
});
|
|
40
|
+
export const VERDICT_SCHEMA = Object.freeze({
|
|
41
|
+
type: 'object', additionalProperties: false,
|
|
42
|
+
properties: {
|
|
43
|
+
verdicts: {
|
|
44
|
+
type: 'array',
|
|
45
|
+
items: {
|
|
46
|
+
type: 'object', additionalProperties: false,
|
|
47
|
+
properties: { id: { type: 'string' }, answers: { type: 'string', enum: ['yes', 'no'] }, reason: { type: 'string' } },
|
|
48
|
+
required: ['id', 'answers', 'reason'],
|
|
49
|
+
},
|
|
50
|
+
},
|
|
51
|
+
},
|
|
52
|
+
required: ['verdicts'],
|
|
53
|
+
});
|
|
54
|
+
|
|
55
|
+
export const CLAUDE_SYSTEM_PROMPT = 'You write retrieval-benchmark labels for source text. You see only the units in the message, '
|
|
56
|
+
+ 'you have no tools, and you must not use anything outside them. Return only the structured output.';
|
|
57
|
+
export const JUDGE_SYSTEM_PROMPT = 'You are an independent verifier of retrieval-benchmark labels. You see only the items in the '
|
|
58
|
+
+ 'message, you have no tools, and you use no outside knowledge. Return only the structured output.';
|
|
59
|
+
|
|
60
|
+
/** The GENERATION prompt. Host-neutral: the exact same text goes to whichever host generates. */
|
|
61
|
+
export function claudePrompt(batch) {
|
|
62
|
+
const head = [
|
|
63
|
+
'Write labels for each UNIT below. Return exactly one object per unit, in order:',
|
|
64
|
+
'- unitId: copy exactly.',
|
|
65
|
+
'- direct: one specific question (8-30 words) answerable ONLY from this unit\'s text. Do not reuse more than 4 consecutive words of the answer text inside the question.',
|
|
66
|
+
'- paraphrase: reword the direct question so it keeps the same meaning and the same answer but uses different words and a different sentence structure.',
|
|
67
|
+
'- span: the EXACT contiguous substring of the unit text that answers the question. Copy it character-for-character: same spelling, casing, punctuation, whitespace and line breaks. 1-4 sentences of prose or 1-12 lines of code. Never the heading line alone; never the whole unit.',
|
|
68
|
+
'- spanStartLine, spanEndLine: 1-based line numbers of that span INSIDE the unit text (line 1 is the first line after the ===UNIT marker).',
|
|
69
|
+
'- skip: true only if the unit has no askable factual content; then set direct/paraphrase/span to "" and the line numbers to 0 and explain in reason. Otherwise reason is "".',
|
|
70
|
+
'Do not invent facts. The text between ===UNIT and ===END is verbatim source.',
|
|
71
|
+
'',
|
|
72
|
+
];
|
|
73
|
+
const body = batch.map(({ unit, text, truncated }) => [
|
|
74
|
+
`===UNIT unitId=${unit.unitId} path=${unit.path} kind=${unit.kind} lines=${text.split('\n').length}${truncated ? ' truncated=true' : ''}`,
|
|
75
|
+
text,
|
|
76
|
+
'===END',
|
|
77
|
+
].join('\n'));
|
|
78
|
+
return head.concat(body).join('\n');
|
|
79
|
+
}
|
|
80
|
+
export const generatorPrompt = claudePrompt;
|
|
81
|
+
|
|
82
|
+
/**
|
|
83
|
+
* The JUDGE prompt. Host-neutral. SUPPORT items carry {id, question, span}; EQUIVALENCE items carry
|
|
84
|
+
* {id, questionA, questionB} and ask whether the paraphrase preserves the direct question's meaning —
|
|
85
|
+
* the pair-equivalence judgment Dual required, because "support for two questions does not establish
|
|
86
|
+
* that they mean the same thing". The judge never sees the unit, its path or its repository.
|
|
87
|
+
*/
|
|
88
|
+
export function codexPrompt(items) {
|
|
89
|
+
const support = items.filter((item) => typeof item.span === 'string');
|
|
90
|
+
const equivalence = items.filter((item) => typeof item.questionA === 'string');
|
|
91
|
+
const head = [
|
|
92
|
+
'You are an independent verifier. For each item you receive a QUESTION and a SPAN of text.',
|
|
93
|
+
'Decide whether the SPAN ALONE contains information sufficient to answer the QUESTION correctly and specifically.',
|
|
94
|
+
'Use no outside knowledge. Do not read files or run commands. Return one verdict per item with the same id.',
|
|
95
|
+
'',
|
|
96
|
+
];
|
|
97
|
+
const body = support.map(({ id, question, span }) => `--- id=${id}\nQUESTION: ${question}\nSPAN:\n${span}\n`);
|
|
98
|
+
const lines = head.concat(body);
|
|
99
|
+
if (equivalence.length) {
|
|
100
|
+
lines.push(
|
|
101
|
+
'EQUIVALENCE ITEMS: each gives QUESTION A and QUESTION B. Answer "yes" only if they ask for the same information and would have exactly the same correct answer; "no" if the meaning, scope or expected answer differs.',
|
|
102
|
+
'',
|
|
103
|
+
...equivalence.map(({ id, questionA, questionB }) => `--- id=${id}\nQUESTION A: ${questionA}\nQUESTION B: ${questionB}\n`),
|
|
104
|
+
);
|
|
105
|
+
}
|
|
106
|
+
return lines.join('\n');
|
|
107
|
+
}
|
|
108
|
+
export const judgePrompt = codexPrompt;
|
|
109
|
+
|
|
110
|
+
export function claudeArgs({ model = PRODUCER_MODELS.claude, effort = 'medium', schema = LABEL_SCHEMA, systemPrompt = CLAUDE_SYSTEM_PROMPT } = {}) {
|
|
111
|
+
return [
|
|
112
|
+
'-p', '--output-format', 'json', '--model', model, '--effort', effort, '--tools', '', '--no-session-persistence',
|
|
113
|
+
'--disable-slash-commands', '--strict-mcp-config', '--mcp-config', '{"mcpServers":{}}', '--setting-sources', '',
|
|
114
|
+
'--system-prompt', systemPrompt, '--json-schema', JSON.stringify(schema),
|
|
115
|
+
];
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
export function codexArgs({ model = PRODUCER_MODELS.codex, effort = 'medium', schemaFile } = {}) {
|
|
119
|
+
return [
|
|
120
|
+
'exec', '--ephemeral', '--sandbox', 'read-only', '--skip-git-repo-check', '--color', 'never', '--json',
|
|
121
|
+
'-m', model, '-c', `model_reasoning_effort="${effort}"`, '--output-schema', schemaFile,
|
|
122
|
+
];
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
function claudeStructured(stdout, key) {
|
|
126
|
+
let envelope;
|
|
127
|
+
try { envelope = JSON.parse(stdout); } catch { return { error: 'claude stdout is not a JSON envelope' }; }
|
|
128
|
+
if (envelope.is_error) return { error: `claude reported is_error (${envelope.subtype})`, envelope };
|
|
129
|
+
let structured = envelope.structured_output;
|
|
130
|
+
if (structured === undefined && typeof envelope.result === 'string') {
|
|
131
|
+
try { structured = JSON.parse(envelope.result); } catch { return { error: 'claude result is not JSON', envelope }; }
|
|
132
|
+
}
|
|
133
|
+
if (!structured || !Array.isArray(structured[key])) return { error: `claude output lacks ${key}[]`, envelope };
|
|
134
|
+
return { structured, envelope };
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
export function parseClaudeEnvelope(stdout) { return claudeStructured(stdout, 'labels'); }
|
|
138
|
+
export function parseClaudeVerdicts(stdout) { return claudeStructured(stdout, 'verdicts'); }
|
|
139
|
+
|
|
140
|
+
/** Codex emits JSONL; hooks on this machine inject unrelated agent_messages first, so only the LAST one counts. */
|
|
141
|
+
function codexStructured(stdout, key) {
|
|
142
|
+
const messages = [];
|
|
143
|
+
const errors = [];
|
|
144
|
+
for (const line of String(stdout).split('\n')) {
|
|
145
|
+
let value;
|
|
146
|
+
try { value = JSON.parse(line); } catch { continue; }
|
|
147
|
+
if (value.type !== 'item.completed') continue;
|
|
148
|
+
if (value.item?.type === 'agent_message') messages.push(value.item.text);
|
|
149
|
+
if (value.item?.type === 'error') errors.push(value.item.message);
|
|
150
|
+
}
|
|
151
|
+
const last = messages.at(-1);
|
|
152
|
+
if (last === undefined) return { error: 'codex emitted no agent_message', errors };
|
|
153
|
+
try {
|
|
154
|
+
const structured = JSON.parse(last);
|
|
155
|
+
if (!Array.isArray(structured[key])) return { error: `codex output lacks ${key}[]`, errors };
|
|
156
|
+
return { structured, errors };
|
|
157
|
+
} catch {
|
|
158
|
+
return { error: 'codex final message is not JSON', errors };
|
|
159
|
+
}
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
export function parseCodexJsonl(stdout) { return codexStructured(stdout, 'verdicts'); }
|
|
163
|
+
export function parseCodexLabels(stdout) { return codexStructured(stdout, 'labels'); }
|
|
164
|
+
|
|
165
|
+
/**
|
|
166
|
+
* A host refusing for capacity is an ordinary outcome, not a crash. The Step 14 spike's own session was
|
|
167
|
+
* killed by a usage limit mid-run. On refusal the producer SUSPENDS — no further calls to any host —
|
|
168
|
+
* and every remaining unit is recorded as unproduced, so the denominator never silently shrinks.
|
|
169
|
+
*/
|
|
170
|
+
export function isQuotaRefusal(text) {
|
|
171
|
+
return /usage limit|rate[ -]?limit|quota|capacity|too many requests|\b429\b|limit reached|try again later/i.test(String(text || ''));
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
/**
|
|
175
|
+
* Resolve {generator, judge} into concrete invocations. `schemaFiles` are paths the caller has already
|
|
176
|
+
* written (codex takes an --output-schema FILE; claude takes the schema inline).
|
|
177
|
+
*/
|
|
178
|
+
export function hostAdapters({
|
|
179
|
+
roles = DEFAULT_ROLES, models = PRODUCER_MODELS, effort = 'medium', schemaFiles, workDir, codexCwd,
|
|
180
|
+
allowSameVendor = false,
|
|
181
|
+
} = {}) {
|
|
182
|
+
const { generator, judge } = roles;
|
|
183
|
+
if (!PRODUCER_HOSTS.includes(generator) || !PRODUCER_HOSTS.includes(judge)) {
|
|
184
|
+
throw new Error(`unknown producer role host (generator=${generator}, judge=${judge})`);
|
|
185
|
+
}
|
|
186
|
+
if (generator === judge && !allowSameVendor) {
|
|
187
|
+
throw new Error(`generator and judge are both ${generator}: same-vendor production is not independent validation and cannot qualify labels for C3`);
|
|
188
|
+
}
|
|
189
|
+
const make = (host, stage) => {
|
|
190
|
+
const isGen = stage === 'generator';
|
|
191
|
+
if (host === 'claude') {
|
|
192
|
+
return {
|
|
193
|
+
host, stage, binary: 'claude', model: models.claude, cwd: workDir, timeoutMs: CLAUDE_TIMEOUT_MS,
|
|
194
|
+
args: claudeArgs({
|
|
195
|
+
model: models.claude, effort,
|
|
196
|
+
schema: isGen ? LABEL_SCHEMA : VERDICT_SCHEMA,
|
|
197
|
+
systemPrompt: isGen ? CLAUDE_SYSTEM_PROMPT : JUDGE_SYSTEM_PROMPT,
|
|
198
|
+
}),
|
|
199
|
+
prompt: isGen ? generatorPrompt : judgePrompt,
|
|
200
|
+
parse: isGen ? parseClaudeEnvelope : parseClaudeVerdicts,
|
|
201
|
+
};
|
|
202
|
+
}
|
|
203
|
+
return {
|
|
204
|
+
host, stage, binary: 'codex', model: models.codex, cwd: codexCwd, timeoutMs: CODEX_TIMEOUT_MS,
|
|
205
|
+
args: codexArgs({ model: models.codex, effort, schemaFile: isGen ? schemaFiles.labels : schemaFiles.verdicts }),
|
|
206
|
+
prompt: isGen ? generatorPrompt : judgePrompt,
|
|
207
|
+
parse: isGen ? parseCodexLabels : parseCodexJsonl,
|
|
208
|
+
};
|
|
209
|
+
};
|
|
210
|
+
return { generator: make(generator, 'generator'), judge: make(judge, 'judge') };
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
export function spawnHost(binary, args, { cwd, env, timeoutMs }, input) {
|
|
214
|
+
return new Promise((resolve) => {
|
|
215
|
+
const started = Date.now();
|
|
216
|
+
const child = spawn(binary, args, { cwd, env, stdio: ['pipe', 'pipe', 'pipe'] });
|
|
217
|
+
let stdout = '';
|
|
218
|
+
let stderr = '';
|
|
219
|
+
let timedOut = false;
|
|
220
|
+
let settled = false;
|
|
221
|
+
const timer = setTimeout(() => { timedOut = true; try { child.kill('SIGKILL'); } catch { /* gone */ } }, timeoutMs);
|
|
222
|
+
const finish = (status, error) => {
|
|
223
|
+
if (settled) return;
|
|
224
|
+
settled = true;
|
|
225
|
+
clearTimeout(timer);
|
|
226
|
+
resolve({ status, stdout, stderr: error ? `${stderr}\n${error.message}` : stderr, timedOut, durationMs: Date.now() - started });
|
|
227
|
+
};
|
|
228
|
+
child.stdout.on('data', (chunk) => { stdout += chunk; });
|
|
229
|
+
child.stderr.on('data', (chunk) => { stderr += chunk; });
|
|
230
|
+
child.stdin.on('error', () => { /* host closed stdin early; the close event still settles */ });
|
|
231
|
+
child.on('error', (error) => finish(null, error));
|
|
232
|
+
child.on('close', (status) => finish(status));
|
|
233
|
+
child.stdin.end(input);
|
|
234
|
+
});
|
|
235
|
+
}
|