openmerit 0.1.4 → 0.1.6-preview.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +40 -0
- package/README.md +121 -386
- package/dist/core/src/index.d.ts +101 -0
- package/dist/core/src/index.js +1649 -0
- package/dist/core/src/store.d.ts +35 -0
- package/dist/core/src/store.js +102 -0
- package/dist/pi/src/index.d.ts +32 -0
- package/dist/pi/src/index.js +794 -0
- package/dist/pi/src/scheduler.d.ts +11 -0
- package/dist/pi/src/scheduler.js +137 -0
- package/dist/pi/src/wakeup.d.ts +2 -0
- package/dist/pi/src/wakeup.js +108 -0
- package/dist/protocol/src/index.d.ts +484 -0
- package/dist/protocol/src/index.js +47 -0
- package/dist/protocol/src/schemas.d.ts +576 -0
- package/dist/protocol/src/schemas.js +280 -0
- package/dist/terminal/public/app.js +297 -0
- package/dist/terminal/public/brands/anthropic.png +0 -0
- package/dist/terminal/public/brands/baai.png +0 -0
- package/dist/terminal/public/brands/baseten.png +0 -0
- package/dist/terminal/public/brands/cerebras.png +0 -0
- package/dist/terminal/public/brands/cohere.png +0 -0
- package/dist/terminal/public/brands/deepseek.ico +0 -0
- package/dist/terminal/public/brands/google.png +0 -0
- package/dist/terminal/public/brands/groq.ico +0 -0
- package/dist/terminal/public/brands/lm-studio.png +0 -0
- package/dist/terminal/public/brands/meta.ico +0 -0
- package/dist/terminal/public/brands/mistral.png +0 -0
- package/dist/terminal/public/brands/nomic.png +0 -0
- package/dist/terminal/public/brands/ollama.png +0 -0
- package/dist/terminal/public/brands/openai.png +0 -0
- package/dist/terminal/public/brands/openrouter.png +0 -0
- package/dist/terminal/public/brands/qwen.png +0 -0
- package/dist/terminal/public/brands/vllm.ico +0 -0
- package/dist/terminal/public/brands/vllm.png +0 -0
- package/dist/terminal/public/favicon.svg +1 -0
- package/dist/terminal/public/flow.css +1 -0
- package/dist/terminal/public/flow.js +770 -0
- package/dist/terminal/public/index.html +21 -0
- package/dist/terminal/public/styles.css +779 -0
- package/dist/terminal/src/activity-merge.mjs +64 -0
- package/dist/terminal/src/browser.mjs +29 -0
- package/dist/terminal/src/cli.mjs +60 -0
- package/dist/terminal/src/collect.mjs +311 -0
- package/dist/terminal/src/discovery.mjs +93 -0
- package/dist/terminal/src/hardware.mjs +57 -0
- package/dist/terminal/src/project-activity.mjs +156 -0
- package/dist/terminal/src/sample.mjs +171 -0
- package/dist/terminal/src/server.mjs +56 -0
- package/dist/terminal/src/services.mjs +62 -0
- package/dist/terminal/src/topology.mjs +30 -0
- package/docs/adapter-guide.md +189 -0
- package/docs/architecture.md +59 -0
- package/docs/automation.md +74 -0
- package/docs/budgets.md +37 -0
- package/docs/commands.md +85 -0
- package/docs/demo-backfill.md +29 -0
- package/docs/demo-fieldkit.md +47 -0
- package/docs/demo-placement.md +30 -0
- package/docs/demo-spam.md +15 -0
- package/docs/demo-support.md +42 -0
- package/docs/demo.md +57 -0
- package/docs/first-trial.md +60 -0
- package/docs/getting-started.md +65 -0
- package/docs/index.md +40 -0
- package/docs/inference-terminal.md +439 -0
- package/docs/lifecycle.md +30 -0
- package/docs/memo.md +126 -0
- package/docs/metrics-and-evidence.md +48 -0
- package/docs/operations.md +40 -0
- package/docs/pareto-spec.md +76 -0
- package/docs/pi-extension.md +54 -0
- package/docs/roadmap.md +28 -0
- package/docs/security.md +37 -0
- package/docs/site-artwork-linocut.md +23 -0
- package/docs/site-artwork-miniature-diverse.md +28 -0
- package/docs/site-artwork-miniature.md +26 -0
- package/docs/site-demo.md +177 -0
- package/docs/site-design.md +94 -0
- package/docs/site-documentation.md +83 -0
- package/docs/site-dynamic-og.md +35 -0
- package/docs/site-faq-maintenance.md +115 -0
- package/docs/site-hero-resolution.md +60 -0
- package/docs/site-illustration-sequences.md +227 -0
- package/docs/site-inference-terminal.md +203 -0
- package/docs/site-memo.md +39 -0
- package/docs/site-og-image.md +38 -0
- package/docs/site-og-workshop.md +21 -0
- package/docs/site-section-artwork.md +56 -0
- package/docs/site-skill-review.md +57 -0
- package/docs/site-terminal-preview.md +85 -0
- package/docs/testing.md +118 -0
- package/docs/troubleshooting.md +55 -0
- package/docs/ux-reference.md +32 -0
- package/package.json +74 -42
- package/benchmark/invoice_ocr/data/invoice_01_ground_truth.json +0 -38
- package/benchmark/invoice_ocr/data/invoice_01_row_2.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_02_ground_truth.json +0 -32
- package/benchmark/invoice_ocr/data/invoice_02_row_5.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_03_ground_truth.json +0 -26
- package/benchmark/invoice_ocr/data/invoice_03_row_6.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_04_ground_truth.json +0 -26
- package/benchmark/invoice_ocr/data/invoice_04_row_7.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_05_ground_truth.json +0 -38
- package/benchmark/invoice_ocr/data/invoice_05_row_947.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_06_ground_truth.json +0 -38
- package/benchmark/invoice_ocr/data/invoice_06_row_948.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_07_ground_truth.json +0 -20
- package/benchmark/invoice_ocr/data/invoice_07_row_949.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_08_ground_truth.json +0 -38
- package/benchmark/invoice_ocr/data/invoice_08_row_1888.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_09_ground_truth.json +0 -26
- package/benchmark/invoice_ocr/data/invoice_09_row_1890.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_10_ground_truth.json +0 -20
- package/benchmark/invoice_ocr/data/invoice_10_row_1892.jpg +0 -0
- package/benchmark/invoice_ocr/data/manifest.json +0 -97
- package/dist/benchmarks.js +0 -98
- package/dist/catalog.js +0 -61
- package/dist/cli.js +0 -188
- package/dist/daemon.js +0 -407
- package/dist/diagnostics.js +0 -227
- package/dist/frontier.js +0 -56
- package/dist/harness.js +0 -1
- package/dist/integrations.js +0 -19
- package/dist/invoice-eval.js +0 -33
- package/dist/invoice-score.js +0 -124
- package/dist/judge.js +0 -43
- package/dist/llm.js +0 -207
- package/dist/pi-config.js +0 -46
- package/dist/pi-trials.js +0 -373
- package/dist/policy.js +0 -185
- package/dist/providers.js +0 -1
- package/dist/recommend.js +0 -76
- package/dist/routes.js +0 -74
- package/dist/standalone.js +0 -224
- package/dist/store.js +0 -89
- package/dist/strategist.js +0 -68
- package/dist/task-input.js +0 -54
- package/dist/traces.js +0 -127
- package/dist/trials.js +0 -140
- package/dist/types.js +0 -2
- package/examples/invoice-prompt.txt +0 -19
- package/examples/task.example.json +0 -7
- package/extension/openmerit.ts +0 -947
- package/instructions/OPENMERIT.md +0 -63
- package/instructions/openmerit.policy.json +0 -37
- package/rules.md +0 -43
|
@@ -0,0 +1,794 @@
|
|
|
1
|
+
import { readdirSync, readFileSync, statSync } from "node:fs";
|
|
2
|
+
import { createHash } from "node:crypto";
|
|
3
|
+
import { basename, join, relative } from "node:path";
|
|
4
|
+
import { fileURLToPath } from "node:url";
|
|
5
|
+
import { StringEnum } from "@earendil-works/pi-ai";
|
|
6
|
+
import { nextImprovementAction, OpenMeritCoordinator, ProjectStore, } from "../../core/src/index.js";
|
|
7
|
+
import { completionToolSchemaFor, OPENMERIT_SCHEMA_DIALECT, PROTOCOL_VERSION, MetricRecordSchema, } from "../../protocol/src/index.js";
|
|
8
|
+
import { Type } from "typebox";
|
|
9
|
+
import { piWakeupPlistPath, provisionPiWakeup, removePiWakeup } from "./scheduler.js";
|
|
10
|
+
const INTENT_ENTRY_TYPE = "openmerit.intent";
|
|
11
|
+
const RESULT_ENTRY_TYPE = "openmerit.result";
|
|
12
|
+
const INTENT_MESSAGE_TYPE = "openmerit.intent-context";
|
|
13
|
+
export const PI_HARNESS_DESCRIPTOR = {
|
|
14
|
+
id: "pi",
|
|
15
|
+
name: "Pi coding agent",
|
|
16
|
+
version: "0.87",
|
|
17
|
+
protocolVersions: [PROTOCOL_VERSION],
|
|
18
|
+
capabilities: [
|
|
19
|
+
"user_confirmation",
|
|
20
|
+
"evaluation_authoring",
|
|
21
|
+
"observability_instrumentation",
|
|
22
|
+
"production_observation",
|
|
23
|
+
"candidate_discovery",
|
|
24
|
+
"challenger_execution",
|
|
25
|
+
"frontier_calculation",
|
|
26
|
+
"model_mutation",
|
|
27
|
+
"post_swap_verification",
|
|
28
|
+
"rollback",
|
|
29
|
+
],
|
|
30
|
+
executionModes: ["interactive", "non_interactive"],
|
|
31
|
+
structuredOutput: {
|
|
32
|
+
schemaDialect: OPENMERIT_SCHEMA_DIALECT,
|
|
33
|
+
presentation: "dynamic_tool",
|
|
34
|
+
enforcement: "harness",
|
|
35
|
+
},
|
|
36
|
+
automation: {
|
|
37
|
+
supportedSignals: [
|
|
38
|
+
"product_task_completed", "metric_window_available", "scheduled_tick",
|
|
39
|
+
"model_catalog_changed", "verification_window_completed",
|
|
40
|
+
],
|
|
41
|
+
persistentScheduling: process.platform === "darwin",
|
|
42
|
+
backgroundExecution: process.platform === "darwin",
|
|
43
|
+
wakeupProvisioning: process.platform === "darwin" ? "infrastructure_change" : "none",
|
|
44
|
+
},
|
|
45
|
+
};
|
|
46
|
+
export function createInitialUiState() {
|
|
47
|
+
return {
|
|
48
|
+
proactiveNudgesPaused: false,
|
|
49
|
+
applicationTarget: "not_detected",
|
|
50
|
+
taskProfile: "not_configured",
|
|
51
|
+
metricReadiness: "not_assessed",
|
|
52
|
+
evaluationReadiness: "not_assessed",
|
|
53
|
+
activeIntent: null,
|
|
54
|
+
nextAssessment: null,
|
|
55
|
+
lifecycleStage: "not_initialized",
|
|
56
|
+
incumbentCandidate: null,
|
|
57
|
+
candidateCount: 0,
|
|
58
|
+
assessmentCount: 0,
|
|
59
|
+
evaluationSpend: 0,
|
|
60
|
+
proposedCandidate: null,
|
|
61
|
+
};
|
|
62
|
+
}
|
|
63
|
+
export function renderStatus(state) {
|
|
64
|
+
return [
|
|
65
|
+
"OpenMerit evidence readiness",
|
|
66
|
+
`application target : ${state.applicationTarget.replaceAll("_", " ")}`,
|
|
67
|
+
`task profile : ${state.taskProfile.replaceAll("_", " ")}`,
|
|
68
|
+
`metric readiness : ${state.metricReadiness.replaceAll("_", " ")}`,
|
|
69
|
+
`evaluation status : ${state.evaluationReadiness.replaceAll("_", " ")}`,
|
|
70
|
+
`active work : ${state.activeIntent ?? "none"}`,
|
|
71
|
+
`next assessment : ${state.nextAssessment ?? "not scheduled"}`,
|
|
72
|
+
`frontier lifecycle : ${state.lifecycleStage.replaceAll("_", " ")}`,
|
|
73
|
+
`incumbent candidate: ${state.incumbentCandidate ?? "not confirmed"}`,
|
|
74
|
+
`candidate set : ${state.candidateCount}`,
|
|
75
|
+
`assessments : ${state.assessmentCount}`,
|
|
76
|
+
`evaluation spend : $${state.evaluationSpend.toFixed(6)}`,
|
|
77
|
+
`proposed candidate : ${state.proposedCandidate ?? "none"}`,
|
|
78
|
+
`proactive nudges : ${state.proactiveNudgesPaused ? "paused" : "enabled"}`,
|
|
79
|
+
].join("\n");
|
|
80
|
+
}
|
|
81
|
+
const DETECTION_IGNORES = new Set([
|
|
82
|
+
".git", ".openmerit", "node_modules", "dist", "build", "coverage", ".next", "vendor", "tests", "test", "__tests__",
|
|
83
|
+
]);
|
|
84
|
+
const DETECTION_EXTENSIONS = new Set([".js", ".jsx", ".mjs", ".cjs", ".ts", ".tsx", ".py", ".go", ".rs", ".java", ".kt"]);
|
|
85
|
+
const APPLICATION_CALL_PATTERNS = [
|
|
86
|
+
{ provider: "openai-compatible", pattern: /\b(?:chat\.completions|responses)\.create\s*\(/ },
|
|
87
|
+
{ provider: "anthropic", pattern: /\bmessages\.create\s*\(/ },
|
|
88
|
+
{ provider: "google", pattern: /\bgenerateContent(?:Stream)?\s*\(/ },
|
|
89
|
+
{ provider: "vercel-ai", pattern: /\b(?:generateText|streamText|generateObject|streamObject)\s*\(/ },
|
|
90
|
+
{ provider: "langchain", pattern: /\b(?:ChatOpenAI|ChatAnthropic|ChatGoogleGenerativeAI)\s*\(/ },
|
|
91
|
+
{ provider: "openrouter", pattern: /openrouter\.ai\/api\/v1[\s\S]{0,5000}\/(?:chat\/completions|responses)/ },
|
|
92
|
+
];
|
|
93
|
+
function extensionOf(path) {
|
|
94
|
+
const match = path.match(/(\.[^.\/]+)$/);
|
|
95
|
+
return match?.[1]?.toLowerCase() ?? "";
|
|
96
|
+
}
|
|
97
|
+
/** Bounded, adapter-local detection. It requires an actual application call pattern, not only an SDK dependency. */
|
|
98
|
+
export function detectApplicationLlmTargets(projectRoot) {
|
|
99
|
+
const files = [];
|
|
100
|
+
const pending = [projectRoot];
|
|
101
|
+
while (pending.length && files.length < 500) {
|
|
102
|
+
const directory = pending.shift();
|
|
103
|
+
let entries;
|
|
104
|
+
try {
|
|
105
|
+
entries = readdirSync(directory, { withFileTypes: true });
|
|
106
|
+
}
|
|
107
|
+
catch {
|
|
108
|
+
continue;
|
|
109
|
+
}
|
|
110
|
+
for (const entry of entries) {
|
|
111
|
+
if (DETECTION_IGNORES.has(entry.name))
|
|
112
|
+
continue;
|
|
113
|
+
const path = join(directory, entry.name);
|
|
114
|
+
if (entry.isDirectory())
|
|
115
|
+
pending.push(path);
|
|
116
|
+
else if (entry.isFile() && DETECTION_EXTENSIONS.has(extensionOf(path)))
|
|
117
|
+
files.push(path);
|
|
118
|
+
if (files.length >= 500)
|
|
119
|
+
break;
|
|
120
|
+
}
|
|
121
|
+
}
|
|
122
|
+
const applicationId = basename(projectRoot);
|
|
123
|
+
const targets = [];
|
|
124
|
+
for (const path of files) {
|
|
125
|
+
let source;
|
|
126
|
+
try {
|
|
127
|
+
if (statSync(path).size > 1_000_000)
|
|
128
|
+
continue;
|
|
129
|
+
source = readFileSync(path, "utf8");
|
|
130
|
+
}
|
|
131
|
+
catch {
|
|
132
|
+
continue;
|
|
133
|
+
}
|
|
134
|
+
for (const { provider, pattern } of APPLICATION_CALL_PATTERNS) {
|
|
135
|
+
const match = pattern.exec(source);
|
|
136
|
+
if (!match)
|
|
137
|
+
continue;
|
|
138
|
+
const projectPath = relative(projectRoot, path).replaceAll("\\", "/");
|
|
139
|
+
const line = source.slice(0, match.index).split("\n").length;
|
|
140
|
+
const identity = `${applicationId}:${projectPath}:${provider}`;
|
|
141
|
+
targets.push({
|
|
142
|
+
id: `app-llm-${createHash("sha256").update(identity).digest("hex").slice(0, 12)}`,
|
|
143
|
+
kind: "application_llm_call",
|
|
144
|
+
name: `${provider} application call in ${projectPath}`,
|
|
145
|
+
applicationId,
|
|
146
|
+
callSites: [`${projectPath}:${line}`],
|
|
147
|
+
routeKey: `${projectPath}#${provider}`,
|
|
148
|
+
...((source.match(/(?:DEFAULT_MODEL|defaultModel|model)\s*=\s*["'`]([^"'`]+)["'`]/)?.[1])
|
|
149
|
+
? { configuredModelId: source.match(/(?:DEFAULT_MODEL|defaultModel|model)\s*=\s*["'`]([^"'`]+)["'`]/)[1] }
|
|
150
|
+
: {}),
|
|
151
|
+
});
|
|
152
|
+
}
|
|
153
|
+
}
|
|
154
|
+
return targets;
|
|
155
|
+
}
|
|
156
|
+
function detectedTargetsConstraint(targets) {
|
|
157
|
+
return targets.map((target) => ({
|
|
158
|
+
id: target.id,
|
|
159
|
+
kind: target.kind,
|
|
160
|
+
name: target.name,
|
|
161
|
+
applicationId: target.applicationId,
|
|
162
|
+
callSites: [...target.callSites],
|
|
163
|
+
routeKey: target.routeKey,
|
|
164
|
+
...(target.configuredModelId ? { configuredModelId: target.configuredModelId } : {}),
|
|
165
|
+
}));
|
|
166
|
+
}
|
|
167
|
+
function validateReportedMetrics(metrics, targetId) {
|
|
168
|
+
for (const metric of metrics) {
|
|
169
|
+
if (metric.targetId !== targetId) {
|
|
170
|
+
throw new Error(`${metric.metricId} target does not match the reported application LLM target.`);
|
|
171
|
+
}
|
|
172
|
+
if (metric.state === "measured" && (!metric.evidence.length || metric.value === undefined)) {
|
|
173
|
+
throw new Error(`Measured metric ${metric.metricId} requires a numeric value and evidence.`);
|
|
174
|
+
}
|
|
175
|
+
if (metric.metricId === "tool_use_performance" && metric.state === "measured" &&
|
|
176
|
+
!metric.evidence.some((item) => item.source === "harness_trace")) {
|
|
177
|
+
throw new Error("tool_use_performance requires harness-trace evidence; test coverage is not tool-use evidence.");
|
|
178
|
+
}
|
|
179
|
+
}
|
|
180
|
+
}
|
|
181
|
+
function modelCatalogFingerprint(ctx) {
|
|
182
|
+
const catalogue = ctx.modelRegistry.getAvailable().map((model) => ({
|
|
183
|
+
provider: model.provider,
|
|
184
|
+
id: model.id,
|
|
185
|
+
reasoning: model.reasoning,
|
|
186
|
+
input: model.input,
|
|
187
|
+
cost: model.cost,
|
|
188
|
+
contextWindow: model.contextWindow,
|
|
189
|
+
maxTokens: model.maxTokens,
|
|
190
|
+
})).sort((left, right) => `${left.provider}/${left.id}`.localeCompare(`${right.provider}/${right.id}`));
|
|
191
|
+
return createHash("sha256").update(JSON.stringify(catalogue)).digest("hex");
|
|
192
|
+
}
|
|
193
|
+
function privateIntentMessage(intent) {
|
|
194
|
+
return [
|
|
195
|
+
"OpenMerit has issued the typed intent below.",
|
|
196
|
+
"Perform the intelligent work using the current project, session context, and available tools.",
|
|
197
|
+
"Use harness-configured fast decision resources such as Jev System One aggressively for bounded ranking, triage, and judgment when available; OpenMerit does not call those services itself.",
|
|
198
|
+
"You, the coding harness, own all research, evaluation design, instrumentation, measurement, Pareto calculation, and authorized configuration work. OpenMerit only validates the returned contract and evidence.",
|
|
199
|
+
"The system under test is the user-confirmed application LLM target. Pi's own model, tokens, cost, latency, and session are orchestration data and must never be submitted as application evidence.",
|
|
200
|
+
intent.targetId
|
|
201
|
+
? `Every result and metric must reference application target ${intent.targetId}.`
|
|
202
|
+
: "Before setup succeeds, identify an actual application LLM call or shared application route and ask the user to confirm its stable target ID, call sites, and route key.",
|
|
203
|
+
"For setup, infer task-relevant metrics from the application rather than requiring the user to invent them. Before any setup run, show the user the proposed metrics, rationale, sampling strategy, representative case count, repetitions per case, aggregation, true performance thresholds if any, and estimated cost, then obtain confirmation or edits. Rate, reliability, distribution, percentile, variance, and average claims require multiple runs. Distinguish repeating the same input for stochastic variance from running representative task instances for category performance; use both when the claim needs both. minimumSamples is evidence quantity and must never be copied into constraint.value. A constraint is optional and represents only a real metric threshold in that metric's unit. For an active-session demo, default baselineAssessment and postSwapVerification to afterCompletedTasks: 1 unless the user changes the cadence; never return an empty cadence. Instrument every confirmed required metric. For setup or instrumentation, actually run each task-specific grader and collector through the application path with working provider authentication, inspect the resulting metric values, and cite the run artifacts. Report explicit covered and missing metric IDs; never describe observability as ready when a required metric cannot be produced. A setup smoke run is verification, not a production baseline sample.",
|
|
204
|
+
"Capture cost, token use, and task latency from the application LLM provider or application observability. Do not substitute Pi provider telemetry or zero for unavailable application telemetry.",
|
|
205
|
+
intent.kind === "instrument_observability"
|
|
206
|
+
? "For instrumentation, execute the confirmed application target itself and return one measured MetricRecord for every covered required metric. Map provider/application usage fields to input_token_consumption and cost_per_successful_task, use the exact metric IDs and objective aggregation, include sampleCount and valid startedAt/completedAt windows, and attach durable observability_record or harness_trace evidence. Do not merely describe telemetry or add arbitrary fields to the application response; those do not count until converted into MetricRecords in the completion result."
|
|
207
|
+
: "",
|
|
208
|
+
"Tool-use performance must be derived from tool-call traces and outcomes; code coverage is not tool-use evidence.",
|
|
209
|
+
intent.constraints.modelMutationAllowed
|
|
210
|
+
? `This intent authorizes only the application model mutation for target ${intent.targetId ?? "(missing target)"}. Never change Pi's model.`
|
|
211
|
+
: "Do not change Pi's model or application model configuration.",
|
|
212
|
+
"Before finishing, call openmerit_complete_intent exactly once with the result and durable evidence references.",
|
|
213
|
+
JSON.stringify(intent),
|
|
214
|
+
].join("\n\n");
|
|
215
|
+
}
|
|
216
|
+
function piHarnessAdapter(pi, ctx) {
|
|
217
|
+
return {
|
|
218
|
+
descriptor: PI_HARNESS_DESCRIPTOR,
|
|
219
|
+
async dispatch(intent) {
|
|
220
|
+
pi.appendEntry(INTENT_ENTRY_TYPE, { intent });
|
|
221
|
+
if (ctx.mode === "tui" || ctx.mode === "rpc") {
|
|
222
|
+
pi.sendMessage({
|
|
223
|
+
customType: INTENT_MESSAGE_TYPE,
|
|
224
|
+
content: privateIntentMessage(intent),
|
|
225
|
+
display: false,
|
|
226
|
+
details: { intentId: intent.id, kind: intent.kind },
|
|
227
|
+
}, { deliverAs: "followUp", triggerTurn: true });
|
|
228
|
+
}
|
|
229
|
+
return { accepted: true, harnessJobId: intent.id };
|
|
230
|
+
},
|
|
231
|
+
async scheduleWakeup(request) {
|
|
232
|
+
if (request.signal !== "scheduled_tick")
|
|
233
|
+
return { accepted: false, reason: "Pi only schedules OpenMerit ticks." };
|
|
234
|
+
if (!ctx.model)
|
|
235
|
+
return { accepted: false, reason: "Select an authenticated Pi model before enabling background checks." };
|
|
236
|
+
try {
|
|
237
|
+
const path = await provisionPiWakeup({
|
|
238
|
+
projectRoot: ctx.cwd,
|
|
239
|
+
extensionPath: fileURLToPath(import.meta.url),
|
|
240
|
+
provider: ctx.model.provider,
|
|
241
|
+
modelId: ctx.model.id,
|
|
242
|
+
});
|
|
243
|
+
return { accepted: true, harnessJobId: path };
|
|
244
|
+
}
|
|
245
|
+
catch (error) {
|
|
246
|
+
return { accepted: false, reason: error instanceof Error ? error.message : String(error) };
|
|
247
|
+
}
|
|
248
|
+
},
|
|
249
|
+
};
|
|
250
|
+
}
|
|
251
|
+
async function dispatchIntent(pi, ctx, state, store, kind, prepareCompletionTool, additionalConstraints) {
|
|
252
|
+
if (state.activeIntent) {
|
|
253
|
+
ctx.ui.notify(`openmerit: work is already active (${state.activeIntent})`, "warning");
|
|
254
|
+
return;
|
|
255
|
+
}
|
|
256
|
+
prepareCompletionTool(kind);
|
|
257
|
+
const issued = await new OpenMeritCoordinator(store, piHarnessAdapter(pi, ctx)).issue(kind, { additionalConstraints });
|
|
258
|
+
const { intent } = issued;
|
|
259
|
+
state.activeIntent = intent.id;
|
|
260
|
+
if (kind === "establish_evals") {
|
|
261
|
+
state.applicationTarget = "detected";
|
|
262
|
+
state.taskProfile = "pending";
|
|
263
|
+
state.metricReadiness = "pending";
|
|
264
|
+
state.evaluationReadiness = "pending";
|
|
265
|
+
}
|
|
266
|
+
else {
|
|
267
|
+
state.metricReadiness = "pending";
|
|
268
|
+
}
|
|
269
|
+
state.lifecycleStage = issued.state.cycle.stage;
|
|
270
|
+
const label = kind === "establish_evals" ? "setup" : kind.replaceAll("_", " ");
|
|
271
|
+
ctx.ui.notify(`openmerit: ${label} started (${intent.id})`, "info");
|
|
272
|
+
}
|
|
273
|
+
async function reportAutomationSignal(pi, ctx, state, store, signal, prepareCompletionTool) {
|
|
274
|
+
const result = await new OpenMeritCoordinator(store, piHarnessAdapter(pi, ctx)).recordAutomationSignal(signal);
|
|
275
|
+
const persistedState = state.activeIntent ? await store.readState() : undefined;
|
|
276
|
+
if (state.activeIntent && persistedState?.activeIntent?.kind === "instrument_observability" && signal.type === "metric_window_available") {
|
|
277
|
+
const projectConfig = await store.readConfig();
|
|
278
|
+
const profile = projectConfig?.taskProfiles.find((item) => item.id === projectConfig.activeTaskProfileId);
|
|
279
|
+
const requiredObjectives = profile?.objectives.filter((item) => item.required) ?? [];
|
|
280
|
+
const measuredMetricIds = new Set(signal.metrics
|
|
281
|
+
.filter((metric) => metric.targetId === signal.targetId && metric.state === "measured" && metric.value !== undefined && metric.sampleCount > 0 && metric.evidence.length > 0)
|
|
282
|
+
.map((metric) => metric.metricId));
|
|
283
|
+
const coveredMetricIds = requiredObjectives.filter((objective) => measuredMetricIds.has(objective.metricId)).map((objective) => objective.metricId);
|
|
284
|
+
const missingMetricIds = requiredObjectives.filter((objective) => !measuredMetricIds.has(objective.metricId)).map((objective) => objective.metricId);
|
|
285
|
+
const evidence = [...new Map(signal.metrics.flatMap((metric) => metric.evidence).map((item) => [item.id, item])).values()];
|
|
286
|
+
const completion = {
|
|
287
|
+
protocolVersion: PROTOCOL_VERSION,
|
|
288
|
+
intentId: state.activeIntent,
|
|
289
|
+
status: "succeeded",
|
|
290
|
+
completedAt: new Date().toISOString(),
|
|
291
|
+
summary: `Instrumentation verified ${coveredMetricIds.length} required metric(s); ${missingMetricIds.length} remain missing.`,
|
|
292
|
+
evidence,
|
|
293
|
+
outputs: {
|
|
294
|
+
targetId: signal.targetId,
|
|
295
|
+
coveredMetricIds,
|
|
296
|
+
missingMetricIds,
|
|
297
|
+
},
|
|
298
|
+
};
|
|
299
|
+
const accepted = await new OpenMeritCoordinator(store, piHarnessAdapter(pi, ctx)).acceptResult(completion);
|
|
300
|
+
state.activeIntent = null;
|
|
301
|
+
state.lifecycleStage = accepted.state.cycle.stage;
|
|
302
|
+
state.metricReadiness = accepted.state.observability?.status === "ready" ? "assessed" : "pending";
|
|
303
|
+
prepareCompletionTool("instrument_observability");
|
|
304
|
+
return;
|
|
305
|
+
}
|
|
306
|
+
if (result.decision.action === "issue_intent" && !state.activeIntent) {
|
|
307
|
+
await dispatchIntent(pi, ctx, state, store, result.decision.intentKind, prepareCompletionTool);
|
|
308
|
+
}
|
|
309
|
+
}
|
|
310
|
+
export default function openMeritExtension(pi) {
|
|
311
|
+
const state = createInitialUiState();
|
|
312
|
+
let store;
|
|
313
|
+
const queuedUserInputs = [];
|
|
314
|
+
const releaseQueuedUserInput = () => {
|
|
315
|
+
if (state.activeIntent || !queuedUserInputs.length)
|
|
316
|
+
return;
|
|
317
|
+
const queued = queuedUserInputs.splice(0);
|
|
318
|
+
const text = queued.map((item) => item.text).join("\n\n");
|
|
319
|
+
const images = queued.flatMap((item) => item.images ?? []);
|
|
320
|
+
pi.sendUserMessage(images.length ? [{ type: "text", text }, ...images] : text, { deliverAs: "followUp" });
|
|
321
|
+
};
|
|
322
|
+
const registerCompletionTool = (kind) => {
|
|
323
|
+
const parameters = kind
|
|
324
|
+
? completionToolSchemaFor(kind)
|
|
325
|
+
: Type.Object({
|
|
326
|
+
intentId: Type.String(),
|
|
327
|
+
status: StringEnum(["failed", "cancelled"]),
|
|
328
|
+
summary: Type.String(),
|
|
329
|
+
evidence: Type.Array(Type.Object({
|
|
330
|
+
id: Type.String(),
|
|
331
|
+
source: StringEnum([
|
|
332
|
+
"harness_trace", "evaluation_artifact", "observability_record", "external_source", "user_feedback",
|
|
333
|
+
]),
|
|
334
|
+
uri: Type.String(),
|
|
335
|
+
mediaType: Type.Optional(Type.String()),
|
|
336
|
+
digest: Type.Optional(Type.String()),
|
|
337
|
+
})),
|
|
338
|
+
errorCode: Type.Optional(Type.String()),
|
|
339
|
+
errorMessage: Type.Optional(Type.String()),
|
|
340
|
+
retryable: Type.Optional(Type.Boolean()),
|
|
341
|
+
}, { additionalProperties: false });
|
|
342
|
+
pi.registerTool({
|
|
343
|
+
name: "openmerit_complete_intent",
|
|
344
|
+
label: "Complete OpenMerit Intent",
|
|
345
|
+
description: kind
|
|
346
|
+
? `Complete the active ${kind} OpenMerit intent with its required structured output and durable evidence references.`
|
|
347
|
+
: "Complete the active OpenMerit intent. A successful result is available only after an intent-specific schema is activated.",
|
|
348
|
+
promptSnippet: "Complete the active OpenMerit intent with structured evidence",
|
|
349
|
+
promptGuidelines: [
|
|
350
|
+
"Call openmerit_complete_intent exactly once after completing OpenMerit-requested work.",
|
|
351
|
+
"Never use openmerit_complete_intent to claim success without verified evidence references.",
|
|
352
|
+
"A succeeded result must include intent-specific outputs and at least one evidence reference; failed or cancelled results may omit outputs and use an empty evidence list.",
|
|
353
|
+
],
|
|
354
|
+
parameters,
|
|
355
|
+
constrainedSampling: { type: "json_schema", strict: "prefer" },
|
|
356
|
+
async execute(_toolCallId, input, _signal, _onUpdate, ctx) {
|
|
357
|
+
if (!state.activeIntent)
|
|
358
|
+
throw new Error("No OpenMerit intent is active.");
|
|
359
|
+
if (input.intentId !== state.activeIntent) {
|
|
360
|
+
throw new Error(`Intent ${input.intentId} is not active.`);
|
|
361
|
+
}
|
|
362
|
+
if (input.status === "succeeded" && input.evidence.length === 0) {
|
|
363
|
+
throw new Error("A successful OpenMerit intent requires at least one evidence reference.");
|
|
364
|
+
}
|
|
365
|
+
store ??= new ProjectStore(ctx.cwd);
|
|
366
|
+
const outputs = input.outputs ?? {};
|
|
367
|
+
const result = {
|
|
368
|
+
protocolVersion: PROTOCOL_VERSION,
|
|
369
|
+
intentId: input.intentId,
|
|
370
|
+
status: input.status,
|
|
371
|
+
completedAt: new Date().toISOString(),
|
|
372
|
+
summary: input.summary,
|
|
373
|
+
evidence: input.evidence,
|
|
374
|
+
outputs,
|
|
375
|
+
...(input.status === "failed"
|
|
376
|
+
? {
|
|
377
|
+
error: {
|
|
378
|
+
code: input.errorCode ?? "harness_work_failed",
|
|
379
|
+
message: input.errorMessage ?? input.summary,
|
|
380
|
+
retryable: input.retryable ?? false,
|
|
381
|
+
},
|
|
382
|
+
}
|
|
383
|
+
: {}),
|
|
384
|
+
};
|
|
385
|
+
const accepted = await new OpenMeritCoordinator(store, piHarnessAdapter(pi, ctx)).acceptResult(result);
|
|
386
|
+
if (accepted.intentKind === "establish_evals" &&
|
|
387
|
+
(accepted.state.cycle.policy?.checkPolicy.execution.mode !== "persistent" ||
|
|
388
|
+
!accepted.state.cycle.policy?.checkPolicy.execution.infrastructureChangesAllowed)) {
|
|
389
|
+
try {
|
|
390
|
+
await removePiWakeup(ctx.cwd);
|
|
391
|
+
}
|
|
392
|
+
catch (error) {
|
|
393
|
+
ctx.ui.notify(`openmerit: could not remove prior wakeup: ${error instanceof Error ? error.message : String(error)}`, "warning");
|
|
394
|
+
}
|
|
395
|
+
}
|
|
396
|
+
pi.appendEntry(RESULT_ENTRY_TYPE, { result });
|
|
397
|
+
const { intentKind, state: nextState } = accepted;
|
|
398
|
+
const stage = nextState.cycle.stage;
|
|
399
|
+
state.activeIntent = null;
|
|
400
|
+
state.lifecycleStage = stage;
|
|
401
|
+
state.incumbentCandidate = nextState.cycle.activeCandidateId ?? null;
|
|
402
|
+
state.candidateCount = nextState.candidates?.length ?? 0;
|
|
403
|
+
state.assessmentCount = nextState.assessments?.length ?? 0;
|
|
404
|
+
state.evaluationSpend = nextState.automation.evaluationSpend ?? 0;
|
|
405
|
+
state.proposedCandidate = nextState.cycle.proposedCandidateId ?? null;
|
|
406
|
+
registerCompletionTool();
|
|
407
|
+
state.nextAssessment = accepted.automation.plan.nextWakeupAt ??
|
|
408
|
+
(accepted.automation.plan.scheduling === "external_scheduler_required" ? "external scheduler required" : "signal driven");
|
|
409
|
+
if (accepted.automation.plan.gaps.length) {
|
|
410
|
+
ctx.ui.notify(`openmerit automation: ${accepted.automation.plan.gaps.join(" ")}`, "warning");
|
|
411
|
+
}
|
|
412
|
+
if (input.status === "succeeded" && intentKind === "establish_evals") {
|
|
413
|
+
state.applicationTarget = "confirmed";
|
|
414
|
+
state.taskProfile = "configured";
|
|
415
|
+
state.metricReadiness = nextState.observability?.status === "ready" ? "assessed" : "pending";
|
|
416
|
+
state.evaluationReadiness = "assessed";
|
|
417
|
+
}
|
|
418
|
+
else if (input.status === "succeeded" && intentKind === "instrument_observability") {
|
|
419
|
+
state.metricReadiness = nextState.observability?.status === "ready" ? "assessed" : "pending";
|
|
420
|
+
}
|
|
421
|
+
if (input.status === "succeeded" && !state.proactiveNudgesPaused) {
|
|
422
|
+
const decision = nextImprovementAction(nextState.cycle);
|
|
423
|
+
if (decision.action === "issue_intent") {
|
|
424
|
+
await dispatchIntent(pi, ctx, state, store, decision.intentKind, registerCompletionTool);
|
|
425
|
+
}
|
|
426
|
+
else if (decision.action === "await_user") {
|
|
427
|
+
ctx.ui.notify(`openmerit: ${decision.reason}`, "warning");
|
|
428
|
+
}
|
|
429
|
+
}
|
|
430
|
+
// A successful result can immediately issue another OpenMerit intent
|
|
431
|
+
// (for example, setup -> instrumentation). Release product work only
|
|
432
|
+
// after that chain has no active follow-up.
|
|
433
|
+
releaseQueuedUserInput();
|
|
434
|
+
return {
|
|
435
|
+
content: [{ type: "text", text: `Recorded OpenMerit intent ${input.intentId} as ${input.status}.` }],
|
|
436
|
+
details: { result },
|
|
437
|
+
};
|
|
438
|
+
},
|
|
439
|
+
});
|
|
440
|
+
};
|
|
441
|
+
registerCompletionTool();
|
|
442
|
+
pi.registerTool({
|
|
443
|
+
name: "openmerit_report_signal",
|
|
444
|
+
label: "Report OpenMerit Signal",
|
|
445
|
+
description: "Report a verified application LLM task, observability metric window, or completed post-swap verification window to OpenMerit.",
|
|
446
|
+
promptSnippet: "Report verified application-target OpenMerit signals",
|
|
447
|
+
promptGuidelines: [
|
|
448
|
+
"Use stable signalId values so retried deliveries remain idempotent.",
|
|
449
|
+
"Every signal and metric must reference the confirmed application LLM target.",
|
|
450
|
+
"Report only application-provider telemetry and measured metrics backed by referenced application observability or evaluation artifacts. Never report Pi's own model usage.",
|
|
451
|
+
],
|
|
452
|
+
parameters: Type.Object({
|
|
453
|
+
signalId: Type.String({ minLength: 1 }),
|
|
454
|
+
type: StringEnum(["product_task_completed", "metric_window_available", "verification_window_completed"]),
|
|
455
|
+
occurredAt: Type.Optional(Type.String({ minLength: 1 })),
|
|
456
|
+
targetId: Type.String({ minLength: 1 }),
|
|
457
|
+
metrics: Type.Optional(Type.Array(MetricRecordSchema)),
|
|
458
|
+
modelId: Type.Optional(Type.String({ minLength: 1 })),
|
|
459
|
+
evidence: Type.Optional(Type.Array(Type.Object({
|
|
460
|
+
id: Type.String({ minLength: 1 }),
|
|
461
|
+
source: StringEnum(["harness_trace", "evaluation_artifact", "observability_record", "external_source", "user_feedback"]),
|
|
462
|
+
uri: Type.String({ minLength: 1 }),
|
|
463
|
+
mediaType: Type.Optional(Type.String({ minLength: 1 })),
|
|
464
|
+
digest: Type.Optional(Type.String({ minLength: 1 })),
|
|
465
|
+
}, { additionalProperties: false }))),
|
|
466
|
+
telemetry: Type.Optional(Type.Object({
|
|
467
|
+
startedAt: Type.String({ minLength: 1 }),
|
|
468
|
+
completedAt: Type.String({ minLength: 1 }),
|
|
469
|
+
durationMs: Type.Number({ minimum: 0 }),
|
|
470
|
+
inputTokens: Type.Number({ minimum: 0 }),
|
|
471
|
+
outputTokens: Type.Number({ minimum: 0 }),
|
|
472
|
+
cacheReadTokens: Type.Number({ minimum: 0 }),
|
|
473
|
+
cacheWriteTokens: Type.Number({ minimum: 0 }),
|
|
474
|
+
cost: Type.Number({ minimum: 0 }),
|
|
475
|
+
currency: Type.Literal("USD"),
|
|
476
|
+
}, { additionalProperties: false })),
|
|
477
|
+
}, { additionalProperties: false }),
|
|
478
|
+
constrainedSampling: { type: "json_schema", strict: "prefer" },
|
|
479
|
+
async execute(_toolCallId, input, _signal, _onUpdate, ctx) {
|
|
480
|
+
store ??= new ProjectStore(ctx.cwd);
|
|
481
|
+
if (input.type === "product_task_completed") {
|
|
482
|
+
if (!input.modelId || !input.evidence?.length) {
|
|
483
|
+
throw new Error("A completed application task requires the application modelId and durable evidence.");
|
|
484
|
+
}
|
|
485
|
+
await reportAutomationSignal(pi, ctx, state, store, {
|
|
486
|
+
protocolVersion: PROTOCOL_VERSION,
|
|
487
|
+
id: input.signalId,
|
|
488
|
+
type: input.type,
|
|
489
|
+
occurredAt: input.occurredAt ?? new Date().toISOString(),
|
|
490
|
+
targetId: input.targetId,
|
|
491
|
+
modelId: input.modelId,
|
|
492
|
+
evidence: input.evidence,
|
|
493
|
+
...(input.telemetry ? { telemetry: input.telemetry } : {}),
|
|
494
|
+
}, registerCompletionTool);
|
|
495
|
+
}
|
|
496
|
+
else {
|
|
497
|
+
if (!input.metrics)
|
|
498
|
+
throw new Error(`${input.type} requires application-target metrics.`);
|
|
499
|
+
validateReportedMetrics(input.metrics, input.targetId);
|
|
500
|
+
await reportAutomationSignal(pi, ctx, state, store, {
|
|
501
|
+
protocolVersion: PROTOCOL_VERSION,
|
|
502
|
+
id: input.signalId,
|
|
503
|
+
type: input.type,
|
|
504
|
+
occurredAt: input.occurredAt ?? new Date().toISOString(),
|
|
505
|
+
targetId: input.targetId,
|
|
506
|
+
metrics: input.metrics,
|
|
507
|
+
}, registerCompletionTool);
|
|
508
|
+
}
|
|
509
|
+
return {
|
|
510
|
+
content: [{ type: "text", text: `Recorded OpenMerit automation signal ${input.signalId}.` }],
|
|
511
|
+
details: { signalId: input.signalId, type: input.type },
|
|
512
|
+
};
|
|
513
|
+
},
|
|
514
|
+
});
|
|
515
|
+
pi.on("input", (event, ctx) => {
|
|
516
|
+
if (ctx.mode !== "tui" || !state.activeIntent || event.source !== "interactive" || ctx.isIdle())
|
|
517
|
+
return;
|
|
518
|
+
queuedUserInputs.push({ text: event.text, ...(event.images ? { images: event.images } : {}) });
|
|
519
|
+
ctx.ui.notify(`openmerit: queued your request until active setup work finishes (${state.activeIntent})`, "info");
|
|
520
|
+
return { action: "handled" };
|
|
521
|
+
});
|
|
522
|
+
// A fresh demo can start in an empty directory. Re-scan only after Pi has
|
|
523
|
+
// settled the user's build turn, so setup sees a complete application call
|
|
524
|
+
// instead of racing partially written files.
|
|
525
|
+
pi.on("turn_end", async (_event, ctx) => {
|
|
526
|
+
if (!ctx.hasUI || state.proactiveNudgesPaused || state.activeIntent)
|
|
527
|
+
return;
|
|
528
|
+
store ??= new ProjectStore(ctx.cwd);
|
|
529
|
+
const projectState = await store.readState();
|
|
530
|
+
if (projectState?.cycle.targetId || (projectState && projectState.cycle.stage !== "unconfigured"))
|
|
531
|
+
return;
|
|
532
|
+
const detectedApplicationTargets = detectApplicationLlmTargets(ctx.cwd);
|
|
533
|
+
if (!detectedApplicationTargets.length)
|
|
534
|
+
return;
|
|
535
|
+
ctx.ui.notify(`openmerit: detected ${detectedApplicationTargets.length} application LLM target candidate(s) after the build; starting evidence setup`, "info");
|
|
536
|
+
await dispatchIntent(pi, ctx, state, store, "establish_evals", registerCompletionTool, {
|
|
537
|
+
applicationTargetRequired: true,
|
|
538
|
+
detectedApplicationTargets: detectedTargetsConstraint(detectedApplicationTargets),
|
|
539
|
+
...(detectedApplicationTargets.length === 1 && detectedApplicationTargets[0].configuredModelId ? { expectedIncumbentModelId: detectedApplicationTargets[0].configuredModelId } : {}),
|
|
540
|
+
});
|
|
541
|
+
});
|
|
542
|
+
pi.on("session_start", async (_event, ctx) => {
|
|
543
|
+
store = new ProjectStore(ctx.cwd);
|
|
544
|
+
let projectState = await store.readState();
|
|
545
|
+
if (projectState) {
|
|
546
|
+
state.lifecycleStage = projectState.cycle.stage;
|
|
547
|
+
state.proactiveNudgesPaused = projectState.automation.paused === true;
|
|
548
|
+
state.activeIntent = projectState.activeIntent?.id ?? null;
|
|
549
|
+
if (projectState.activeIntent) {
|
|
550
|
+
registerCompletionTool(projectState.activeIntent.kind);
|
|
551
|
+
if (ctx.hasUI || ctx.mode === "rpc") {
|
|
552
|
+
pi.sendMessage({
|
|
553
|
+
customType: INTENT_MESSAGE_TYPE,
|
|
554
|
+
content: privateIntentMessage(projectState.activeIntent),
|
|
555
|
+
display: false,
|
|
556
|
+
details: { intentId: projectState.activeIntent.id, kind: projectState.activeIntent.kind, resumed: true },
|
|
557
|
+
}, { deliverAs: "followUp", triggerTurn: true });
|
|
558
|
+
}
|
|
559
|
+
}
|
|
560
|
+
state.applicationTarget = projectState.cycle.targetId ? "confirmed" : "not_detected";
|
|
561
|
+
state.taskProfile = projectState.cycle.taskProfileId ? "configured" : "not_configured";
|
|
562
|
+
state.metricReadiness = projectState.observability?.status === "ready" ? "assessed" : "pending";
|
|
563
|
+
state.evaluationReadiness = projectState.cycle.taskProfileId ? "assessed" : "not_assessed";
|
|
564
|
+
state.incumbentCandidate = projectState.cycle.activeCandidateId ?? null;
|
|
565
|
+
state.candidateCount = projectState.candidates?.length ?? 0;
|
|
566
|
+
state.assessmentCount = projectState.assessments?.length ?? 0;
|
|
567
|
+
state.evaluationSpend = projectState.automation.evaluationSpend ?? 0;
|
|
568
|
+
state.proposedCandidate = projectState.cycle.proposedCandidateId ?? null;
|
|
569
|
+
if (!projectState.cycle.targetId) {
|
|
570
|
+
if (ctx.hasUI && !state.proactiveNudgesPaused && projectState.cycle.stage === "unconfigured") {
|
|
571
|
+
const detectedApplicationTargets = detectApplicationLlmTargets(ctx.cwd);
|
|
572
|
+
if (detectedApplicationTargets.length) {
|
|
573
|
+
await dispatchIntent(pi, ctx, state, store, "establish_evals", registerCompletionTool, {
|
|
574
|
+
applicationTargetRequired: true,
|
|
575
|
+
detectedApplicationTargets: detectedTargetsConstraint(detectedApplicationTargets),
|
|
576
|
+
...(detectedApplicationTargets.length === 1 && detectedApplicationTargets[0].configuredModelId ? { expectedIncumbentModelId: detectedApplicationTargets[0].configuredModelId } : {}),
|
|
577
|
+
});
|
|
578
|
+
}
|
|
579
|
+
}
|
|
580
|
+
else if (state.proactiveNudgesPaused) {
|
|
581
|
+
ctx.ui.notify("openmerit: proactive checks remain paused; use /openmerit resume to re-enable them", "info");
|
|
582
|
+
}
|
|
583
|
+
else {
|
|
584
|
+
ctx.ui.notify("openmerit: existing state has no confirmed application LLM target; use /openmerit setup", "warning");
|
|
585
|
+
}
|
|
586
|
+
return;
|
|
587
|
+
}
|
|
588
|
+
if (state.proactiveNudgesPaused || projectState.cycle.policy?.checkPolicy.execution.mode !== "persistent" ||
|
|
589
|
+
!projectState.cycle.policy?.checkPolicy.execution.infrastructureChangesAllowed) {
|
|
590
|
+
try {
|
|
591
|
+
await removePiWakeup(ctx.cwd);
|
|
592
|
+
}
|
|
593
|
+
catch (error) {
|
|
594
|
+
ctx.ui.notify(`openmerit: could not remove prior wakeup: ${error instanceof Error ? error.message : String(error)}`, "warning");
|
|
595
|
+
}
|
|
596
|
+
}
|
|
597
|
+
const automation = await new OpenMeritCoordinator(store, piHarnessAdapter(pi, ctx)).reconcileAutomation();
|
|
598
|
+
state.nextAssessment = automation.plan.nextWakeupAt ??
|
|
599
|
+
(automation.plan.scheduling === "external_scheduler_required" ? "external scheduler required" : "signal driven");
|
|
600
|
+
if (automation.plan.gaps.length) {
|
|
601
|
+
ctx.ui.notify(`openmerit automation: ${automation.plan.gaps.join(" ")}`, "warning");
|
|
602
|
+
}
|
|
603
|
+
const fingerprint = modelCatalogFingerprint(ctx);
|
|
604
|
+
await reportAutomationSignal(pi, ctx, state, store, {
|
|
605
|
+
protocolVersion: PROTOCOL_VERSION,
|
|
606
|
+
id: `signal-${crypto.randomUUID()}`,
|
|
607
|
+
type: "model_catalog_changed",
|
|
608
|
+
occurredAt: new Date().toISOString(),
|
|
609
|
+
targetId: projectState.cycle.targetId,
|
|
610
|
+
fingerprint,
|
|
611
|
+
}, registerCompletionTool);
|
|
612
|
+
if (!state.activeIntent) {
|
|
613
|
+
await reportAutomationSignal(pi, ctx, state, store, {
|
|
614
|
+
protocolVersion: PROTOCOL_VERSION,
|
|
615
|
+
id: `signal-${crypto.randomUUID()}`,
|
|
616
|
+
type: "scheduled_tick",
|
|
617
|
+
occurredAt: new Date().toISOString(),
|
|
618
|
+
targetId: projectState.cycle.targetId,
|
|
619
|
+
}, registerCompletionTool);
|
|
620
|
+
}
|
|
621
|
+
}
|
|
622
|
+
else if (ctx.hasUI && !state.proactiveNudgesPaused) {
|
|
623
|
+
const detectedApplicationTargets = detectApplicationLlmTargets(ctx.cwd);
|
|
624
|
+
if (detectedApplicationTargets.length) {
|
|
625
|
+
ctx.ui.notify(`openmerit: detected ${detectedApplicationTargets.length} application LLM target candidate(s); asking Pi to confirm the target before setup`, "info");
|
|
626
|
+
await dispatchIntent(pi, ctx, state, store, "establish_evals", registerCompletionTool, {
|
|
627
|
+
applicationTargetRequired: true,
|
|
628
|
+
detectedApplicationTargets: detectedTargetsConstraint(detectedApplicationTargets),
|
|
629
|
+
...(detectedApplicationTargets.length === 1 && detectedApplicationTargets[0].configuredModelId ? { expectedIncumbentModelId: detectedApplicationTargets[0].configuredModelId } : {}),
|
|
630
|
+
});
|
|
631
|
+
}
|
|
632
|
+
else {
|
|
633
|
+
ctx.ui.notify("openmerit: no application LLM call detected yet; automatic setup will check again after Pi finishes a build turn (/openmerit setup is available for recovery)", "info");
|
|
634
|
+
}
|
|
635
|
+
}
|
|
636
|
+
if (projectState?.activeIntent) {
|
|
637
|
+
ctx.ui.notify(`openmerit: recovered interrupted work (${projectState.activeIntent.id}); use /openmerit cancel to abandon it`, "warning");
|
|
638
|
+
}
|
|
639
|
+
});
|
|
640
|
+
pi.registerCommand("openmerit", {
|
|
641
|
+
description: "OpenMerit: status, setup, assess, frontier, approve, cancel, pause/resume, logs, doctor",
|
|
642
|
+
handler: async (args, ctx) => {
|
|
643
|
+
const subcommand = args.trim().split(/\s+/)[0]?.toLowerCase() || "status";
|
|
644
|
+
switch (subcommand) {
|
|
645
|
+
case "status":
|
|
646
|
+
ctx.ui.notify(renderStatus(state), "info");
|
|
647
|
+
return;
|
|
648
|
+
case "setup":
|
|
649
|
+
store ??= new ProjectStore(ctx.cwd);
|
|
650
|
+
{
|
|
651
|
+
const detectedApplicationTargets = detectApplicationLlmTargets(ctx.cwd);
|
|
652
|
+
await dispatchIntent(pi, ctx, state, store, "establish_evals", registerCompletionTool, {
|
|
653
|
+
applicationTargetRequired: true,
|
|
654
|
+
detectedApplicationTargets: detectedTargetsConstraint(detectedApplicationTargets),
|
|
655
|
+
...(detectedApplicationTargets.length === 1 && detectedApplicationTargets[0].configuredModelId ? { expectedIncumbentModelId: detectedApplicationTargets[0].configuredModelId } : {}),
|
|
656
|
+
});
|
|
657
|
+
}
|
|
658
|
+
return;
|
|
659
|
+
case "assess":
|
|
660
|
+
store ??= new ProjectStore(ctx.cwd);
|
|
661
|
+
{
|
|
662
|
+
const assessment = await new OpenMeritCoordinator(store, piHarnessAdapter(pi, ctx)).assessBaseline();
|
|
663
|
+
state.lifecycleStage = assessment.state.cycle.stage;
|
|
664
|
+
ctx.ui.notify(assessment.errors.length
|
|
665
|
+
? `openmerit: baseline is not ready: ${assessment.errors.join("; ")}`
|
|
666
|
+
: "openmerit: baseline meets the required metric thresholds; starting candidate discovery", assessment.errors.length ? "warning" : "info");
|
|
667
|
+
if (assessment.decision.action === "issue_intent") {
|
|
668
|
+
await dispatchIntent(pi, ctx, state, store, assessment.decision.intentKind, registerCompletionTool);
|
|
669
|
+
}
|
|
670
|
+
}
|
|
671
|
+
return;
|
|
672
|
+
case "frontier":
|
|
673
|
+
store ??= new ProjectStore(ctx.cwd);
|
|
674
|
+
await dispatchIntent(pi, ctx, state, store, "calculate_frontier", registerCompletionTool);
|
|
675
|
+
return;
|
|
676
|
+
case "approve": {
|
|
677
|
+
store ??= new ProjectStore(ctx.cwd);
|
|
678
|
+
const projectState = await store.readState();
|
|
679
|
+
if (!projectState?.cycle.proposedCandidateId || projectState.cycle.stage !== "frontier_ready") {
|
|
680
|
+
ctx.ui.notify("openmerit: there is no verified model proposal awaiting approval", "warning");
|
|
681
|
+
return;
|
|
682
|
+
}
|
|
683
|
+
const now = new Date().toISOString();
|
|
684
|
+
await store.writeState({
|
|
685
|
+
...projectState,
|
|
686
|
+
cycle: { ...projectState.cycle, stage: "swap_approved" },
|
|
687
|
+
updatedAt: now,
|
|
688
|
+
});
|
|
689
|
+
await store.appendEvent({
|
|
690
|
+
schemaVersion: 1,
|
|
691
|
+
id: `event-${crypto.randomUUID()}`,
|
|
692
|
+
type: "cycle_advanced",
|
|
693
|
+
occurredAt: now,
|
|
694
|
+
cycleId: projectState.cycle.id,
|
|
695
|
+
data: { stage: "swap_approved", candidateId: projectState.cycle.proposedCandidateId },
|
|
696
|
+
});
|
|
697
|
+
state.lifecycleStage = "swap_approved";
|
|
698
|
+
await dispatchIntent(pi, ctx, state, store, "apply_model_swap", registerCompletionTool);
|
|
699
|
+
return;
|
|
700
|
+
}
|
|
701
|
+
case "cancel": {
|
|
702
|
+
store ??= new ProjectStore(ctx.cwd);
|
|
703
|
+
const projectState = await store.readState();
|
|
704
|
+
if (!projectState?.activeIntent) {
|
|
705
|
+
ctx.ui.notify("openmerit: there is no active work to cancel", "warning");
|
|
706
|
+
return;
|
|
707
|
+
}
|
|
708
|
+
const cancelled = await new OpenMeritCoordinator(store, piHarnessAdapter(pi, ctx)).acceptResult({
|
|
709
|
+
protocolVersion: PROTOCOL_VERSION,
|
|
710
|
+
intentId: projectState.activeIntent.id,
|
|
711
|
+
status: "cancelled",
|
|
712
|
+
completedAt: new Date().toISOString(),
|
|
713
|
+
summary: "Cancelled explicitly by the user after an interrupted harness run.",
|
|
714
|
+
evidence: [],
|
|
715
|
+
});
|
|
716
|
+
state.activeIntent = null;
|
|
717
|
+
state.lifecycleStage = cancelled.state.cycle.stage;
|
|
718
|
+
registerCompletionTool();
|
|
719
|
+
releaseQueuedUserInput();
|
|
720
|
+
ctx.ui.notify(`openmerit: cancelled ${projectState.activeIntent.id}`, "info");
|
|
721
|
+
return;
|
|
722
|
+
}
|
|
723
|
+
case "pause": {
|
|
724
|
+
store ??= new ProjectStore(ctx.cwd);
|
|
725
|
+
const existing = await store.readState();
|
|
726
|
+
const now = new Date().toISOString();
|
|
727
|
+
const paused = existing
|
|
728
|
+
? { ...existing, automation: { ...existing.automation, paused: true }, updatedAt: now }
|
|
729
|
+
: {
|
|
730
|
+
schemaVersion: 1,
|
|
731
|
+
cycle: { id: `cycle-${crypto.randomUUID()}`, stage: "unconfigured", verifiedSwapCount: 0, baselineEvidenceSufficient: false },
|
|
732
|
+
automation: { setupNudgeShown: false, observedCompletedTasks: 0, lastAssessmentTaskCount: 0, paused: true },
|
|
733
|
+
updatedAt: now,
|
|
734
|
+
};
|
|
735
|
+
await store.writeState(paused);
|
|
736
|
+
state.proactiveNudgesPaused = true;
|
|
737
|
+
state.nextAssessment = "paused";
|
|
738
|
+
try {
|
|
739
|
+
await removePiWakeup(ctx.cwd);
|
|
740
|
+
}
|
|
741
|
+
catch (error) {
|
|
742
|
+
ctx.ui.notify(`openmerit: could not unload wakeup: ${error instanceof Error ? error.message : String(error)}`, "warning");
|
|
743
|
+
}
|
|
744
|
+
ctx.ui.notify("openmerit: proactive nudges paused", "info");
|
|
745
|
+
return;
|
|
746
|
+
}
|
|
747
|
+
case "resume": {
|
|
748
|
+
store ??= new ProjectStore(ctx.cwd);
|
|
749
|
+
const existing = await store.readState();
|
|
750
|
+
if (existing) {
|
|
751
|
+
await store.writeState({ ...existing, automation: { ...existing.automation, paused: false }, updatedAt: new Date().toISOString() });
|
|
752
|
+
}
|
|
753
|
+
state.proactiveNudgesPaused = false;
|
|
754
|
+
if (existing?.cycle.targetId) {
|
|
755
|
+
const automation = await new OpenMeritCoordinator(store, piHarnessAdapter(pi, ctx)).reconcileAutomation();
|
|
756
|
+
state.nextAssessment = automation.plan.nextWakeupAt ??
|
|
757
|
+
(automation.plan.scheduling === "external_scheduler_required" ? "external scheduler required" : "signal driven");
|
|
758
|
+
if (automation.plan.gaps.length) {
|
|
759
|
+
ctx.ui.notify(`openmerit automation: ${automation.plan.gaps.join(" ")}`, "warning");
|
|
760
|
+
}
|
|
761
|
+
}
|
|
762
|
+
ctx.ui.notify("openmerit: proactive nudges resumed", "info");
|
|
763
|
+
return;
|
|
764
|
+
}
|
|
765
|
+
case "doctor":
|
|
766
|
+
ctx.ui.notify([
|
|
767
|
+
"OpenMerit doctor",
|
|
768
|
+
"Pi adapter : loaded",
|
|
769
|
+
"intent executor : connected",
|
|
770
|
+
`audit log : ${store?.eventsPath ?? `${ctx.cwd}/.openmerit/events.jsonl`}`,
|
|
771
|
+
"Pi model mutation : prohibited",
|
|
772
|
+
"application swap : target-bound after approval",
|
|
773
|
+
`persistent schedule: ${process.platform === "darwin" ? "opt-in LaunchAgent when policy permits" : "unavailable on this platform"}`,
|
|
774
|
+
`schedule file : ${piWakeupPlistPath(ctx.cwd)}`,
|
|
775
|
+
`schedule status : ${ctx.cwd}/.openmerit/scheduler-status.json`,
|
|
776
|
+
`interactive UI : ${ctx.hasUI ? "available" : "unavailable"}`,
|
|
777
|
+
].join("\n"), "info");
|
|
778
|
+
return;
|
|
779
|
+
case "logs":
|
|
780
|
+
store ??= new ProjectStore(ctx.cwd);
|
|
781
|
+
ctx.ui.notify([
|
|
782
|
+
"OpenMerit observability files",
|
|
783
|
+
`audit events : ${store.eventsPath}`,
|
|
784
|
+
`export errors: ${store.exportErrorsPath}`,
|
|
785
|
+
`evidence : ${store.evidenceDirectory}`,
|
|
786
|
+
`state : ${store.statePath}`,
|
|
787
|
+
].join("\n"), "info");
|
|
788
|
+
return;
|
|
789
|
+
default:
|
|
790
|
+
ctx.ui.notify("openmerit: usage: /openmerit [status|setup|assess|frontier|approve|cancel|pause|resume|logs|doctor]", "warning");
|
|
791
|
+
}
|
|
792
|
+
},
|
|
793
|
+
});
|
|
794
|
+
}
|