opencode-agent-skill 9.0.0 → 11.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +116 -0
- package/README.md +742 -675
- package/bin/ocskill.mjs +354 -5
- package/docs/V11-PERCEPTION-ADAPTIVE-EXECUTION.md +75 -0
- package/docs/V11-PERCEPTION-ADAPTIVE.md +220 -0
- package/evals/router-triggers.json +82 -0
- package/evals/routing.json +76 -0
- package/evals/v11/tasks.json +122 -0
- package/global-config/AGENTS.md +78 -163
- package/global-config/agents/merge-arbiter.md +12 -0
- package/global-config/agents/visual-verifier.md +12 -0
- package/global-config/plugins/ues-router/index.js +683 -59
- package/global-config/plugins/ues-router/router.js +62 -3
- package/global-config/plugins/ues-router/runtime-guard.js +265 -0
- package/global-config/skills/browser-qa/SKILL.md +14 -0
- package/global-config/skills/browser-qa/references/workflow.md +11 -0
- package/global-config/skills/browser-security/SKILL.md +12 -0
- package/global-config/skills/component-visual-testing/SKILL.md +10 -0
- package/global-config/skills/design-source/SKILL.md +10 -0
- package/global-config/skills/design-source/references/workflow.md +12 -0
- package/global-config/skills/dynamic-workflow/SKILL.md +18 -0
- package/global-config/skills/dynamic-workflow/references/workflow.md +19 -0
- package/global-config/skills/responsive-verification/SKILL.md +10 -0
- package/global-config/skills/skill-authoring/SKILL.md +12 -0
- package/global-config/skills/skill-evaluation/SKILL.md +17 -0
- package/global-config/skills/visual-fidelity/SKILL.md +14 -0
- package/global-config/skills/visual-fidelity/references/workflow.md +14 -0
- package/lib/benchmark-confidence.mjs +49 -11
- package/lib/browser-adapter.mjs +82 -0
- package/lib/browser-runtime.mjs +193 -0
- package/lib/capability-registry.mjs +109 -0
- package/lib/context-engine-v11.mjs +146 -0
- package/lib/context-manifest.mjs +16 -3
- package/lib/control-center.mjs +12 -2
- package/lib/dynamic-workflow.mjs +179 -0
- package/lib/eval-ablation.mjs +146 -0
- package/lib/eval-report.mjs +83 -0
- package/lib/eval-telemetry.mjs +64 -0
- package/lib/evidence-budget.mjs +84 -0
- package/lib/evidence-store.mjs +178 -0
- package/lib/hermes-bridge.mjs +45 -1
- package/lib/model-config.mjs +9 -1
- package/lib/model-policy.mjs +58 -1
- package/lib/orchestrator-policy.mjs +100 -7
- package/lib/png-diff.mjs +229 -0
- package/lib/prompt-cache.mjs +60 -0
- package/lib/skill-quality.mjs +72 -0
- package/lib/task-engine.mjs +223 -12
- package/lib/ui-inspector.mjs +152 -0
- package/lib/v11-metrics.mjs +64 -0
- package/lib/visual-spec.mjs +159 -0
- package/package.json +11 -5
- package/scripts/eval-ablation.mjs +47 -0
- package/scripts/eval-matrix.mjs +13 -2
- package/scripts/validate-v11-suite.mjs +58 -0
- package/scripts/validate.mjs +16 -4
package/lib/task-engine.mjs
CHANGED
|
@@ -9,10 +9,15 @@ import { relevantAcceptedLearnings } from "./learning-engine.mjs"
|
|
|
9
9
|
import { validateVerificationReceipt } from "./evidence-receipt.mjs"
|
|
10
10
|
import { appendRuntimeEvent, readRuntimeEvents } from "./runtime-events.mjs"
|
|
11
11
|
import { createGateReceipt, validateGateReceipt } from "./gate-receipt.mjs"
|
|
12
|
-
import { classifyEngineeringTask } from "./orchestrator-policy.mjs"
|
|
12
|
+
import { classifyEngineeringTask, recoveryPolicyForAttempt } from "./orchestrator-policy.mjs"
|
|
13
|
+
import { planEvidenceBudget } from "./evidence-budget.mjs"
|
|
14
|
+
import { putEvidence } from "./evidence-store.mjs"
|
|
15
|
+
import { inferTaskCapabilities } from "./capability-registry.mjs"
|
|
16
|
+
import { buildPromptEnvelope } from "./prompt-cache.mjs"
|
|
17
|
+
import { externalizeContextExcerpts } from "./context-engine-v11.mjs"
|
|
13
18
|
|
|
14
19
|
const WORK_DIR = ".ues-work"
|
|
15
|
-
const STATE_SCHEMA =
|
|
20
|
+
const STATE_SCHEMA = 4
|
|
16
21
|
const LOCK_TIMEOUT_MS = 30_000
|
|
17
22
|
const LOCK_STALE_MS = 120_000
|
|
18
23
|
const LOCK_HEARTBEAT_MS = 15_000
|
|
@@ -284,6 +289,7 @@ export async function initWork(root, slug, goal) {
|
|
|
284
289
|
tasks: {},
|
|
285
290
|
decisions: [],
|
|
286
291
|
blockers: [],
|
|
292
|
+
checkpoint: null,
|
|
287
293
|
nextAction: "Complete SPEC.md, create PLAN.json, then run ocskill work plan.",
|
|
288
294
|
})
|
|
289
295
|
await writeJson(paths.evidence, {
|
|
@@ -387,6 +393,7 @@ export async function importPlan(root, slug, planInput) {
|
|
|
387
393
|
planHash: hash,
|
|
388
394
|
planApproval: { status: "pending", planHash: hash, at: updatedAt, evidence: null },
|
|
389
395
|
integrationVerification: null,
|
|
396
|
+
checkpoint: null,
|
|
390
397
|
evidencePolicy: evidencePolicyForPlan(plan),
|
|
391
398
|
tasks: stateTasks,
|
|
392
399
|
nextAction: "Run ues-plan-checker, then record PASS with 'ocskill work approve-plan'.",
|
|
@@ -489,7 +496,7 @@ export function readyTasks(plan, state) {
|
|
|
489
496
|
const ready = []
|
|
490
497
|
for (const task of plan.tasks) {
|
|
491
498
|
const status = state.tasks?.[task.id]?.status
|
|
492
|
-
if (!["pending", "failed"].includes(status)) continue
|
|
499
|
+
if (!["pending", "failed", "retryable"].includes(status)) continue
|
|
493
500
|
if (!(task.dependsOn || []).every((dep) => completed.has(dep))) continue
|
|
494
501
|
const files = taskFiles(task)
|
|
495
502
|
if (files.length === 0 && runningFiles.size > 0) continue
|
|
@@ -501,7 +508,7 @@ export function readyTasks(plan, state) {
|
|
|
501
508
|
|
|
502
509
|
export async function workStatus(root, slug) {
|
|
503
510
|
const loaded = await loadWork(root, slug)
|
|
504
|
-
const counts = { pending: 0, running: 0, completed: 0, failed: 0 }
|
|
511
|
+
const counts = { pending: 0, running: 0, completed: 0, failed: 0, retryable: 0 }
|
|
505
512
|
for (const value of Object.values(loaded.state.tasks || {})) {
|
|
506
513
|
if (Object.hasOwn(counts, value.status)) counts[value.status] += 1
|
|
507
514
|
}
|
|
@@ -541,6 +548,7 @@ export async function workStatus(root, slug) {
|
|
|
541
548
|
gateReceipts: (loaded.evidence.gateReceipts || []).length,
|
|
542
549
|
},
|
|
543
550
|
nextAction: loaded.state.nextAction,
|
|
551
|
+
checkpoint: loaded.state.checkpoint || null,
|
|
544
552
|
updatedAt: loaded.state.updatedAt,
|
|
545
553
|
}
|
|
546
554
|
}
|
|
@@ -673,7 +681,7 @@ export async function recoverTask(root, slug, taskID, options = {}) {
|
|
|
673
681
|
const timestamp = now()
|
|
674
682
|
const previousOwner = record.owner ? { ...record.owner } : null
|
|
675
683
|
const previousRunId = record.runId || null
|
|
676
|
-
record.status = "
|
|
684
|
+
record.status = "retryable"
|
|
677
685
|
record.lastFailure = String(options.reason || "stale executor lease recovered after interruption")
|
|
678
686
|
record.lastOwner = previousOwner
|
|
679
687
|
record.lastRunId = previousRunId
|
|
@@ -709,7 +717,7 @@ export async function recoverStaleTasks(root, slug, options = {}) {
|
|
|
709
717
|
if (!stale) continue
|
|
710
718
|
const previousOwner = record.owner ? { ...record.owner } : null
|
|
711
719
|
const previousRunId = record.runId || null
|
|
712
|
-
record.status = "
|
|
720
|
+
record.status = "retryable"
|
|
713
721
|
record.lastFailure = "stale executor lease recovered after interruption"
|
|
714
722
|
record.lastOwner = previousOwner
|
|
715
723
|
record.lastRunId = previousRunId
|
|
@@ -1065,11 +1073,19 @@ async function buildContextPack(loaded, taskID) {
|
|
|
1065
1073
|
const task = taskByID(loaded.plan, taskID)
|
|
1066
1074
|
const spec = await readFile(loaded.paths.spec, "utf8").catch(() => "")
|
|
1067
1075
|
const dependencyReports = {}
|
|
1076
|
+
const dependencyReportRefs = {}
|
|
1068
1077
|
|
|
1069
1078
|
for (const dep of task.dependsOn || []) {
|
|
1070
1079
|
const file = path.join(loaded.paths.reports, `${dep}.md`)
|
|
1071
1080
|
if (!existsSync(file)) continue
|
|
1072
|
-
|
|
1081
|
+
const report = await readFile(file, "utf8")
|
|
1082
|
+
dependencyReports[dep] = report.slice(0, 6000)
|
|
1083
|
+
const stored = await putEvidence(loaded.paths.root, report, {
|
|
1084
|
+
kind: "dependency-report",
|
|
1085
|
+
source: path.relative(loaded.paths.root, file).replaceAll("\\", "/"),
|
|
1086
|
+
summary: `Dependency report for ${dep}`,
|
|
1087
|
+
}).catch(() => null)
|
|
1088
|
+
if (stored?.ref) dependencyReportRefs[dep] = stored.ref
|
|
1073
1089
|
}
|
|
1074
1090
|
|
|
1075
1091
|
const taskText = [
|
|
@@ -1082,26 +1098,98 @@ async function buildContextPack(loaded, taskID) {
|
|
|
1082
1098
|
changedFiles: taskFiles(task).length,
|
|
1083
1099
|
risk: task.risk,
|
|
1084
1100
|
})
|
|
1085
|
-
const
|
|
1101
|
+
const attempt = Math.max(1, Number(loaded.state.tasks?.[taskID]?.attempts || 1))
|
|
1102
|
+
const recovery = recoveryPolicyForAttempt(contextPolicy, attempt)
|
|
1103
|
+
const effectiveContextPolicy = {
|
|
1104
|
+
...contextPolicy,
|
|
1105
|
+
contextBudget: recovery.contextBudget,
|
|
1106
|
+
maxSkills: recovery.maxSkills,
|
|
1107
|
+
effectiveContextBudget: recovery.contextBudget,
|
|
1108
|
+
effectiveMaxSkills: recovery.maxSkills,
|
|
1109
|
+
recovery,
|
|
1110
|
+
}
|
|
1111
|
+
const capabilities = inferTaskCapabilities(taskText, {
|
|
1112
|
+
longContext: contextPolicy.mode === "long-horizon",
|
|
1113
|
+
coding: true,
|
|
1114
|
+
toolCalling: true,
|
|
1115
|
+
filesystem: true,
|
|
1116
|
+
})
|
|
1117
|
+
const evidenceBudget = planEvidenceBudget(effectiveContextPolicy, task, capabilities.required)
|
|
1118
|
+
const rawContextManifest = await buildContextManifest(
|
|
1086
1119
|
loaded.paths.root,
|
|
1087
1120
|
task,
|
|
1088
1121
|
{
|
|
1089
|
-
budget:
|
|
1090
|
-
strategy:
|
|
1122
|
+
budget: evidenceBudget.total,
|
|
1123
|
+
strategy: recovery.contextStrategy,
|
|
1124
|
+
evidenceBudget,
|
|
1091
1125
|
},
|
|
1092
1126
|
).catch(() => null)
|
|
1127
|
+
const externalizedContext = rawContextManifest
|
|
1128
|
+
? await externalizeContextExcerpts(loaded.paths.root, rawContextManifest, {
|
|
1129
|
+
task,
|
|
1130
|
+
threshold: Math.max(1800, Math.round(evidenceBudget.total * 0.12)),
|
|
1131
|
+
inlineChars: Math.max(600, Math.min(1600, Math.round(evidenceBudget.total * 0.08))),
|
|
1132
|
+
}).catch(() => ({ manifest: rawContextManifest, externalized: [], externalizedBytes: 0 }))
|
|
1133
|
+
: { manifest: null, externalized: [], externalizedBytes: 0 }
|
|
1134
|
+
const contextManifest = externalizedContext.manifest
|
|
1093
1135
|
const learnings = await relevantAcceptedLearnings(
|
|
1094
1136
|
loaded.paths.root,
|
|
1095
1137
|
[task.title, task.summary, ...(task.acceptance || [])].join(" "),
|
|
1096
1138
|
).catch(() => [])
|
|
1097
1139
|
|
|
1140
|
+
const specStored = spec
|
|
1141
|
+
? await putEvidence(loaded.paths.root, spec, {
|
|
1142
|
+
kind: "work-spec",
|
|
1143
|
+
source: path.relative(loaded.paths.root, loaded.paths.spec).replaceAll("\\", "/"),
|
|
1144
|
+
summary: "Durable work specification",
|
|
1145
|
+
}).catch(() => null)
|
|
1146
|
+
: null
|
|
1147
|
+
const promptEnvelope = buildPromptEnvelope({
|
|
1148
|
+
role: "executor",
|
|
1149
|
+
invariants: "evidence-first; scoped edits; fresh verification; no unsupported completion claims",
|
|
1150
|
+
skills: contextPolicy.domains || [],
|
|
1151
|
+
projectFacts: {
|
|
1152
|
+
instructions: contextManifest?.instructions || [],
|
|
1153
|
+
strategy: recovery.contextStrategy,
|
|
1154
|
+
},
|
|
1155
|
+
task,
|
|
1156
|
+
evidence: [
|
|
1157
|
+
...(specStored?.ref ? [specStored.ref] : []),
|
|
1158
|
+
...Object.values(dependencyReportRefs),
|
|
1159
|
+
...(contextManifest?.evidencePointers || []).map((item) => item.ref),
|
|
1160
|
+
...(contextManifest?.rankedReferences || []).slice(0, 12).map((item) => item.path),
|
|
1161
|
+
],
|
|
1162
|
+
recentFailure: recovery.stage === "initial" ? null : loaded.state.tasks?.[taskID]?.lastError || null,
|
|
1163
|
+
nextAction: loaded.state.checkpoint?.nextAction || loaded.state.nextAction || null,
|
|
1164
|
+
})
|
|
1165
|
+
|
|
1098
1166
|
return {
|
|
1099
1167
|
schemaVersion: STATE_SCHEMA,
|
|
1100
|
-
|
|
1168
|
+
contextSchemaVersion: 6,
|
|
1169
|
+
contextPolicy: effectiveContextPolicy,
|
|
1170
|
+
capabilities,
|
|
1171
|
+
evidenceBudget,
|
|
1172
|
+
promptCache: {
|
|
1173
|
+
stablePrefixHash: promptEnvelope.stablePrefixHash,
|
|
1174
|
+
dynamicHash: promptEnvelope.dynamicHash,
|
|
1175
|
+
stableChars: promptEnvelope.stableChars,
|
|
1176
|
+
dynamicChars: promptEnvelope.dynamicChars,
|
|
1177
|
+
cacheableRatio: promptEnvelope.cacheableRatio,
|
|
1178
|
+
},
|
|
1179
|
+
attempt,
|
|
1101
1180
|
slug: loaded.state.slug,
|
|
1102
1181
|
task,
|
|
1103
1182
|
taskBrief: path.relative(loaded.paths.root, path.join(loaded.paths.tasks, `${taskID}.md`)).replaceAll("\\", "/"),
|
|
1104
|
-
spec: spec.slice(0,
|
|
1183
|
+
spec: spec.slice(0, Math.min(12000, evidenceBudget.buckets.instructions + evidenceBudget.buckets.task)),
|
|
1184
|
+
evidencePointers: {
|
|
1185
|
+
spec: specStored?.ref || null,
|
|
1186
|
+
dependencyReports: dependencyReportRefs,
|
|
1187
|
+
context: (contextManifest?.evidencePointers || []).map((item) => item.ref),
|
|
1188
|
+
},
|
|
1189
|
+
evidenceStore: {
|
|
1190
|
+
refs: externalizedContext.externalized.length,
|
|
1191
|
+
externalizedBytes: externalizedContext.externalizedBytes,
|
|
1192
|
+
},
|
|
1105
1193
|
dependencyReports,
|
|
1106
1194
|
decisions: loaded.state.decisions || [],
|
|
1107
1195
|
blockers: loaded.state.blockers || [],
|
|
@@ -1111,6 +1199,7 @@ async function buildContextPack(loaded, taskID) {
|
|
|
1111
1199
|
workingState: {
|
|
1112
1200
|
status: loaded.state.status,
|
|
1113
1201
|
task: loaded.state.tasks?.[taskID],
|
|
1202
|
+
checkpoint: loaded.state.checkpoint || null,
|
|
1114
1203
|
},
|
|
1115
1204
|
}
|
|
1116
1205
|
}
|
|
@@ -1151,6 +1240,128 @@ export async function contextPack(root, slug, taskID) {
|
|
|
1151
1240
|
return buildContextPack(await loadWork(root, slug), taskID)
|
|
1152
1241
|
}
|
|
1153
1242
|
|
|
1243
|
+
function checkpointEvidencePointers(evidence = {}) {
|
|
1244
|
+
const pointers = []
|
|
1245
|
+
for (const entry of (evidence.entries || []).slice(-12)) {
|
|
1246
|
+
pointers.push({
|
|
1247
|
+
kind: "evidence-entry",
|
|
1248
|
+
task: entry.task || null,
|
|
1249
|
+
at: entry.at || entry.recordedAt || null,
|
|
1250
|
+
runId: entry.runId || null,
|
|
1251
|
+
})
|
|
1252
|
+
}
|
|
1253
|
+
for (const receipt of (evidence.receipts || []).slice(-12)) {
|
|
1254
|
+
pointers.push({
|
|
1255
|
+
kind: "verification-receipt",
|
|
1256
|
+
id: receipt.id || null,
|
|
1257
|
+
task: receipt.task || null,
|
|
1258
|
+
runId: receipt.runId || null,
|
|
1259
|
+
verdict: receipt.verdict || null,
|
|
1260
|
+
})
|
|
1261
|
+
}
|
|
1262
|
+
for (const receipt of (evidence.gateReceipts || []).slice(-8)) {
|
|
1263
|
+
pointers.push({
|
|
1264
|
+
kind: "gate-receipt",
|
|
1265
|
+
id: receipt.id || null,
|
|
1266
|
+
receiptKind: receipt.kind || null,
|
|
1267
|
+
verdict: receipt.verdict || null,
|
|
1268
|
+
})
|
|
1269
|
+
}
|
|
1270
|
+
return pointers.slice(-24)
|
|
1271
|
+
}
|
|
1272
|
+
|
|
1273
|
+
function checkpointNextAction(loaded, taskID, runId) {
|
|
1274
|
+
const record = taskID ? loaded.state.tasks?.[taskID] : null
|
|
1275
|
+
if (record?.status === "running") {
|
|
1276
|
+
return {
|
|
1277
|
+
type: "continue-task",
|
|
1278
|
+
taskId: taskID,
|
|
1279
|
+
runId: runId || record.runId || null,
|
|
1280
|
+
instruction: "Continue the active scoped task from durable evidence; execute the next required tool/action before narrative summary.",
|
|
1281
|
+
}
|
|
1282
|
+
}
|
|
1283
|
+
|
|
1284
|
+
const ready = readyTasks(loaded.plan, loaded.state)
|
|
1285
|
+
if (ready.length) {
|
|
1286
|
+
const nextTask = ready[0]
|
|
1287
|
+
return {
|
|
1288
|
+
type: "dispatch-task",
|
|
1289
|
+
taskId: nextTask,
|
|
1290
|
+
attempt: Number(loaded.state.tasks?.[nextTask]?.attempts || 0) + 1,
|
|
1291
|
+
instruction: "Dispatch the next ready task in a fresh executor from .ues-work.",
|
|
1292
|
+
}
|
|
1293
|
+
}
|
|
1294
|
+
|
|
1295
|
+
return {
|
|
1296
|
+
type: "inspect-state",
|
|
1297
|
+
instruction: loaded.state.nextAction || "Inspect durable work state and choose the next deterministic action.",
|
|
1298
|
+
}
|
|
1299
|
+
}
|
|
1300
|
+
|
|
1301
|
+
export async function checkpointWork(root, slug, options = {}) {
|
|
1302
|
+
return withWorkLock(root, slug, async () => {
|
|
1303
|
+
const loaded = await loadWork(root, slug)
|
|
1304
|
+
const taskID = String(options.taskID || "").trim() ||
|
|
1305
|
+
Object.entries(loaded.state.tasks || {}).find(([, value]) => value.status === "running")?.[0] ||
|
|
1306
|
+
null
|
|
1307
|
+
const record = taskID ? loaded.state.tasks?.[taskID] : null
|
|
1308
|
+
if (taskID && !record) throw new Error(`task '${taskID}' is not tracked`)
|
|
1309
|
+
if (record?.status === "running") assertRunFence(record, { runId: options.runId || null })
|
|
1310
|
+
|
|
1311
|
+
const timestamp = now()
|
|
1312
|
+
const checkpoint = {
|
|
1313
|
+
schemaVersion: 1,
|
|
1314
|
+
reason: String(options.reason || "runtime-checkpoint"),
|
|
1315
|
+
createdAt: timestamp,
|
|
1316
|
+
currentTaskId: taskID,
|
|
1317
|
+
runId: record?.runId || options.runId || null,
|
|
1318
|
+
planHash: loaded.plan ? planHash(loaded.plan) : loaded.state.planHash || null,
|
|
1319
|
+
nextAction: checkpointNextAction(loaded, taskID, record?.runId || options.runId || null),
|
|
1320
|
+
workspaceFingerprint: workspaceFingerprint(loaded.paths.root),
|
|
1321
|
+
evidencePointers: checkpointEvidencePointers(loaded.evidence),
|
|
1322
|
+
resumedAt: null,
|
|
1323
|
+
}
|
|
1324
|
+
|
|
1325
|
+
loaded.state.schemaVersion = STATE_SCHEMA
|
|
1326
|
+
loaded.state.checkpoint = checkpoint
|
|
1327
|
+
loaded.state.updatedAt = timestamp
|
|
1328
|
+
await writeJson(loaded.paths.state, loaded.state)
|
|
1329
|
+
await journal(loaded.paths, "work.checkpoint", {
|
|
1330
|
+
task: taskID,
|
|
1331
|
+
runId: checkpoint.runId,
|
|
1332
|
+
planHash: checkpoint.planHash,
|
|
1333
|
+
reason: checkpoint.reason,
|
|
1334
|
+
nextAction: checkpoint.nextAction,
|
|
1335
|
+
})
|
|
1336
|
+
return checkpoint
|
|
1337
|
+
})
|
|
1338
|
+
}
|
|
1339
|
+
|
|
1340
|
+
export async function markCheckpointResumed(root, slug, options = {}) {
|
|
1341
|
+
return withWorkLock(root, slug, async () => {
|
|
1342
|
+
const loaded = await loadWork(root, slug)
|
|
1343
|
+
const checkpoint = loaded.state.checkpoint
|
|
1344
|
+
if (!checkpoint) return null
|
|
1345
|
+
if (options.taskID && checkpoint.currentTaskId && checkpoint.currentTaskId !== options.taskID) {
|
|
1346
|
+
throw new Error("checkpoint task mismatch")
|
|
1347
|
+
}
|
|
1348
|
+
if (checkpoint.runId && options.runId && checkpoint.runId !== options.runId) {
|
|
1349
|
+
throw new Error("checkpoint runId mismatch")
|
|
1350
|
+
}
|
|
1351
|
+
const timestamp = now()
|
|
1352
|
+
checkpoint.resumedAt = timestamp
|
|
1353
|
+
checkpoint.resumeReason = String(options.reason || "deterministic-action-observed")
|
|
1354
|
+
loaded.state.updatedAt = timestamp
|
|
1355
|
+
await writeJson(loaded.paths.state, loaded.state)
|
|
1356
|
+
await journal(loaded.paths, "work.checkpoint-resumed", {
|
|
1357
|
+
task: checkpoint.currentTaskId || null,
|
|
1358
|
+
runId: checkpoint.runId || null,
|
|
1359
|
+
reason: checkpoint.resumeReason,
|
|
1360
|
+
})
|
|
1361
|
+
return checkpoint
|
|
1362
|
+
})
|
|
1363
|
+
}
|
|
1364
|
+
|
|
1154
1365
|
export async function runtimeEvents(root, slug, options = {}) {
|
|
1155
1366
|
const paths = workPaths(root, slug)
|
|
1156
1367
|
return readRuntimeEvents(paths.events, options)
|
|
@@ -0,0 +1,152 @@
|
|
|
1
|
+
function uniq(values) {
|
|
2
|
+
return [...new Set(values.filter((value) => value !== null && value !== undefined && value !== ""))]
|
|
3
|
+
}
|
|
4
|
+
|
|
5
|
+
function round(value, digits = 3) {
|
|
6
|
+
const factor = 10 ** digits
|
|
7
|
+
return Math.round(Number(value) * factor) / factor
|
|
8
|
+
}
|
|
9
|
+
|
|
10
|
+
function normalizeBox(item = {}) {
|
|
11
|
+
const box = item.box && typeof item.box === "object" ? item.box : item
|
|
12
|
+
const x = Number(box.x)
|
|
13
|
+
const y = Number(box.y)
|
|
14
|
+
const width = Number(box.width)
|
|
15
|
+
const height = Number(box.height)
|
|
16
|
+
if (![x,y,width,height].every(Number.isFinite)) return null
|
|
17
|
+
return {
|
|
18
|
+
id: String(item.id || item.name || "").trim() || null,
|
|
19
|
+
role: item.role || null,
|
|
20
|
+
x, y, width, height,
|
|
21
|
+
right: x + width,
|
|
22
|
+
bottom: y + height,
|
|
23
|
+
}
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
function containsBox(a, b) {
|
|
27
|
+
return a.x <= b.x && a.y <= b.y && a.right >= b.right && a.bottom >= b.bottom
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
function intersectionArea(a, b) {
|
|
31
|
+
const width = Math.max(0, Math.min(a.right, b.right) - Math.max(a.x, b.x))
|
|
32
|
+
const height = Math.max(0, Math.min(a.bottom, b.bottom) - Math.max(a.y, b.y))
|
|
33
|
+
return width * height
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
export function inspectResponsiveLayout(items = [], viewport = {}, options = {}) {
|
|
37
|
+
const width = Math.max(1, Number(viewport.width || 0))
|
|
38
|
+
const height = Math.max(1, Number(viewport.height || 0))
|
|
39
|
+
const boxes = items.map(normalizeBox).filter(Boolean)
|
|
40
|
+
const issues = []
|
|
41
|
+
const minTouch = Math.max(1, Number(options.minTouchTarget || 44))
|
|
42
|
+
const touchRoles = new Set(["button","link","checkbox","radio","switch","tab","menuitem"])
|
|
43
|
+
|
|
44
|
+
for (const box of boxes) {
|
|
45
|
+
if (box.x < 0 || box.y < 0 || box.right > width || box.bottom > height) {
|
|
46
|
+
issues.push({
|
|
47
|
+
kind: "viewport-overflow",
|
|
48
|
+
id: box.id,
|
|
49
|
+
box,
|
|
50
|
+
viewport: { width, height },
|
|
51
|
+
})
|
|
52
|
+
}
|
|
53
|
+
if (touchRoles.has(String(box.role || "").toLowerCase()) && (box.width < minTouch || box.height < minTouch)) {
|
|
54
|
+
issues.push({
|
|
55
|
+
kind: "small-touch-target",
|
|
56
|
+
id: box.id,
|
|
57
|
+
role: box.role,
|
|
58
|
+
actual: { width: box.width, height: box.height },
|
|
59
|
+
minimum: minTouch,
|
|
60
|
+
})
|
|
61
|
+
}
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
const maxPairChecks = Math.max(0, Math.min(10000, Number(options.maxPairChecks || 3000)))
|
|
65
|
+
let checks = 0
|
|
66
|
+
for (let i = 0; i < boxes.length; i += 1) {
|
|
67
|
+
for (let j = i + 1; j < boxes.length && checks < maxPairChecks; j += 1) {
|
|
68
|
+
checks += 1
|
|
69
|
+
const a = boxes[i]
|
|
70
|
+
const b = boxes[j]
|
|
71
|
+
if (containsBox(a,b) || containsBox(b,a)) continue
|
|
72
|
+
const area = intersectionArea(a,b)
|
|
73
|
+
if (!area) continue
|
|
74
|
+
const smaller = Math.max(1, Math.min(a.width*a.height, b.width*b.height))
|
|
75
|
+
const ratio = area / smaller
|
|
76
|
+
if (ratio >= Number(options.overlapRatio ?? 0.15)) {
|
|
77
|
+
issues.push({
|
|
78
|
+
kind: "element-overlap",
|
|
79
|
+
a: a.id,
|
|
80
|
+
b: b.id,
|
|
81
|
+
intersectionArea: area,
|
|
82
|
+
smallerElementRatio: round(ratio),
|
|
83
|
+
})
|
|
84
|
+
}
|
|
85
|
+
}
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
return {
|
|
89
|
+
schemaVersion: 1,
|
|
90
|
+
viewport: { width, height },
|
|
91
|
+
elementCount: boxes.length,
|
|
92
|
+
pairChecks: checks,
|
|
93
|
+
verdict: issues.length ? "FAIL" : "PASS",
|
|
94
|
+
issues,
|
|
95
|
+
}
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
function normalizeCssValue(value) {
|
|
99
|
+
return String(value || "").trim().replace(/\s+/g, " ")
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
export function extractDesignTokens(css = "") {
|
|
103
|
+
const source = String(css || "")
|
|
104
|
+
const variables = {}
|
|
105
|
+
for (const match of source.matchAll(/--([a-zA-Z0-9_-]+)\s*:\s*([^;}{]+)\s*;/g)) {
|
|
106
|
+
variables["--" + match[1]] = normalizeCssValue(match[2])
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
const colors = uniq([
|
|
110
|
+
...source.matchAll(/#[0-9a-fA-F]{3,8}\b/g),
|
|
111
|
+
...source.matchAll(/\b(?:rgb|rgba|hsl|hsla)\([^)]*\)/g),
|
|
112
|
+
].map((match) => match[0].toLowerCase())).slice(0, 128)
|
|
113
|
+
|
|
114
|
+
const lengths = uniq([...source.matchAll(/-?\d*\.?\d+(?:px|rem|em)\b/g)].map((m)=>m[0]))
|
|
115
|
+
const px = lengths.filter((value)=>value.endsWith("px")).map((value)=>Number.parseFloat(value)).filter((value)=>Number.isFinite(value) && value >= 0)
|
|
116
|
+
const radii = uniq([...source.matchAll(/border-radius\s*:\s*([^;}{]+)/g)].map((m)=>normalizeCssValue(m[1]))).slice(0,64)
|
|
117
|
+
const fontSizes = uniq([...source.matchAll(/font-size\s*:\s*([^;}{]+)/g)].map((m)=>normalizeCssValue(m[1]))).slice(0,64)
|
|
118
|
+
const shadows = uniq([...source.matchAll(/box-shadow\s*:\s*([^;}{]+)/g)].map((m)=>normalizeCssValue(m[1]))).slice(0,64)
|
|
119
|
+
|
|
120
|
+
const spacingCandidates = uniq(px.filter((value)=>value <= 128).sort((a,b)=>a-b)).slice(0,32)
|
|
121
|
+
|
|
122
|
+
return {
|
|
123
|
+
schemaVersion: 1,
|
|
124
|
+
variables,
|
|
125
|
+
colors,
|
|
126
|
+
spacingCandidates,
|
|
127
|
+
radii,
|
|
128
|
+
fontSizes,
|
|
129
|
+
shadows,
|
|
130
|
+
stats: {
|
|
131
|
+
variableCount: Object.keys(variables).length,
|
|
132
|
+
colorCount: colors.length,
|
|
133
|
+
lengthCount: lengths.length,
|
|
134
|
+
},
|
|
135
|
+
}
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
export function designTokenEvidence(tokens = {}) {
|
|
139
|
+
const variableEntries = Object.entries(tokens.variables || {})
|
|
140
|
+
return {
|
|
141
|
+
schemaVersion: 1,
|
|
142
|
+
summary: {
|
|
143
|
+
variables: variableEntries.slice(0,40),
|
|
144
|
+
colors: (tokens.colors || []).slice(0,24),
|
|
145
|
+
spacingCandidates: (tokens.spacingCandidates || []).slice(0,20),
|
|
146
|
+
radii: (tokens.radii || []).slice(0,16),
|
|
147
|
+
fontSizes: (tokens.fontSizes || []).slice(0,16),
|
|
148
|
+
shadows: (tokens.shadows || []).slice(0,12),
|
|
149
|
+
},
|
|
150
|
+
stats: tokens.stats || {},
|
|
151
|
+
}
|
|
152
|
+
}
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
function finite(value) {
|
|
2
|
+
const n = Number(value)
|
|
3
|
+
return Number.isFinite(n) ? n : 0
|
|
4
|
+
}
|
|
5
|
+
|
|
6
|
+
function ratio(numerator, denominator) {
|
|
7
|
+
return denominator > 0 ? numerator / denominator : null
|
|
8
|
+
}
|
|
9
|
+
|
|
10
|
+
export function summarizeV11Efficiency(samples = []) {
|
|
11
|
+
const totals = {
|
|
12
|
+
samples: samples.length,
|
|
13
|
+
stableChars: 0,
|
|
14
|
+
dynamicChars: 0,
|
|
15
|
+
repeatedStableChars: 0,
|
|
16
|
+
externalizedEvidenceBytes: 0,
|
|
17
|
+
evidenceRefs: 0,
|
|
18
|
+
visualRepairAttempts: 0,
|
|
19
|
+
contextExpansions: 0,
|
|
20
|
+
modelEscalations: 0,
|
|
21
|
+
duplicateToolBlocks: 0,
|
|
22
|
+
loopBlocks: 0,
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
for (const sample of samples) {
|
|
26
|
+
const cache = sample.promptCache || sample.contextPack?.promptCache || {}
|
|
27
|
+
totals.stableChars += finite(cache.stableChars)
|
|
28
|
+
totals.dynamicChars += finite(cache.dynamicChars)
|
|
29
|
+
totals.repeatedStableChars += finite(sample.repeatedStableChars ?? cache.repeatedStableChars)
|
|
30
|
+
|
|
31
|
+
const evidence = sample.evidenceStore || sample.evidence || {}
|
|
32
|
+
totals.externalizedEvidenceBytes += finite(evidence.externalizedBytes ?? evidence.bytes)
|
|
33
|
+
totals.evidenceRefs += finite(evidence.refs ?? sample.evidenceRefs)
|
|
34
|
+
|
|
35
|
+
totals.visualRepairAttempts += finite(sample.visualRepairAttempts)
|
|
36
|
+
totals.contextExpansions += finite(sample.contextExpansions)
|
|
37
|
+
totals.modelEscalations += finite(sample.modelEscalations)
|
|
38
|
+
totals.duplicateToolBlocks += finite(sample.duplicateToolBlocks)
|
|
39
|
+
totals.loopBlocks += finite(sample.loopBlocks)
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
const inputChars = totals.stableChars + totals.dynamicChars
|
|
43
|
+
return {
|
|
44
|
+
schemaVersion: 1,
|
|
45
|
+
totals,
|
|
46
|
+
cacheablePrefixRatio: ratio(totals.stableChars, inputChars),
|
|
47
|
+
repeatedStableRatio: ratio(totals.repeatedStableChars, totals.stableChars),
|
|
48
|
+
externalizedEvidenceBytesPerSample: ratio(totals.externalizedEvidenceBytes, totals.samples),
|
|
49
|
+
evidenceRefsPerSample: ratio(totals.evidenceRefs, totals.samples),
|
|
50
|
+
visualRepairsPerSample: ratio(totals.visualRepairAttempts, totals.samples),
|
|
51
|
+
contextExpansionsPerSample: ratio(totals.contextExpansions, totals.samples),
|
|
52
|
+
modelEscalationsPerSample: ratio(totals.modelEscalations, totals.samples),
|
|
53
|
+
}
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
export function verifiedSuccessPer100kTokens(results = []) {
|
|
57
|
+
let success = 0
|
|
58
|
+
let tokens = 0
|
|
59
|
+
for (const item of results) {
|
|
60
|
+
if (item.passed === true && item.verified !== false) success += 1
|
|
61
|
+
tokens += finite(item.tokens ?? item.telemetry?.tokens?.total)
|
|
62
|
+
}
|
|
63
|
+
return tokens > 0 ? success / (tokens / 100_000) : null
|
|
64
|
+
}
|