skill-harness 0.26.2 → 0.27.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (2) hide show
  1. package/dist/index.js +135 -10
  2. package/package.json +55 -55
package/dist/index.js CHANGED
@@ -16924,6 +16924,88 @@ async function writePreparedExport({ prepared, directory }) {
16924
16924
  return { directory, trainingEligible: prepared.metadata.trainingEligible, files: Object.fromEntries(Object.entries(prepared.files).map(([name, s]) => [name, { sha256: learningDigest(s), bytes: Buffer.byteLength(s) }])) };
16925
16925
  }
16926
16926
 
16927
+ // experiments/decision-shadow/workflow-fixtures.mjs
16928
+ var WORKFLOW_ORACLE_VERSION = "workflow-convergence-oracles-v1";
16929
+ function evaluateWorkflowFixture(family, facts) {
16930
+ if (!facts || typeof facts !== "object" || Array.isArray(facts)) return false;
16931
+ switch (family) {
16932
+ case "workspace":
16933
+ return typeof facts.requestedRoot === "string" && facts.requestedRoot.startsWith("/") && facts.observedRoot === facts.requestedRoot && facts.pathsObserved === true && Array.isArray(facts.requiredPaths) && facts.requiredPaths.length > 0 && Array.isArray(facts.readablePaths) && facts.requiredPaths.every((path) => typeof path === "string" && path.startsWith(facts.requestedRoot + "/") && facts.readablePaths.includes(path));
16934
+ case "identity":
16935
+ return facts.expected?.algorithm === "principal-candidate-v1" && facts.actual?.algorithm === facts.expected.algorithm && typeof facts.expected.id === "string" && /^[a-f0-9]{64}$/.test(facts.expected.id) && facts.actual.id === facts.expected.id;
16936
+ case "evidence":
16937
+ return facts.candidateMatched === true && facts.referencesVerified === true && Array.isArray(facts.obligations) && facts.obligations.some((item) => item?.due === "review") && facts.obligations.every((item) => item && ["review", "finish"].includes(item.due) && typeof item.id === "string" && item.id.length > 0 && (item.due === "review" ? item.state === "verified" : ["pending", "verified"].includes(item.state)));
16938
+ case "routing": {
16939
+ if (facts.observationComplete !== true || !Array.isArray(facts.productFindings) || !Array.isArray(facts.dueEvidenceGaps)) return false;
16940
+ const expected = facts.productFindings.length ? "build" : facts.dueEvidenceGaps.length ? "evidence" : "git-ops";
16941
+ return facts.next === expected;
16942
+ }
16943
+ case "reuse":
16944
+ return facts.operation === "static-inspection" && facts.purpose === "same-check" && facts.previous?.state === "complete" && facts.previous.outputsVerified === true && typeof facts.currentInput === "string" && /^[a-f0-9]{64}$/.test(facts.currentInput) && facts.previous.input === facts.currentInput && facts.previous.environment === facts.currentEnvironment && typeof facts.currentEnvironment === "string" && facts.currentEnvironment.length > 0;
16945
+ default:
16946
+ throw new TypeError("Unknown workflow fixture family");
16947
+ }
16948
+ }
16949
+ function workflowFixtureFamilies() {
16950
+ const root = "/fixture/pilot", hash4 = "a".repeat(64);
16951
+ const workspace = {
16952
+ requestedRoot: root,
16953
+ observedRoot: root,
16954
+ pathsObserved: true,
16955
+ requiredPaths: [`${root}/report.md`],
16956
+ readablePaths: [`${root}/report.md`]
16957
+ };
16958
+ const identity2 = {
16959
+ expected: { algorithm: "principal-candidate-v1", id: hash4 },
16960
+ actual: { algorithm: "principal-candidate-v1", id: hash4 }
16961
+ };
16962
+ const evidence = {
16963
+ candidateMatched: true,
16964
+ referencesVerified: true,
16965
+ obligations: [{ id: "behavior-check", due: "review", state: "verified" }, { id: "archive", due: "finish", state: "pending" }]
16966
+ };
16967
+ const routing = { observationComplete: true, productFindings: [], dueEvidenceGaps: ["missing-report"], next: "evidence" };
16968
+ const reuse = {
16969
+ operation: "static-inspection",
16970
+ purpose: "same-check",
16971
+ currentInput: hash4,
16972
+ currentEnvironment: "node-fixture-v1",
16973
+ previous: { state: "complete", input: hash4, environment: "node-fixture-v1", outputsVerified: true }
16974
+ };
16975
+ return [
16976
+ {
16977
+ family: "workspace",
16978
+ split: "train",
16979
+ question: "Do these explicitly observed roots and readable absolute paths establish access to the requested fixture workspace? Missing access evidence is not a permission-denial finding.",
16980
+ facts: [workspace, { ...workspace, observedRoot: "/fixture/original" }, { ...workspace, readablePaths: [] }, { ...workspace, pathsObserved: false }]
16981
+ },
16982
+ {
16983
+ family: "identity",
16984
+ split: "train",
16985
+ question: "Do these records contain matching full candidate IDs under the same principal-candidate-v1 algorithm? A diff hash or another algorithm is not an interchangeable candidate ID.",
16986
+ facts: [identity2, { ...identity2, actual: { algorithm: "git-diff-sha256", id: hash4 } }, { ...identity2, actual: { algorithm: "principal-candidate-v1", id: "b".repeat(64) } }, { expected: { algorithm: "principal-candidate-v1", id: "aaaaaaa" }, actual: { algorithm: "principal-candidate-v1", id: "aaaaaaa" } }]
16987
+ },
16988
+ {
16989
+ family: "evidence",
16990
+ split: "validation",
16991
+ question: "Are all review-due obligations verified by matching-candidate, verified references? Finish-due obligations may remain pending; a new runtime replay is not required by this fixture rule.",
16992
+ facts: [evidence, { ...evidence, obligations: [{ id: "behavior-check", due: "review", state: "pending" }] }, { ...evidence, candidateMatched: false }, { ...evidence, referencesVerified: false }]
16993
+ },
16994
+ {
16995
+ family: "routing",
16996
+ split: "test",
16997
+ question: "Does Next follow this explicit routing rule: observed product findings -> build, otherwise due evidence gaps -> evidence, otherwise git-ops? Incomplete observations cannot establish routing.",
16998
+ facts: [routing, { ...routing, next: "build" }, { ...routing, productFindings: ["blocked-api-shutdown"], next: "build" }, { ...routing, observationComplete: false }]
16999
+ },
17000
+ {
17001
+ family: "reuse",
17002
+ split: "test",
17003
+ question: "May this completed static inspection be reused for the same check, unchanged input/environment and verified outputs? A new independent review or live runtime observation requires a distinct operation.",
17004
+ facts: [reuse, { ...reuse, previous: { ...reuse.previous, outputsVerified: false } }, { ...reuse, purpose: "independent-review" }, { ...reuse, operation: "live-database-observation" }]
17005
+ }
17006
+ ];
17007
+ }
17008
+
16927
17009
  // experiments/decision-shadow/learning-fixtures.mjs
16928
17010
  var FIXTURE_ORACLE_VERSION = "mechanical-fixture-oracles-v1";
16929
17011
  var COMMIT = /^[a-f0-9]{40}$/;
@@ -16964,9 +17046,10 @@ function evaluateFixture(family, e) {
16964
17046
  throw new TypeError("Unknown fixture oracle family");
16965
17047
  }
16966
17048
  }
16967
- function createLearningFixtures({ recordedAt = "2026-10-08T00:00:00.000Z" } = {}) {
17049
+ function createLearningFixtures({ recordedAt = "2026-10-08T00:00:00.000Z", set: set2 = "mechanical" } = {}) {
17050
+ if (!["mechanical", "workflow"].includes(set2)) throw new TypeError("Fixture set must be mechanical or workflow");
16968
17051
  const commit = "a".repeat(40), complete = Object.fromEntries(PHASES.map((p) => [p, "complete"]));
16969
- const families = [
17052
+ const mechanicalFamilies = [
16970
17053
  { family: "candidate", split: "train", question: "Do the records establish one exact full 40-character lowercase Git candidate shared by candidate, tested and reviewed?", facts: [{ candidate: commit, tested: commit, reviewed: commit }, { candidate: commit, tested: "b".repeat(40), reviewed: commit }, { candidate: commit, tested: commit, reviewed: "c".repeat(40) }, { candidate: "aaaaaaa", tested: "aaaaaaa", reviewed: "aaaaaaa" }] },
16971
17054
  { family: "phases", split: "train", question: "Do the latest snapshots cover exactly every required step with all five named phases complete? Earlier complete phases cannot fill later omissions.", facts: [{ required: ["step-1"], records: [{ step: "step-1", phases: complete }] }, { required: ["step-1", "step-2"], records: [{ step: "step-1", phases: complete }] }, { required: ["step-1"], records: [{ step: "step-1", phases: complete }, { step: "step-extra", phases: complete }] }, { required: ["step-1"], records: [{ step: "step-1", phases: complete }, { step: "step-1", phases: { ...complete, verified: "unknown" } }] }] },
16972
17055
  { family: "findings", split: "train", question: "Does the observed finding history establish that every finding identity (step, source, id) has an explicit latest verified disposition? Omission never clears an earlier finding.", facts: [{ historyObserved: true, records: [{ step: "step-1", findings: [] }] }, { historyObserved: true, records: [{ step: "step-1", findings: [{ id: "f1", source: "report-a", status: "open" }] }, { step: "step-1", findings: [{ id: "f1", source: "report-a", status: "verified" }] }] }, { historyObserved: true, records: [{ step: "step-1", findings: [{ id: "f1", source: "report-a", status: "open" }] }, { step: "step-1", findings: [] }] }, { historyObserved: true, records: [{ step: "step-1", findings: [{ id: "f1", source: "report-a", status: "open" }] }, { step: "step-1", findings: [{ id: "f1", source: "report-b", status: "verified" }] }] }] },
@@ -16974,15 +17057,20 @@ function createLearningFixtures({ recordedAt = "2026-10-08T00:00:00.000Z" } = {}
16974
17057
  { family: "report", split: "test", question: "Does this structured requested delivery comply with the selected saved-full-report/five-line-summary protocol, including its explicit unavailable-persistence blocked exception?", facts: [{ protocol: "saved-full-report-and-five-line-summary", persistence: "available", requestedDelivery: "saved-full-report-and-five-line-summary", reportReference: "required" }, { protocol: "saved-full-report-and-five-line-summary", persistence: "available", requestedDelivery: "complete-report-only-in-final", reportReference: "forbidden" }, { protocol: "saved-full-report-and-five-line-summary", persistence: "available", requestedDelivery: "saved-full-report-and-five-line-summary", reportReference: "omitted" }, { protocol: "saved-full-report-and-five-line-summary", persistence: "unavailable", requestedDelivery: "blocked-no-invented-report", reportReference: "not-saved" }] },
16975
17058
  { family: "cleanup", split: "test", question: "Do these records establish settled cleanup with a matching execution identity and reapedAll true? This does not establish task success or approval.", facts: [{ executionId: "exec:one", state: "settled", receipt: { state: "settled", executionId: "exec:one", reapedAll: true } }, { executionId: "exec:one", state: "settled", receipt: null }, { executionId: "exec:one", state: "settled", receipt: { state: "settled", executionId: "exec:other", reapedAll: true } }, { executionId: "exec:one", state: "settled", receipt: { state: "settled", executionId: "exec:one", reapedAll: false } }] }
16976
17059
  ];
17060
+ const families = set2 === "workflow" ? workflowFixtureFamilies() : mechanicalFamilies;
17061
+ const evaluate = set2 === "workflow" ? evaluateWorkflowFixture : evaluateFixture;
17062
+ const oracleVersion = set2 === "workflow" ? WORKFLOW_ORACLE_VERSION : FIXTURE_ORACLE_VERSION;
17063
+ const experimentId = set2 === "workflow" ? "workflow-convergence-20-v1" : "mechanical-fixtures-24-v1";
17064
+ const scope = set2 === "workflow" ? "20 synthetic workflow convergence cases; no observed sessions or model-quality claims." : "24 synthetic mechanical plumbing cases; no observed examples or model-quality claims.";
16977
17065
  const caseDocument = { schema: 1, cases: families.flatMap((f) => f.facts.map((facts, i) => {
16978
17066
  const input = JSON.stringify(facts);
16979
- return { id: `fixture-${f.family}-${i + 1}`, input, question: f.question, provenance: "synthetic", source: { sha256: learningDigest(input), recordId: `${FIXTURE_ORACLE_VERSION}/${f.family}/${i + 1}` }, visibility: "public" };
17067
+ return { id: `fixture-${f.family}-${i + 1}`, input, question: f.question, provenance: "synthetic", source: { sha256: learningDigest(input), recordId: `${oracleVersion}/${f.family}/${i + 1}` }, visibility: "public" };
16980
17068
  })) };
16981
17069
  const cases = parseCases(caseDocument), receipts = [], labels = [], entries = [];
16982
17070
  for (const c of cases) {
16983
17071
  const family = c.id.split("-")[1], f = families.find((f2) => f2.family === family);
16984
- const value = evaluateFixture(family, JSON.parse(c.input));
16985
- const receipt = { schema: 1, kind: "independent-decision-label", caseId: c.id, caseHash: c.hash, source: c.source, value, labelKind: "test", actor: FIXTURE_ORACLE_VERSION, independent: true, recordedAt, method: { kind: "deterministic-test", id: family, version: FIXTURE_ORACLE_VERSION + ":" + learningDigest(evaluateFixture.toString()) } };
17072
+ const value = evaluate(family, JSON.parse(c.input));
17073
+ const receipt = { schema: 1, kind: "independent-decision-label", caseId: c.id, caseHash: c.hash, source: c.source, value, labelKind: "test", actor: oracleVersion, independent: true, recordedAt, method: { kind: "deterministic-test", id: family, version: oracleVersion + ":" + learningDigest(evaluate.toString()) } };
16986
17074
  const bytes = JSON.stringify(receipt, null, 2) + "\n";
16987
17075
  receipts.push({ caseId: c.id, filename: c.id + "-label.json", bytes, sha256: learningDigest(bytes) });
16988
17076
  labels.push({ caseId: c.id, caseHash: c.hash, value, kind: "test", actor: receipt.actor, evidenceSha256: learningDigest(bytes), independent: true });
@@ -16990,8 +17078,8 @@ function createLearningFixtures({ recordedAt = "2026-10-08T00:00:00.000Z" } = {}
16990
17078
  }
16991
17079
  const labelDocument = { schema: 1, labels };
16992
17080
  parseLabels(labelDocument, cases);
16993
- const manifest = parseExperiment(cases, { schema: 1, kind: "decision-learning-experiment", id: "mechanical-fixtures-24-v1", frozenAt: recordedAt, entries });
16994
- return { caseDocument, labelDocument, manifest, labelReceipts: receipts, fixtureOnly: true, trainingEligible: false, scope: "24 synthetic mechanical plumbing cases; no observed examples or model-quality claims." };
17081
+ const manifest = parseExperiment(cases, { schema: 1, kind: "decision-learning-experiment", id: experimentId, frozenAt: recordedAt, entries });
17082
+ return { caseDocument, labelDocument, manifest, labelReceipts: receipts, fixtureOnly: true, trainingEligible: false, scope };
16995
17083
  }
16996
17084
 
16997
17085
  // experiments/decision-shadow/research.mjs
@@ -17436,13 +17524,14 @@ var RESEARCH_OPTIONS = {
17436
17524
  fixtures: ["out"]
17437
17525
  };
17438
17526
  var RESEARCH_OPTIONAL = {
17527
+ fixtures: ["set"],
17439
17528
  "run-pi": ["pi-node", "auth-path", "timeout-ms"]
17440
17529
  };
17441
17530
  async function researchCommand(command, flags, helpers, options) {
17442
17531
  const { cases, jsonFile: jsonFile2, textFile: textFile2, writeNew: writeNew3, readLegacyRun } = helpers;
17443
17532
  const { emit = console.log, piRunner } = options;
17444
17533
  if (command === "fixtures") {
17445
- const f = createLearningFixtures({});
17534
+ const f = createLearningFixtures({ set: flags.set ?? "mechanical" });
17446
17535
  const dir = resolve15(flags.out);
17447
17536
  await mkdir4(dir, { mode: 448 });
17448
17537
  const refs = [];
@@ -18326,7 +18415,7 @@ var help = `Decision research workflow (Node >=20; qualified Pi execution needs
18326
18415
  score --cases FILE --labels FILE --run RESULT.jsonl [--run RESULT2.jsonl]
18327
18416
  corpus --cases FILE --labels FILE --out NEW.json
18328
18417
  verify-sources --cases FILE --sources SELECTION.json --evidence-root DIR --out NEW.json
18329
- fixtures --out NEW_DIR
18418
+ fixtures --out NEW_DIR [--set mechanical|workflow]
18330
18419
  validate-experiment --cases FILE --experiment FILE
18331
18420
  preview-pi --cases FILE --experiment FILE --model openai-codex:MODEL --thinking LEVEL --split test
18332
18421
  run-pi --cases FILE --experiment FILE --model openai-codex:MODEL --thinking LEVEL --split test --pi-package DIR --out NEW.jsonl --allow-subscription [--pi-node PATH] [--auth-path FILE] [--timeout-ms N]
@@ -19132,6 +19221,41 @@ function createJevController(pi, options = {}) {
19132
19221
 
19133
19222
  // packages/pi-extension/src/jev-advice.ts
19134
19223
  import { Type as Type3 } from "typebox";
19224
+ var nullableNumber = Type3.Union([Type3.Number(), Type3.Null()]);
19225
+ var nullableString = Type3.Union([Type3.String(), Type3.Null()]);
19226
+ var adviceOutput = Type3.Union([
19227
+ Type3.Object({
19228
+ enabled: Type3.Boolean(),
19229
+ mode: Type3.Union([Type3.Literal("disabled"), Type3.Literal("manual"), Type3.Literal("workflow")]),
19230
+ storage: Type3.Union([Type3.Literal("granted"), Type3.Literal("declined"), Type3.Null()]),
19231
+ remaining: Type3.Integer({ minimum: 0 }),
19232
+ availability: Type3.String(),
19233
+ providerReadiness: Type3.Union([Type3.Literal("key-present"), Type3.Literal("missing-key")]),
19234
+ advisory: Type3.Literal(true)
19235
+ }, { additionalProperties: false }),
19236
+ Type3.Object({
19237
+ advisory: Type3.Literal(true),
19238
+ status: Type3.Union([Type3.Literal("answered"), Type3.Literal("unavailable")]),
19239
+ probability: Type3.Union([Type3.Number({ minimum: 0, maximum: 1 }), Type3.Null()]),
19240
+ provider: Type3.String(),
19241
+ requestedModel: Type3.String(),
19242
+ resolvedModel: nullableString,
19243
+ usage: Type3.Object({ inputTokens: nullableNumber, outputTokens: nullableNumber, costUsd: nullableNumber }, { additionalProperties: false }),
19244
+ latencyMs: nullableNumber,
19245
+ reason: Type3.Optional(Type3.String()),
19246
+ remaining: Type3.Integer({ minimum: 0 }),
19247
+ inputSha256: Type3.Optional(Type3.String()),
19248
+ reused: Type3.Optional(Type3.Boolean()),
19249
+ source: Type3.Optional(Type3.Object({
19250
+ kind: Type3.String(),
19251
+ sha256: Type3.String(),
19252
+ sessionId: Type3.String(),
19253
+ toolCallId: Type3.String(),
19254
+ retained: Type3.Boolean(),
19255
+ path: Type3.Optional(Type3.String())
19256
+ }, { additionalProperties: false }))
19257
+ }, { additionalProperties: false })
19258
+ ]);
19135
19259
  function createJevAdviceTool(controller) {
19136
19260
  return {
19137
19261
  name: "jev_advice",
@@ -19148,6 +19272,7 @@ function createJevAdviceTool(controller) {
19148
19272
  requirements: Type3.Optional(Type3.String({ minLength: 1, maxLength: 16e3, description: "Explicit acceptance checks, before seeing independent review or JEV outcomes." })),
19149
19273
  evidence: Type3.Optional(Type3.String({ minLength: 1, maxLength: 16e3, description: "Selected decision-time handoff evidence. Complete JSON packet is limited to 16000 characters." }))
19150
19274
  }, { additionalProperties: false }),
19275
+ outputSchema: adviceOutput,
19151
19276
  async execute(id, params, signal, _onUpdate, ctx) {
19152
19277
  requireIdentity(id, "Pi toolCallId");
19153
19278
  if (!params || typeof params !== "object" || Array.isArray(params)) throw new Error("Invalid jev_advice parameters");
@@ -19165,7 +19290,7 @@ function createJevAdviceTool(controller) {
19165
19290
  ctx
19166
19291
  );
19167
19292
  else throw new Error("jev_advice action must be status or evaluate");
19168
- return { content: [{ type: "text", text: JSON.stringify(details) }], details };
19293
+ return { content: [{ type: "text", text: JSON.stringify(details) }], details, structuredContent: details };
19169
19294
  }
19170
19295
  };
19171
19296
  }
package/package.json CHANGED
@@ -1,58 +1,58 @@
1
1
  {
2
- "name": "skill-harness",
3
- "version": "0.26.2",
4
- "description": "Test/optimize loop for agent skills \u2014 run spec'd scenarios on pi, LLM-judge, score, review, re-run",
5
- "type": "module",
6
- "keywords": [
7
- "agent-skills",
8
- "skill",
9
- "llm",
10
- "testing",
11
- "eval",
12
- "grading",
13
- "pi",
14
- "ci",
15
- "harness"
16
- ],
17
- "bin": {
18
- "skill-harness": "bin.js"
19
- },
20
- "pi": {
21
- "extensions": [
22
- "./dist/index.js"
23
- ]
24
- },
25
- "files": [
26
- "bin.js",
27
- "dist/index.js",
28
- "assets/report.template.html",
29
- "assets/report.grade.js",
30
- "LICENSE",
31
- "README.md"
32
- ],
33
- "dependencies": {
34
- "@skill-harness/cli": "0.26.2"
35
- },
36
- "peerDependencies": {
37
- "typebox": "*"
38
- },
39
- "peerDependenciesMeta": {
40
- "typebox": {
41
- "optional": true
2
+ "name": "skill-harness",
3
+ "version": "0.27.0",
4
+ "description": "Test/optimize loop for agent skills \u2014 run spec'd scenarios on pi, LLM-judge, score, review, re-run",
5
+ "type": "module",
6
+ "keywords": [
7
+ "agent-skills",
8
+ "skill",
9
+ "llm",
10
+ "testing",
11
+ "eval",
12
+ "grading",
13
+ "pi",
14
+ "ci",
15
+ "harness"
16
+ ],
17
+ "bin": {
18
+ "skill-harness": "bin.js"
19
+ },
20
+ "pi": {
21
+ "extensions": [
22
+ "./dist/index.js"
23
+ ]
24
+ },
25
+ "files": [
26
+ "bin.js",
27
+ "dist/index.js",
28
+ "assets/report.template.html",
29
+ "assets/report.grade.js",
30
+ "LICENSE",
31
+ "README.md"
32
+ ],
33
+ "dependencies": {
34
+ "@skill-harness/cli": "0.27.0"
35
+ },
36
+ "peerDependencies": {
37
+ "typebox": "*"
38
+ },
39
+ "peerDependenciesMeta": {
40
+ "typebox": {
41
+ "optional": true
42
+ }
43
+ },
44
+ "devDependencies": {
45
+ "typebox": "^1.1.38"
46
+ },
47
+ "repository": {
48
+ "type": "git",
49
+ "url": "git+https://github.com/mojomanyana/skill-harness.git"
50
+ },
51
+ "license": "MIT",
52
+ "engines": {
53
+ "node": ">=20"
54
+ },
55
+ "scripts": {
56
+ "prepack": "node ../../scripts/release-pack.mjs --guard"
42
57
  }
43
- },
44
- "devDependencies": {
45
- "typebox": "^1.1.38"
46
- },
47
- "repository": {
48
- "type": "git",
49
- "url": "git+https://github.com/mojomanyana/skill-harness.git"
50
- },
51
- "license": "MIT",
52
- "engines": {
53
- "node": ">=20"
54
- },
55
- "scripts": {
56
- "prepack": "node ../../scripts/release-pack.mjs --guard"
57
- }
58
58
  }