skill-harness 0.25.0 → 0.25.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (2) hide show
  1. package/dist/index.js +50 -17
  2. package/package.json +3 -3
package/dist/index.js CHANGED
@@ -6293,7 +6293,15 @@ function withExecutionFailure(transcript, failure2) {
6293
6293
  ${transcript}` : transcript;
6294
6294
  }
6295
6295
  function executionFailureFromTranscript(transcript) {
6296
- return failureFromPreamble(transcript, EXECUTION_FAILURE_MARKER);
6296
+ const failure2 = failureFromPreamble(transcript, EXECUTION_FAILURE_MARKER);
6297
+ if (failure2 !== null)
6298
+ return failure2;
6299
+ const legacyPrefix = "[adapter failure] ";
6300
+ if (transcript.startsWith(legacyPrefix)) {
6301
+ const detail = transcript.slice(legacyPrefix.length).split("\n", 1)[0].trim();
6302
+ return `adapter failure \u2014 ${detail || "adapter threw without a message"}`;
6303
+ }
6304
+ return null;
6297
6305
  }
6298
6306
  function withProviderFailure(transcript, failure2) {
6299
6307
  return failure2 ? `${PROVIDER_FAILURE_MARKER} ${failure2}
@@ -7229,13 +7237,13 @@ async function runRep(scenario, rep, repCount, ctx) {
7229
7237
  }
7230
7238
  }
7231
7239
  } catch (e) {
7232
- adapterFailure = e instanceof Error ? e.message : String(e);
7240
+ adapterFailure = (e instanceof Error ? e.message : String(e)) || "adapter threw without a message";
7233
7241
  transcript = `[adapter failure] ${adapterFailure}`;
7234
7242
  gatePrefix = null;
7235
7243
  stagedDiff = null;
7236
7244
  traces = [];
7237
7245
  }
7238
- if (!infrastructureFailure) {
7246
+ if (!infrastructureFailure && adapterFailure === null) {
7239
7247
  const provider = providerFailureFromTranscript(transcript);
7240
7248
  if (provider)
7241
7249
  infrastructureFailure = `provider failure \u2014 ${provider}`;
@@ -7245,14 +7253,15 @@ async function runRep(scenario, rep, repCount, ctx) {
7245
7253
  infrastructureFailure = `execution failure \u2014 ${execution}`;
7246
7254
  }
7247
7255
  }
7248
- executionUnavailable ||= executionFailureFromTranscript(transcript) !== null;
7256
+ executionUnavailable ||= adapterFailure === null && executionFailureFromTranscript(transcript) !== null;
7249
7257
  const deliveredText = traces.length > 0 && traces.every((trace) => trace.final_status === "complete" && trace.final_text.length > 0 && !trace.capture_errors?.length);
7250
7258
  noResponse = !deliveredText && hasEmptyAssistantTurn(transcript);
7251
7259
  if (executionUnavailable || !noResponse && !adapterFailure)
7252
7260
  break;
7253
7261
  }
7254
- if (adapterFailure && !infrastructureFailure) {
7255
- infrastructureFailure = `adapter failure \u2014 ${adapterFailure}`;
7262
+ if (adapterFailure) {
7263
+ infrastructureFailure ??= `adapter failure \u2014 ${adapterFailure}`;
7264
+ transcript = withExecutionFailure(transcript, `adapter failure \u2014 ${adapterFailure}`);
7256
7265
  }
7257
7266
  }
7258
7267
  const repSuffix = repCount > 1 ? rep : void 0;
@@ -16381,10 +16390,20 @@ def ordinary(path):
16381
16390
  def closed(value, fields, name):
16382
16391
  require(isinstance(value, dict) and set(value) == set(fields), name + ' has unsupported or missing fields')
16383
16392
 
16393
+ def jsonl_rows(path):
16394
+ # JSONL records end only at LF. Unicode NEL/LS/PS are legal JSON string data.
16395
+ lines = ordinary(path).decode('utf-8').split('\n')
16396
+ if lines[-1] == '': lines.pop()
16397
+ return [json.loads(line) for line in lines]
16398
+
16384
16399
  def export_data(root):
16385
16400
  root = pathlib.Path(root).absolute()
16386
16401
  raw = ordinary(root / 'export-manifest.json'); m = json.loads(raw)
16387
- require(m.get('schema') == 1 and m.get('kind') == 'decision-learning-export', 'Unsupported export')
16402
+ require(m.get('schema') == 2 and m.get('kind') == 'decision-learning-export', 'Unsupported export schema; re-export with the current skill-harness')
16403
+ normalization = m['inputNormalization']
16404
+ closed(normalization, ['algorithm','runtime','nodeVersion','unicodeVersion'], 'Input normalization producer')
16405
+ require(normalization['algorithm'] == 'ecmascript-nfkc-whitespace-trim-lower-sha256-v1' and normalization['runtime'] == 'node', 'Unsupported input normalization producer')
16406
+ require(all(isinstance(normalization[key], str) and re.fullmatch(r'[0-9]+(?:\.[0-9]+){1,2}', normalization[key]) for key in ['nodeVersion','unicodeVersion']), 'Invalid input normalization producer version')
16388
16407
  require(m.get('trainingExecuted') is False and m.get('providerPredictionsIncluded') is False, 'Unsupported source provenance')
16389
16408
  require(set(m['files']) == {'train.jsonl','validation.jsonl','test.jsonl','train-lora.py','requirements-training.txt','training-config.example.json','TRAINING.md'}, 'Unexpected export files')
16390
16409
  for name, ref in m['files'].items():
@@ -16394,10 +16413,10 @@ def export_data(root):
16394
16413
  require(digest(ordinary(pathlib.Path(__file__).absolute())) == m['files']['train-lora.py']['sha256'], 'Run the exact exported training script')
16395
16414
  groups, seen_inputs, seen_cases = {}, {}, set()
16396
16415
  for split in ['train','validation','test']:
16397
- rows = [json.loads(line) for line in ordinary(root / (split+'.jsonl')).decode('utf-8').splitlines()]
16416
+ rows = jsonl_rows(root / (split+'.jsonl'))
16398
16417
  require(len(rows) == m['counts'][split], 'Split count mismatch')
16399
16418
  for row in rows:
16400
- closed(row, ['caseId','caseHash','taskGroup','lineageGroup','sessionId','fixtureOnly','input','question','answer','label','source'], 'Dataset row')
16419
+ closed(row, ['caseId','caseHash','taskGroup','lineageGroup','sessionId','fixtureOnly','input','inputIdentity','question','answer','label','source'], 'Dataset row')
16401
16420
  require(row['caseId'] not in seen_cases and type(row['answer']) is bool, 'Duplicate case or invalid label')
16402
16421
  seen_cases.add(row['caseId'])
16403
16422
  require(row['label']['caseId'] == row['caseId'] and row['label']['caseHash'] == row['caseHash'] and row['label']['value'] == row['answer'] and row['label']['kind'] in ['human','test'] and row['label']['independent'] is True, 'Label identity mismatch')
@@ -16406,7 +16425,15 @@ def export_data(root):
16406
16425
  identity = key + ':' + row[key]
16407
16426
  require(identity not in groups or groups[identity] == split, 'Related cases cross splits')
16408
16427
  groups[identity] = split
16409
- normalized = ' '.join(row['input'].casefold().split())
16428
+ identity = row['inputIdentity']
16429
+ closed(identity, ['algorithm','inputSha256','normalizedSha256'], 'Input identity receipt')
16430
+ require(identity['algorithm'] == normalization['algorithm'], 'Input identity algorithm mismatch')
16431
+ require(all(isinstance(identity[key], str) and re.fullmatch(r'[0-9a-f]{64}', identity[key]) for key in ['inputSha256','normalizedSha256']), 'Invalid input identity digest')
16432
+ require(isinstance(row['input'], str) and digest(row['input'].encode('utf-8')) == identity['inputSha256'], 'Input identity byte mismatch')
16433
+ # This is producer evidence bound by the exact row/file/manifest hashes.
16434
+ # Recomputing with Python Unicode tables can disagree with the Node producer.
16435
+ # The receipt is not an independent proof that normalization was correct.
16436
+ normalized = identity['normalizedSha256']
16410
16437
  require(normalized not in seen_inputs or seen_inputs[normalized] == split, 'Duplicate input crosses splits')
16411
16438
  seen_inputs[normalized] = split
16412
16439
  if m.get('trainingEligible') is True: require(row['fixtureOnly'] is False and row['sessionId'] is not None, 'Fixture/anonymous data cannot be training eligible')
@@ -16481,8 +16508,7 @@ def main():
16481
16508
  model = get_peft_model(model, LoraConfig(task_type=TaskType.CAUSAL_LM, r=t['loraRank'], lora_alpha=t['loraAlpha'], target_modules=t['targetModules'], lora_dropout=0.0, bias='none'))
16482
16509
  def rows(split):
16483
16510
  data = []
16484
- for line in ordinary(root / (split + '.jsonl')).decode('utf-8').splitlines():
16485
- row = json.loads(line)
16511
+ for row in jsonl_rows(root / (split + '.jsonl')):
16486
16512
  require(row['fixtureOnly'] is False and type(row['answer']) is bool and row['label']['kind'] in ['human','test'] and row['label']['independent'] is True, 'Only independently labeled reviewed real data may train')
16487
16513
  prompt = 'Question: ' + row['question'] + '\nEvidence:\n' + row['input'] + '\nAnswer (true or false): '
16488
16514
  prefix = tokenizer.encode(prompt, add_special_tokens=True)
@@ -16512,6 +16538,8 @@ var TRAINING_DOC = `# Local LoRA workflow
16512
16538
 
16513
16539
  This export contains independent labels, never JEV predictions. A fixture demonstration is not training eligible. No training or installation has been performed by exporting these files.
16514
16540
 
16541
+ Export schema 2 records the producer's normalized-input SHA-256 receipt in every row, the exact input UTF-8 SHA-256, and the Node/Unicode versions and normalization algorithm in the manifest. One JavaScript normalizer validates the split and writes the receipt. Python checks receipt shape, raw-input binding, duplicate identities and exact file hashes; it does not reinterpret those identities with its own Unicode tables. These are trusted producer receipts, not independent proofs of correct normalization. Separate training approval still pins the exact export manifest. Invalid Unicode scalar input is refused by the producer. Re-export older schema-1 packets with the current CLI; do not modify their bound script, rows or manifest in place.
16542
+
16515
16543
  1. Check the intact export offline: python3 train-lora.py --export /absolute/export --check-export-only.
16516
16544
  2. For reviewed real data only, select a base model, immutable full revision, license and an existing canonical local weight directory. Pin every file by SHA-256 in a separate copy of training-config.example.json. Select architecture-specific LoRA target modules. Record an explicit eligibility review bound to the exact export-manifest.json digest. Storage consent alone is insufficient.
16517
16545
  3. Use an isolated Python 3.12 environment with the exact requirements-training.txt versions. Resolve dependencies separately, retain the complete resolved environment/wheel hashes, and assess the host memory requirements. The template pins direct versions; it is not a platform-complete dependency lock or proof of hardware support.
@@ -16587,6 +16615,11 @@ function consentFor(consents, sessionId) {
16587
16615
  check(c.decision === "granted", "session storage declined; advice remains allowed");
16588
16616
  return c;
16589
16617
  }
16618
+ var INPUT_NORMALIZATION_ALGORITHM = "ecmascript-nfkc-whitespace-trim-lower-sha256-v1";
16619
+ function inputIdentity(input) {
16620
+ check(!/[\uD800-\uDFFF]/u.test(input), "learning input must contain valid Unicode scalar values");
16621
+ return { algorithm: INPUT_NORMALIZATION_ALGORITHM, inputSha256: learningDigest(input), normalizedSha256: learningDigest(input.normalize("NFKC").replace(/\s+/gu, " ").trim().toLowerCase()) };
16622
+ }
16590
16623
  var ENTRY_KEYS = ["caseId", "caseHash", "taskGroup", "lineageGroup", "split", "sessionId", "fixtureOnly", "decisionTimeReviewed", "redactionReviewed", "rights", "exportApproved", "trainingApproved", "reviewer"];
16591
16624
  function parseEntry(c, e) {
16592
16625
  keys(e, ENTRY_KEYS, "experiment entry");
@@ -16618,7 +16651,7 @@ function parseExperiment(cases, manifest) {
16618
16651
  bind(groups, "task:" + e.taskGroup, e.split, "task group");
16619
16652
  bind(groups, "lineage:" + e.lineageGroup, e.split, "lineage group");
16620
16653
  if (e.sessionId !== null) bind(groups, "session:" + e.sessionId, e.split, "session");
16621
- bind(duplicates, learningDigest(cases[i].input.normalize("NFKC").replace(/\s+/gu, " ").trim().toLowerCase()), e.split, "duplicate input");
16654
+ bind(duplicates, inputIdentity(cases[i].input).normalizedSha256, e.split, "duplicate input");
16622
16655
  bind(duplicates, "source:" + cases[i].source.sha256 + ":" + cases[i].source.recordId, e.split, "source record");
16623
16656
  });
16624
16657
  return { ...structuredClone(manifest), entries };
@@ -16788,7 +16821,7 @@ async function prepareExport({ cases, labels, manifest, consents, labelEvidence,
16788
16821
  excluded.push({ caseId: c.id, reasons });
16789
16822
  continue;
16790
16823
  }
16791
- const row = { caseId: c.id, caseHash: c.hash, taskGroup: e.taskGroup, lineageGroup: e.lineageGroup, sessionId: e.sessionId, fixtureOnly: e.fixtureOnly, input: c.input, question: c.question, answer: label.label.value, label: { ...label.label }, source: { ...c.source } };
16824
+ const row = { caseId: c.id, caseHash: c.hash, taskGroup: e.taskGroup, lineageGroup: e.lineageGroup, sessionId: e.sessionId, fixtureOnly: e.fixtureOnly, input: c.input, inputIdentity: inputIdentity(c.input), question: c.question, answer: label.label.value, label: { ...label.label }, source: { ...c.source } };
16792
16825
  splitRows[e.split].push(row);
16793
16826
  included.push(e);
16794
16827
  }
@@ -16796,16 +16829,16 @@ async function prepareExport({ cases, labels, manifest, consents, labelEvidence,
16796
16829
  const trainingEligible = mode === "reviewed-data" && excluded.length === 0 && Object.values(splitRows).every((rows) => rows.length > 0) && included.every((e) => !e.fixtureOnly && e.trainingApproved && e.rights === "local-training");
16797
16830
  const files = { ...trainingAssets() };
16798
16831
  for (const [split, rows] of Object.entries(splitRows)) files[`${split}.jsonl`] = rows.map((r) => JSON.stringify(r) + "\n").join("");
16799
- const metadata2 = { schema: 1, kind: "decision-learning-export", mode, experimentId: manifest.id, experimentHash: learningDigest(manifest), trainingEligible, trainingExecuted: false, providerPredictionsIncluded: false, counts: Object.fromEntries(Object.entries(splitRows).map(([s, r]) => [s, r.length])), excluded, review: { frozenAt: manifest.frozenAt, eligibilityRecords: included, independentLabelReceipts: reviewed.reviewed.map((r) => ({ ...r.reference, recordedAt: r.recordedAt })), sessionConsents: [...new Set(included.filter((e) => e.sessionId !== null).map((e) => e.sessionId))].map((sessionId) => {
16832
+ const metadata2 = { schema: 2, kind: "decision-learning-export", inputNormalization: { algorithm: INPUT_NORMALIZATION_ALGORITHM, runtime: "node", nodeVersion: process.versions.node, unicodeVersion: process.versions.unicode }, mode, experimentId: manifest.id, experimentHash: learningDigest(manifest), trainingEligible, trainingExecuted: false, providerPredictionsIncluded: false, counts: Object.fromEntries(Object.entries(splitRows).map(([s, r]) => [s, r.length])), excluded, review: { frozenAt: manifest.frozenAt, eligibilityRecords: included, independentLabelReceipts: reviewed.reviewed.map((r) => ({ ...r.reference, recordedAt: r.recordedAt })), sessionConsents: [...new Set(included.filter((e) => e.sessionId !== null).map((e) => e.sessionId))].map((sessionId) => {
16800
16833
  const c = consentFor(consents, sessionId);
16801
16834
  return { sessionId, sha256: learningDigest(c), decision: c.decision, interactionId: c.interactionId, recordedAt: c.recordedAt };
16802
16835
  }) }, files: Object.fromEntries(Object.entries(files).map(([name, s]) => [name, { sha256: learningDigest(s), bytes: Buffer.byteLength(s) }])) };
16803
16836
  files["export-manifest.json"] = JSON.stringify(metadata2, null, 2) + "\n";
16804
- return { schema: 1, kind: "prepared-decision-learning-export", metadata: metadata2, files };
16837
+ return { schema: 2, kind: "prepared-decision-learning-export", metadata: metadata2, files };
16805
16838
  }
16806
16839
  async function writePreparedExport({ prepared, directory }) {
16807
16840
  keys(prepared, ["schema", "kind", "metadata", "files"], "prepared export");
16808
- check(prepared.schema === 1 && prepared.kind === "prepared-decision-learning-export", "unsupported prepared export");
16841
+ check(prepared.schema === 2 && prepared.metadata?.schema === 2 && prepared.kind === "prepared-decision-learning-export", "unsupported prepared export");
16809
16842
  text(directory, "output directory", 4096);
16810
16843
  check(isAbsolute7(directory), "output directory must be absolute");
16811
16844
  const allowed = ["train.jsonl", "validation.jsonl", "test.jsonl", "export-manifest.json", "train-lora.py", "training-config.example.json", "requirements-training.txt", "TRAINING.md"];
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "skill-harness",
3
- "version": "0.25.0",
4
- "description": "Test/optimize loop for agent skills — run spec'd scenarios on pi, LLM-judge, score, review, re-run",
3
+ "version": "0.25.1",
4
+ "description": "Test/optimize loop for agent skills \u2014 run spec'd scenarios on pi, LLM-judge, score, review, re-run",
5
5
  "type": "module",
6
6
  "keywords": [
7
7
  "agent-skills",
@@ -31,7 +31,7 @@
31
31
  "README.md"
32
32
  ],
33
33
  "dependencies": {
34
- "@skill-harness/cli": "0.25.0"
34
+ "@skill-harness/cli": "0.25.1"
35
35
  },
36
36
  "peerDependencies": {
37
37
  "typebox": "*"