skill-harness 0.25.0 → 0.25.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.js +50 -17
- package/package.json +3 -3
package/dist/index.js
CHANGED
|
@@ -6293,7 +6293,15 @@ function withExecutionFailure(transcript, failure2) {
|
|
|
6293
6293
|
${transcript}` : transcript;
|
|
6294
6294
|
}
|
|
6295
6295
|
function executionFailureFromTranscript(transcript) {
|
|
6296
|
-
|
|
6296
|
+
const failure2 = failureFromPreamble(transcript, EXECUTION_FAILURE_MARKER);
|
|
6297
|
+
if (failure2 !== null)
|
|
6298
|
+
return failure2;
|
|
6299
|
+
const legacyPrefix = "[adapter failure] ";
|
|
6300
|
+
if (transcript.startsWith(legacyPrefix)) {
|
|
6301
|
+
const detail = transcript.slice(legacyPrefix.length).split("\n", 1)[0].trim();
|
|
6302
|
+
return `adapter failure \u2014 ${detail || "adapter threw without a message"}`;
|
|
6303
|
+
}
|
|
6304
|
+
return null;
|
|
6297
6305
|
}
|
|
6298
6306
|
function withProviderFailure(transcript, failure2) {
|
|
6299
6307
|
return failure2 ? `${PROVIDER_FAILURE_MARKER} ${failure2}
|
|
@@ -7229,13 +7237,13 @@ async function runRep(scenario, rep, repCount, ctx) {
|
|
|
7229
7237
|
}
|
|
7230
7238
|
}
|
|
7231
7239
|
} catch (e) {
|
|
7232
|
-
adapterFailure = e instanceof Error ? e.message : String(e);
|
|
7240
|
+
adapterFailure = (e instanceof Error ? e.message : String(e)) || "adapter threw without a message";
|
|
7233
7241
|
transcript = `[adapter failure] ${adapterFailure}`;
|
|
7234
7242
|
gatePrefix = null;
|
|
7235
7243
|
stagedDiff = null;
|
|
7236
7244
|
traces = [];
|
|
7237
7245
|
}
|
|
7238
|
-
if (!infrastructureFailure) {
|
|
7246
|
+
if (!infrastructureFailure && adapterFailure === null) {
|
|
7239
7247
|
const provider = providerFailureFromTranscript(transcript);
|
|
7240
7248
|
if (provider)
|
|
7241
7249
|
infrastructureFailure = `provider failure \u2014 ${provider}`;
|
|
@@ -7245,14 +7253,15 @@ async function runRep(scenario, rep, repCount, ctx) {
|
|
|
7245
7253
|
infrastructureFailure = `execution failure \u2014 ${execution}`;
|
|
7246
7254
|
}
|
|
7247
7255
|
}
|
|
7248
|
-
executionUnavailable ||= executionFailureFromTranscript(transcript) !== null;
|
|
7256
|
+
executionUnavailable ||= adapterFailure === null && executionFailureFromTranscript(transcript) !== null;
|
|
7249
7257
|
const deliveredText = traces.length > 0 && traces.every((trace) => trace.final_status === "complete" && trace.final_text.length > 0 && !trace.capture_errors?.length);
|
|
7250
7258
|
noResponse = !deliveredText && hasEmptyAssistantTurn(transcript);
|
|
7251
7259
|
if (executionUnavailable || !noResponse && !adapterFailure)
|
|
7252
7260
|
break;
|
|
7253
7261
|
}
|
|
7254
|
-
if (adapterFailure
|
|
7255
|
-
infrastructureFailure
|
|
7262
|
+
if (adapterFailure) {
|
|
7263
|
+
infrastructureFailure ??= `adapter failure \u2014 ${adapterFailure}`;
|
|
7264
|
+
transcript = withExecutionFailure(transcript, `adapter failure \u2014 ${adapterFailure}`);
|
|
7256
7265
|
}
|
|
7257
7266
|
}
|
|
7258
7267
|
const repSuffix = repCount > 1 ? rep : void 0;
|
|
@@ -16381,10 +16390,20 @@ def ordinary(path):
|
|
|
16381
16390
|
def closed(value, fields, name):
|
|
16382
16391
|
require(isinstance(value, dict) and set(value) == set(fields), name + ' has unsupported or missing fields')
|
|
16383
16392
|
|
|
16393
|
+
def jsonl_rows(path):
|
|
16394
|
+
# JSONL records end only at LF. Unicode NEL/LS/PS are legal JSON string data.
|
|
16395
|
+
lines = ordinary(path).decode('utf-8').split('\n')
|
|
16396
|
+
if lines[-1] == '': lines.pop()
|
|
16397
|
+
return [json.loads(line) for line in lines]
|
|
16398
|
+
|
|
16384
16399
|
def export_data(root):
|
|
16385
16400
|
root = pathlib.Path(root).absolute()
|
|
16386
16401
|
raw = ordinary(root / 'export-manifest.json'); m = json.loads(raw)
|
|
16387
|
-
require(m.get('schema') ==
|
|
16402
|
+
require(m.get('schema') == 2 and m.get('kind') == 'decision-learning-export', 'Unsupported export schema; re-export with the current skill-harness')
|
|
16403
|
+
normalization = m['inputNormalization']
|
|
16404
|
+
closed(normalization, ['algorithm','runtime','nodeVersion','unicodeVersion'], 'Input normalization producer')
|
|
16405
|
+
require(normalization['algorithm'] == 'ecmascript-nfkc-whitespace-trim-lower-sha256-v1' and normalization['runtime'] == 'node', 'Unsupported input normalization producer')
|
|
16406
|
+
require(all(isinstance(normalization[key], str) and re.fullmatch(r'[0-9]+(?:\.[0-9]+){1,2}', normalization[key]) for key in ['nodeVersion','unicodeVersion']), 'Invalid input normalization producer version')
|
|
16388
16407
|
require(m.get('trainingExecuted') is False and m.get('providerPredictionsIncluded') is False, 'Unsupported source provenance')
|
|
16389
16408
|
require(set(m['files']) == {'train.jsonl','validation.jsonl','test.jsonl','train-lora.py','requirements-training.txt','training-config.example.json','TRAINING.md'}, 'Unexpected export files')
|
|
16390
16409
|
for name, ref in m['files'].items():
|
|
@@ -16394,10 +16413,10 @@ def export_data(root):
|
|
|
16394
16413
|
require(digest(ordinary(pathlib.Path(__file__).absolute())) == m['files']['train-lora.py']['sha256'], 'Run the exact exported training script')
|
|
16395
16414
|
groups, seen_inputs, seen_cases = {}, {}, set()
|
|
16396
16415
|
for split in ['train','validation','test']:
|
|
16397
|
-
rows =
|
|
16416
|
+
rows = jsonl_rows(root / (split+'.jsonl'))
|
|
16398
16417
|
require(len(rows) == m['counts'][split], 'Split count mismatch')
|
|
16399
16418
|
for row in rows:
|
|
16400
|
-
closed(row, ['caseId','caseHash','taskGroup','lineageGroup','sessionId','fixtureOnly','input','question','answer','label','source'], 'Dataset row')
|
|
16419
|
+
closed(row, ['caseId','caseHash','taskGroup','lineageGroup','sessionId','fixtureOnly','input','inputIdentity','question','answer','label','source'], 'Dataset row')
|
|
16401
16420
|
require(row['caseId'] not in seen_cases and type(row['answer']) is bool, 'Duplicate case or invalid label')
|
|
16402
16421
|
seen_cases.add(row['caseId'])
|
|
16403
16422
|
require(row['label']['caseId'] == row['caseId'] and row['label']['caseHash'] == row['caseHash'] and row['label']['value'] == row['answer'] and row['label']['kind'] in ['human','test'] and row['label']['independent'] is True, 'Label identity mismatch')
|
|
@@ -16406,7 +16425,15 @@ def export_data(root):
|
|
|
16406
16425
|
identity = key + ':' + row[key]
|
|
16407
16426
|
require(identity not in groups or groups[identity] == split, 'Related cases cross splits')
|
|
16408
16427
|
groups[identity] = split
|
|
16409
|
-
|
|
16428
|
+
identity = row['inputIdentity']
|
|
16429
|
+
closed(identity, ['algorithm','inputSha256','normalizedSha256'], 'Input identity receipt')
|
|
16430
|
+
require(identity['algorithm'] == normalization['algorithm'], 'Input identity algorithm mismatch')
|
|
16431
|
+
require(all(isinstance(identity[key], str) and re.fullmatch(r'[0-9a-f]{64}', identity[key]) for key in ['inputSha256','normalizedSha256']), 'Invalid input identity digest')
|
|
16432
|
+
require(isinstance(row['input'], str) and digest(row['input'].encode('utf-8')) == identity['inputSha256'], 'Input identity byte mismatch')
|
|
16433
|
+
# This is producer evidence bound by the exact row/file/manifest hashes.
|
|
16434
|
+
# Recomputing with Python Unicode tables can disagree with the Node producer.
|
|
16435
|
+
# The receipt is not an independent proof that normalization was correct.
|
|
16436
|
+
normalized = identity['normalizedSha256']
|
|
16410
16437
|
require(normalized not in seen_inputs or seen_inputs[normalized] == split, 'Duplicate input crosses splits')
|
|
16411
16438
|
seen_inputs[normalized] = split
|
|
16412
16439
|
if m.get('trainingEligible') is True: require(row['fixtureOnly'] is False and row['sessionId'] is not None, 'Fixture/anonymous data cannot be training eligible')
|
|
@@ -16481,8 +16508,7 @@ def main():
|
|
|
16481
16508
|
model = get_peft_model(model, LoraConfig(task_type=TaskType.CAUSAL_LM, r=t['loraRank'], lora_alpha=t['loraAlpha'], target_modules=t['targetModules'], lora_dropout=0.0, bias='none'))
|
|
16482
16509
|
def rows(split):
|
|
16483
16510
|
data = []
|
|
16484
|
-
for
|
|
16485
|
-
row = json.loads(line)
|
|
16511
|
+
for row in jsonl_rows(root / (split + '.jsonl')):
|
|
16486
16512
|
require(row['fixtureOnly'] is False and type(row['answer']) is bool and row['label']['kind'] in ['human','test'] and row['label']['independent'] is True, 'Only independently labeled reviewed real data may train')
|
|
16487
16513
|
prompt = 'Question: ' + row['question'] + '\nEvidence:\n' + row['input'] + '\nAnswer (true or false): '
|
|
16488
16514
|
prefix = tokenizer.encode(prompt, add_special_tokens=True)
|
|
@@ -16512,6 +16538,8 @@ var TRAINING_DOC = `# Local LoRA workflow
|
|
|
16512
16538
|
|
|
16513
16539
|
This export contains independent labels, never JEV predictions. A fixture demonstration is not training eligible. No training or installation has been performed by exporting these files.
|
|
16514
16540
|
|
|
16541
|
+
Export schema 2 records the producer's normalized-input SHA-256 receipt in every row, the exact input UTF-8 SHA-256, and the Node/Unicode versions and normalization algorithm in the manifest. One JavaScript normalizer validates the split and writes the receipt. Python checks receipt shape, raw-input binding, duplicate identities and exact file hashes; it does not reinterpret those identities with its own Unicode tables. These are trusted producer receipts, not independent proofs of correct normalization. Separate training approval still pins the exact export manifest. Invalid Unicode scalar input is refused by the producer. Re-export older schema-1 packets with the current CLI; do not modify their bound script, rows or manifest in place.
|
|
16542
|
+
|
|
16515
16543
|
1. Check the intact export offline: python3 train-lora.py --export /absolute/export --check-export-only.
|
|
16516
16544
|
2. For reviewed real data only, select a base model, immutable full revision, license and an existing canonical local weight directory. Pin every file by SHA-256 in a separate copy of training-config.example.json. Select architecture-specific LoRA target modules. Record an explicit eligibility review bound to the exact export-manifest.json digest. Storage consent alone is insufficient.
|
|
16517
16545
|
3. Use an isolated Python 3.12 environment with the exact requirements-training.txt versions. Resolve dependencies separately, retain the complete resolved environment/wheel hashes, and assess the host memory requirements. The template pins direct versions; it is not a platform-complete dependency lock or proof of hardware support.
|
|
@@ -16587,6 +16615,11 @@ function consentFor(consents, sessionId) {
|
|
|
16587
16615
|
check(c.decision === "granted", "session storage declined; advice remains allowed");
|
|
16588
16616
|
return c;
|
|
16589
16617
|
}
|
|
16618
|
+
var INPUT_NORMALIZATION_ALGORITHM = "ecmascript-nfkc-whitespace-trim-lower-sha256-v1";
|
|
16619
|
+
function inputIdentity(input) {
|
|
16620
|
+
check(!/[\uD800-\uDFFF]/u.test(input), "learning input must contain valid Unicode scalar values");
|
|
16621
|
+
return { algorithm: INPUT_NORMALIZATION_ALGORITHM, inputSha256: learningDigest(input), normalizedSha256: learningDigest(input.normalize("NFKC").replace(/\s+/gu, " ").trim().toLowerCase()) };
|
|
16622
|
+
}
|
|
16590
16623
|
var ENTRY_KEYS = ["caseId", "caseHash", "taskGroup", "lineageGroup", "split", "sessionId", "fixtureOnly", "decisionTimeReviewed", "redactionReviewed", "rights", "exportApproved", "trainingApproved", "reviewer"];
|
|
16591
16624
|
function parseEntry(c, e) {
|
|
16592
16625
|
keys(e, ENTRY_KEYS, "experiment entry");
|
|
@@ -16618,7 +16651,7 @@ function parseExperiment(cases, manifest) {
|
|
|
16618
16651
|
bind(groups, "task:" + e.taskGroup, e.split, "task group");
|
|
16619
16652
|
bind(groups, "lineage:" + e.lineageGroup, e.split, "lineage group");
|
|
16620
16653
|
if (e.sessionId !== null) bind(groups, "session:" + e.sessionId, e.split, "session");
|
|
16621
|
-
bind(duplicates,
|
|
16654
|
+
bind(duplicates, inputIdentity(cases[i].input).normalizedSha256, e.split, "duplicate input");
|
|
16622
16655
|
bind(duplicates, "source:" + cases[i].source.sha256 + ":" + cases[i].source.recordId, e.split, "source record");
|
|
16623
16656
|
});
|
|
16624
16657
|
return { ...structuredClone(manifest), entries };
|
|
@@ -16788,7 +16821,7 @@ async function prepareExport({ cases, labels, manifest, consents, labelEvidence,
|
|
|
16788
16821
|
excluded.push({ caseId: c.id, reasons });
|
|
16789
16822
|
continue;
|
|
16790
16823
|
}
|
|
16791
|
-
const row = { caseId: c.id, caseHash: c.hash, taskGroup: e.taskGroup, lineageGroup: e.lineageGroup, sessionId: e.sessionId, fixtureOnly: e.fixtureOnly, input: c.input, question: c.question, answer: label.label.value, label: { ...label.label }, source: { ...c.source } };
|
|
16824
|
+
const row = { caseId: c.id, caseHash: c.hash, taskGroup: e.taskGroup, lineageGroup: e.lineageGroup, sessionId: e.sessionId, fixtureOnly: e.fixtureOnly, input: c.input, inputIdentity: inputIdentity(c.input), question: c.question, answer: label.label.value, label: { ...label.label }, source: { ...c.source } };
|
|
16792
16825
|
splitRows[e.split].push(row);
|
|
16793
16826
|
included.push(e);
|
|
16794
16827
|
}
|
|
@@ -16796,16 +16829,16 @@ async function prepareExport({ cases, labels, manifest, consents, labelEvidence,
|
|
|
16796
16829
|
const trainingEligible = mode === "reviewed-data" && excluded.length === 0 && Object.values(splitRows).every((rows) => rows.length > 0) && included.every((e) => !e.fixtureOnly && e.trainingApproved && e.rights === "local-training");
|
|
16797
16830
|
const files = { ...trainingAssets() };
|
|
16798
16831
|
for (const [split, rows] of Object.entries(splitRows)) files[`${split}.jsonl`] = rows.map((r) => JSON.stringify(r) + "\n").join("");
|
|
16799
|
-
const metadata2 = { schema:
|
|
16832
|
+
const metadata2 = { schema: 2, kind: "decision-learning-export", inputNormalization: { algorithm: INPUT_NORMALIZATION_ALGORITHM, runtime: "node", nodeVersion: process.versions.node, unicodeVersion: process.versions.unicode }, mode, experimentId: manifest.id, experimentHash: learningDigest(manifest), trainingEligible, trainingExecuted: false, providerPredictionsIncluded: false, counts: Object.fromEntries(Object.entries(splitRows).map(([s, r]) => [s, r.length])), excluded, review: { frozenAt: manifest.frozenAt, eligibilityRecords: included, independentLabelReceipts: reviewed.reviewed.map((r) => ({ ...r.reference, recordedAt: r.recordedAt })), sessionConsents: [...new Set(included.filter((e) => e.sessionId !== null).map((e) => e.sessionId))].map((sessionId) => {
|
|
16800
16833
|
const c = consentFor(consents, sessionId);
|
|
16801
16834
|
return { sessionId, sha256: learningDigest(c), decision: c.decision, interactionId: c.interactionId, recordedAt: c.recordedAt };
|
|
16802
16835
|
}) }, files: Object.fromEntries(Object.entries(files).map(([name, s]) => [name, { sha256: learningDigest(s), bytes: Buffer.byteLength(s) }])) };
|
|
16803
16836
|
files["export-manifest.json"] = JSON.stringify(metadata2, null, 2) + "\n";
|
|
16804
|
-
return { schema:
|
|
16837
|
+
return { schema: 2, kind: "prepared-decision-learning-export", metadata: metadata2, files };
|
|
16805
16838
|
}
|
|
16806
16839
|
async function writePreparedExport({ prepared, directory }) {
|
|
16807
16840
|
keys(prepared, ["schema", "kind", "metadata", "files"], "prepared export");
|
|
16808
|
-
check(prepared.schema ===
|
|
16841
|
+
check(prepared.schema === 2 && prepared.metadata?.schema === 2 && prepared.kind === "prepared-decision-learning-export", "unsupported prepared export");
|
|
16809
16842
|
text(directory, "output directory", 4096);
|
|
16810
16843
|
check(isAbsolute7(directory), "output directory must be absolute");
|
|
16811
16844
|
const allowed = ["train.jsonl", "validation.jsonl", "test.jsonl", "export-manifest.json", "train-lora.py", "training-config.example.json", "requirements-training.txt", "TRAINING.md"];
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "skill-harness",
|
|
3
|
-
"version": "0.25.
|
|
4
|
-
"description": "Test/optimize loop for agent skills
|
|
3
|
+
"version": "0.25.1",
|
|
4
|
+
"description": "Test/optimize loop for agent skills \u2014 run spec'd scenarios on pi, LLM-judge, score, review, re-run",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"keywords": [
|
|
7
7
|
"agent-skills",
|
|
@@ -31,7 +31,7 @@
|
|
|
31
31
|
"README.md"
|
|
32
32
|
],
|
|
33
33
|
"dependencies": {
|
|
34
|
-
"@skill-harness/cli": "0.25.
|
|
34
|
+
"@skill-harness/cli": "0.25.1"
|
|
35
35
|
},
|
|
36
36
|
"peerDependencies": {
|
|
37
37
|
"typebox": "*"
|