skill-harness 0.25.0 → 0.26.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.js +400 -51
- package/package.json +3 -3
package/dist/index.js
CHANGED
|
@@ -6293,7 +6293,15 @@ function withExecutionFailure(transcript, failure2) {
|
|
|
6293
6293
|
${transcript}` : transcript;
|
|
6294
6294
|
}
|
|
6295
6295
|
function executionFailureFromTranscript(transcript) {
|
|
6296
|
-
|
|
6296
|
+
const failure2 = failureFromPreamble(transcript, EXECUTION_FAILURE_MARKER);
|
|
6297
|
+
if (failure2 !== null)
|
|
6298
|
+
return failure2;
|
|
6299
|
+
const legacyPrefix = "[adapter failure] ";
|
|
6300
|
+
if (transcript.startsWith(legacyPrefix)) {
|
|
6301
|
+
const detail = transcript.slice(legacyPrefix.length).split("\n", 1)[0].trim();
|
|
6302
|
+
return `adapter failure \u2014 ${detail || "adapter threw without a message"}`;
|
|
6303
|
+
}
|
|
6304
|
+
return null;
|
|
6297
6305
|
}
|
|
6298
6306
|
function withProviderFailure(transcript, failure2) {
|
|
6299
6307
|
return failure2 ? `${PROVIDER_FAILURE_MARKER} ${failure2}
|
|
@@ -7229,13 +7237,13 @@ async function runRep(scenario, rep, repCount, ctx) {
|
|
|
7229
7237
|
}
|
|
7230
7238
|
}
|
|
7231
7239
|
} catch (e) {
|
|
7232
|
-
adapterFailure = e instanceof Error ? e.message : String(e);
|
|
7240
|
+
adapterFailure = (e instanceof Error ? e.message : String(e)) || "adapter threw without a message";
|
|
7233
7241
|
transcript = `[adapter failure] ${adapterFailure}`;
|
|
7234
7242
|
gatePrefix = null;
|
|
7235
7243
|
stagedDiff = null;
|
|
7236
7244
|
traces = [];
|
|
7237
7245
|
}
|
|
7238
|
-
if (!infrastructureFailure) {
|
|
7246
|
+
if (!infrastructureFailure && adapterFailure === null) {
|
|
7239
7247
|
const provider = providerFailureFromTranscript(transcript);
|
|
7240
7248
|
if (provider)
|
|
7241
7249
|
infrastructureFailure = `provider failure \u2014 ${provider}`;
|
|
@@ -7245,14 +7253,15 @@ async function runRep(scenario, rep, repCount, ctx) {
|
|
|
7245
7253
|
infrastructureFailure = `execution failure \u2014 ${execution}`;
|
|
7246
7254
|
}
|
|
7247
7255
|
}
|
|
7248
|
-
executionUnavailable ||= executionFailureFromTranscript(transcript) !== null;
|
|
7256
|
+
executionUnavailable ||= adapterFailure === null && executionFailureFromTranscript(transcript) !== null;
|
|
7249
7257
|
const deliveredText = traces.length > 0 && traces.every((trace) => trace.final_status === "complete" && trace.final_text.length > 0 && !trace.capture_errors?.length);
|
|
7250
7258
|
noResponse = !deliveredText && hasEmptyAssistantTurn(transcript);
|
|
7251
7259
|
if (executionUnavailable || !noResponse && !adapterFailure)
|
|
7252
7260
|
break;
|
|
7253
7261
|
}
|
|
7254
|
-
if (adapterFailure
|
|
7255
|
-
infrastructureFailure
|
|
7262
|
+
if (adapterFailure) {
|
|
7263
|
+
infrastructureFailure ??= `adapter failure \u2014 ${adapterFailure}`;
|
|
7264
|
+
transcript = withExecutionFailure(transcript, `adapter failure \u2014 ${adapterFailure}`);
|
|
7256
7265
|
}
|
|
7257
7266
|
}
|
|
7258
7267
|
const repSuffix = repCount > 1 ? rep : void 0;
|
|
@@ -10225,8 +10234,8 @@ function normalizePiDaddyRecordLedgerV12(raw) {
|
|
|
10225
10234
|
const record = parsed;
|
|
10226
10235
|
if (record.seq !== index + 1) throw new Error(`${where}: sequence gap`);
|
|
10227
10236
|
if (record.prev !== previous) throw new Error(`${where}: previous line hash mismatch`);
|
|
10228
|
-
const { digest:
|
|
10229
|
-
if (sha4(canonical3(unsigned)) !==
|
|
10237
|
+
const { digest: digest3, ...unsigned } = record;
|
|
10238
|
+
if (sha4(canonical3(unsigned)) !== digest3) throw new Error(`${where}: record digest mismatch`);
|
|
10230
10239
|
if (ids.has(record.id)) throw new Error(`${where}: duplicate record identity`);
|
|
10231
10240
|
ids.add(record.id);
|
|
10232
10241
|
validate2(PI_DADDY_RECORD_V1_GOVERNANCE_SCHEMA2, record.body, `${where} governance body`);
|
|
@@ -12924,7 +12933,7 @@ function normalizeLegacyGrant(record, index) {
|
|
|
12924
12933
|
const effective = record.effective;
|
|
12925
12934
|
const denied = record.denied;
|
|
12926
12935
|
const gated = record.gatedBlocked;
|
|
12927
|
-
const
|
|
12936
|
+
const digest3 = object3(record.definitionDigest);
|
|
12928
12937
|
const common2 = {
|
|
12929
12938
|
event_version: TRAJECTORY_EVENT_VERSION,
|
|
12930
12939
|
source: "pi-daddy-0.17",
|
|
@@ -12943,7 +12952,7 @@ function normalizeLegacyGrant(record, index) {
|
|
|
12943
12952
|
gate_outcome: record.gateOutcome,
|
|
12944
12953
|
human_denied: record.humanDenied === true,
|
|
12945
12954
|
reason: record.reason,
|
|
12946
|
-
definition_name:
|
|
12955
|
+
definition_name: digest3?.name,
|
|
12947
12956
|
legacy_schema: "pi-daddy-grant-ledger/0.17"
|
|
12948
12957
|
});
|
|
12949
12958
|
const refusal = record.blocked ? legacyRefusalCode(record) : void 0;
|
|
@@ -12953,7 +12962,7 @@ function normalizeLegacyGrant(record, index) {
|
|
|
12953
12962
|
requested_capabilities: requested,
|
|
12954
12963
|
effective_capabilities: effective,
|
|
12955
12964
|
refusal_code: refusal,
|
|
12956
|
-
digests: anyDefined({ definition: string(
|
|
12965
|
+
digests: anyDefined({ definition: string(digest3?.sha256) }),
|
|
12957
12966
|
attributes
|
|
12958
12967
|
});
|
|
12959
12968
|
const events = [
|
|
@@ -13561,7 +13570,7 @@ var init_src = __esm({
|
|
|
13561
13570
|
|
|
13562
13571
|
// packages/pi-extension/src/index.ts
|
|
13563
13572
|
import { fileURLToPath as fileURLToPath2 } from "node:url";
|
|
13564
|
-
import { basename as basename4, dirname as dirname10, join as
|
|
13573
|
+
import { basename as basename4, dirname as dirname10, join as join35 } from "node:path";
|
|
13565
13574
|
|
|
13566
13575
|
// packages/pi-extension/src/commands.ts
|
|
13567
13576
|
init_dist();
|
|
@@ -15951,7 +15960,7 @@ async function runViaExtension(opts) {
|
|
|
15951
15960
|
}
|
|
15952
15961
|
|
|
15953
15962
|
// packages/pi-extension/src/commands.ts
|
|
15954
|
-
var USAGE = "usage: /skill-harness run [skill] [--model p:m] [--reps N] [--mode red|green|force] [--canary] [--judge p:m] | judge [run-dir] | review [skill] | coverage [skill] | jev enable|run|status|disable";
|
|
15963
|
+
var USAGE = "usage: /skill-harness run [skill] [--model p:m] [--reps N] [--mode red|green|force] [--canary] [--judge p:m] | judge [run-dir] | review [skill] | coverage [skill] | jev enable [workflow]|run|status|disable";
|
|
15955
15964
|
function parse(argstr) {
|
|
15956
15965
|
const tokens = argstr.trim().length ? argstr.trim().split(/\s+/) : [];
|
|
15957
15966
|
const [sub = "", ...rest] = tokens;
|
|
@@ -16381,10 +16390,20 @@ def ordinary(path):
|
|
|
16381
16390
|
def closed(value, fields, name):
|
|
16382
16391
|
require(isinstance(value, dict) and set(value) == set(fields), name + ' has unsupported or missing fields')
|
|
16383
16392
|
|
|
16393
|
+
def jsonl_rows(path):
|
|
16394
|
+
# JSONL records end only at LF. Unicode NEL/LS/PS are legal JSON string data.
|
|
16395
|
+
lines = ordinary(path).decode('utf-8').split('\n')
|
|
16396
|
+
if lines[-1] == '': lines.pop()
|
|
16397
|
+
return [json.loads(line) for line in lines]
|
|
16398
|
+
|
|
16384
16399
|
def export_data(root):
|
|
16385
16400
|
root = pathlib.Path(root).absolute()
|
|
16386
16401
|
raw = ordinary(root / 'export-manifest.json'); m = json.loads(raw)
|
|
16387
|
-
require(m.get('schema') ==
|
|
16402
|
+
require(m.get('schema') == 2 and m.get('kind') == 'decision-learning-export', 'Unsupported export schema; re-export with the current skill-harness')
|
|
16403
|
+
normalization = m['inputNormalization']
|
|
16404
|
+
closed(normalization, ['algorithm','runtime','nodeVersion','unicodeVersion'], 'Input normalization producer')
|
|
16405
|
+
require(normalization['algorithm'] == 'ecmascript-nfkc-whitespace-trim-lower-sha256-v1' and normalization['runtime'] == 'node', 'Unsupported input normalization producer')
|
|
16406
|
+
require(all(isinstance(normalization[key], str) and re.fullmatch(r'[0-9]+(?:\.[0-9]+){1,2}', normalization[key]) for key in ['nodeVersion','unicodeVersion']), 'Invalid input normalization producer version')
|
|
16388
16407
|
require(m.get('trainingExecuted') is False and m.get('providerPredictionsIncluded') is False, 'Unsupported source provenance')
|
|
16389
16408
|
require(set(m['files']) == {'train.jsonl','validation.jsonl','test.jsonl','train-lora.py','requirements-training.txt','training-config.example.json','TRAINING.md'}, 'Unexpected export files')
|
|
16390
16409
|
for name, ref in m['files'].items():
|
|
@@ -16394,10 +16413,10 @@ def export_data(root):
|
|
|
16394
16413
|
require(digest(ordinary(pathlib.Path(__file__).absolute())) == m['files']['train-lora.py']['sha256'], 'Run the exact exported training script')
|
|
16395
16414
|
groups, seen_inputs, seen_cases = {}, {}, set()
|
|
16396
16415
|
for split in ['train','validation','test']:
|
|
16397
|
-
rows =
|
|
16416
|
+
rows = jsonl_rows(root / (split+'.jsonl'))
|
|
16398
16417
|
require(len(rows) == m['counts'][split], 'Split count mismatch')
|
|
16399
16418
|
for row in rows:
|
|
16400
|
-
closed(row, ['caseId','caseHash','taskGroup','lineageGroup','sessionId','fixtureOnly','input','question','answer','label','source'], 'Dataset row')
|
|
16419
|
+
closed(row, ['caseId','caseHash','taskGroup','lineageGroup','sessionId','fixtureOnly','input','inputIdentity','question','answer','label','source'], 'Dataset row')
|
|
16401
16420
|
require(row['caseId'] not in seen_cases and type(row['answer']) is bool, 'Duplicate case or invalid label')
|
|
16402
16421
|
seen_cases.add(row['caseId'])
|
|
16403
16422
|
require(row['label']['caseId'] == row['caseId'] and row['label']['caseHash'] == row['caseHash'] and row['label']['value'] == row['answer'] and row['label']['kind'] in ['human','test'] and row['label']['independent'] is True, 'Label identity mismatch')
|
|
@@ -16406,7 +16425,15 @@ def export_data(root):
|
|
|
16406
16425
|
identity = key + ':' + row[key]
|
|
16407
16426
|
require(identity not in groups or groups[identity] == split, 'Related cases cross splits')
|
|
16408
16427
|
groups[identity] = split
|
|
16409
|
-
|
|
16428
|
+
identity = row['inputIdentity']
|
|
16429
|
+
closed(identity, ['algorithm','inputSha256','normalizedSha256'], 'Input identity receipt')
|
|
16430
|
+
require(identity['algorithm'] == normalization['algorithm'], 'Input identity algorithm mismatch')
|
|
16431
|
+
require(all(isinstance(identity[key], str) and re.fullmatch(r'[0-9a-f]{64}', identity[key]) for key in ['inputSha256','normalizedSha256']), 'Invalid input identity digest')
|
|
16432
|
+
require(isinstance(row['input'], str) and digest(row['input'].encode('utf-8')) == identity['inputSha256'], 'Input identity byte mismatch')
|
|
16433
|
+
# This is producer evidence bound by the exact row/file/manifest hashes.
|
|
16434
|
+
# Recomputing with Python Unicode tables can disagree with the Node producer.
|
|
16435
|
+
# The receipt is not an independent proof that normalization was correct.
|
|
16436
|
+
normalized = identity['normalizedSha256']
|
|
16410
16437
|
require(normalized not in seen_inputs or seen_inputs[normalized] == split, 'Duplicate input crosses splits')
|
|
16411
16438
|
seen_inputs[normalized] = split
|
|
16412
16439
|
if m.get('trainingEligible') is True: require(row['fixtureOnly'] is False and row['sessionId'] is not None, 'Fixture/anonymous data cannot be training eligible')
|
|
@@ -16481,8 +16508,7 @@ def main():
|
|
|
16481
16508
|
model = get_peft_model(model, LoraConfig(task_type=TaskType.CAUSAL_LM, r=t['loraRank'], lora_alpha=t['loraAlpha'], target_modules=t['targetModules'], lora_dropout=0.0, bias='none'))
|
|
16482
16509
|
def rows(split):
|
|
16483
16510
|
data = []
|
|
16484
|
-
for
|
|
16485
|
-
row = json.loads(line)
|
|
16511
|
+
for row in jsonl_rows(root / (split + '.jsonl')):
|
|
16486
16512
|
require(row['fixtureOnly'] is False and type(row['answer']) is bool and row['label']['kind'] in ['human','test'] and row['label']['independent'] is True, 'Only independently labeled reviewed real data may train')
|
|
16487
16513
|
prompt = 'Question: ' + row['question'] + '\nEvidence:\n' + row['input'] + '\nAnswer (true or false): '
|
|
16488
16514
|
prefix = tokenizer.encode(prompt, add_special_tokens=True)
|
|
@@ -16512,6 +16538,8 @@ var TRAINING_DOC = `# Local LoRA workflow
|
|
|
16512
16538
|
|
|
16513
16539
|
This export contains independent labels, never JEV predictions. A fixture demonstration is not training eligible. No training or installation has been performed by exporting these files.
|
|
16514
16540
|
|
|
16541
|
+
Export schema 2 records the producer's normalized-input SHA-256 receipt in every row, the exact input UTF-8 SHA-256, and the Node/Unicode versions and normalization algorithm in the manifest. One JavaScript normalizer validates the split and writes the receipt. Python checks receipt shape, raw-input binding, duplicate identities and exact file hashes; it does not reinterpret those identities with its own Unicode tables. These are trusted producer receipts, not independent proofs of correct normalization. Separate training approval still pins the exact export manifest. Invalid Unicode scalar input is refused by the producer. Re-export older schema-1 packets with the current CLI; do not modify their bound script, rows or manifest in place.
|
|
16542
|
+
|
|
16515
16543
|
1. Check the intact export offline: python3 train-lora.py --export /absolute/export --check-export-only.
|
|
16516
16544
|
2. For reviewed real data only, select a base model, immutable full revision, license and an existing canonical local weight directory. Pin every file by SHA-256 in a separate copy of training-config.example.json. Select architecture-specific LoRA target modules. Record an explicit eligibility review bound to the exact export-manifest.json digest. Storage consent alone is insufficient.
|
|
16517
16545
|
3. Use an isolated Python 3.12 environment with the exact requirements-training.txt versions. Resolve dependencies separately, retain the complete resolved environment/wheel hashes, and assess the host memory requirements. The template pins direct versions; it is not a platform-complete dependency lock or proof of hardware support.
|
|
@@ -16587,6 +16615,11 @@ function consentFor(consents, sessionId) {
|
|
|
16587
16615
|
check(c.decision === "granted", "session storage declined; advice remains allowed");
|
|
16588
16616
|
return c;
|
|
16589
16617
|
}
|
|
16618
|
+
var INPUT_NORMALIZATION_ALGORITHM = "ecmascript-nfkc-whitespace-trim-lower-sha256-v1";
|
|
16619
|
+
function inputIdentity(input) {
|
|
16620
|
+
check(!/[\uD800-\uDFFF]/u.test(input), "learning input must contain valid Unicode scalar values");
|
|
16621
|
+
return { algorithm: INPUT_NORMALIZATION_ALGORITHM, inputSha256: learningDigest(input), normalizedSha256: learningDigest(input.normalize("NFKC").replace(/\s+/gu, " ").trim().toLowerCase()) };
|
|
16622
|
+
}
|
|
16590
16623
|
var ENTRY_KEYS = ["caseId", "caseHash", "taskGroup", "lineageGroup", "split", "sessionId", "fixtureOnly", "decisionTimeReviewed", "redactionReviewed", "rights", "exportApproved", "trainingApproved", "reviewer"];
|
|
16591
16624
|
function parseEntry(c, e) {
|
|
16592
16625
|
keys(e, ENTRY_KEYS, "experiment entry");
|
|
@@ -16618,7 +16651,7 @@ function parseExperiment(cases, manifest) {
|
|
|
16618
16651
|
bind(groups, "task:" + e.taskGroup, e.split, "task group");
|
|
16619
16652
|
bind(groups, "lineage:" + e.lineageGroup, e.split, "lineage group");
|
|
16620
16653
|
if (e.sessionId !== null) bind(groups, "session:" + e.sessionId, e.split, "session");
|
|
16621
|
-
bind(duplicates,
|
|
16654
|
+
bind(duplicates, inputIdentity(cases[i].input).normalizedSha256, e.split, "duplicate input");
|
|
16622
16655
|
bind(duplicates, "source:" + cases[i].source.sha256 + ":" + cases[i].source.recordId, e.split, "source record");
|
|
16623
16656
|
});
|
|
16624
16657
|
return { ...structuredClone(manifest), entries };
|
|
@@ -16788,7 +16821,7 @@ async function prepareExport({ cases, labels, manifest, consents, labelEvidence,
|
|
|
16788
16821
|
excluded.push({ caseId: c.id, reasons });
|
|
16789
16822
|
continue;
|
|
16790
16823
|
}
|
|
16791
|
-
const row = { caseId: c.id, caseHash: c.hash, taskGroup: e.taskGroup, lineageGroup: e.lineageGroup, sessionId: e.sessionId, fixtureOnly: e.fixtureOnly, input: c.input, question: c.question, answer: label.label.value, label: { ...label.label }, source: { ...c.source } };
|
|
16824
|
+
const row = { caseId: c.id, caseHash: c.hash, taskGroup: e.taskGroup, lineageGroup: e.lineageGroup, sessionId: e.sessionId, fixtureOnly: e.fixtureOnly, input: c.input, inputIdentity: inputIdentity(c.input), question: c.question, answer: label.label.value, label: { ...label.label }, source: { ...c.source } };
|
|
16792
16825
|
splitRows[e.split].push(row);
|
|
16793
16826
|
included.push(e);
|
|
16794
16827
|
}
|
|
@@ -16796,16 +16829,16 @@ async function prepareExport({ cases, labels, manifest, consents, labelEvidence,
|
|
|
16796
16829
|
const trainingEligible = mode === "reviewed-data" && excluded.length === 0 && Object.values(splitRows).every((rows) => rows.length > 0) && included.every((e) => !e.fixtureOnly && e.trainingApproved && e.rights === "local-training");
|
|
16797
16830
|
const files = { ...trainingAssets() };
|
|
16798
16831
|
for (const [split, rows] of Object.entries(splitRows)) files[`${split}.jsonl`] = rows.map((r) => JSON.stringify(r) + "\n").join("");
|
|
16799
|
-
const metadata2 = { schema:
|
|
16832
|
+
const metadata2 = { schema: 2, kind: "decision-learning-export", inputNormalization: { algorithm: INPUT_NORMALIZATION_ALGORITHM, runtime: "node", nodeVersion: process.versions.node, unicodeVersion: process.versions.unicode }, mode, experimentId: manifest.id, experimentHash: learningDigest(manifest), trainingEligible, trainingExecuted: false, providerPredictionsIncluded: false, counts: Object.fromEntries(Object.entries(splitRows).map(([s, r]) => [s, r.length])), excluded, review: { frozenAt: manifest.frozenAt, eligibilityRecords: included, independentLabelReceipts: reviewed.reviewed.map((r) => ({ ...r.reference, recordedAt: r.recordedAt })), sessionConsents: [...new Set(included.filter((e) => e.sessionId !== null).map((e) => e.sessionId))].map((sessionId) => {
|
|
16800
16833
|
const c = consentFor(consents, sessionId);
|
|
16801
16834
|
return { sessionId, sha256: learningDigest(c), decision: c.decision, interactionId: c.interactionId, recordedAt: c.recordedAt };
|
|
16802
16835
|
}) }, files: Object.fromEntries(Object.entries(files).map(([name, s]) => [name, { sha256: learningDigest(s), bytes: Buffer.byteLength(s) }])) };
|
|
16803
16836
|
files["export-manifest.json"] = JSON.stringify(metadata2, null, 2) + "\n";
|
|
16804
|
-
return { schema:
|
|
16837
|
+
return { schema: 2, kind: "prepared-decision-learning-export", metadata: metadata2, files };
|
|
16805
16838
|
}
|
|
16806
16839
|
async function writePreparedExport({ prepared, directory }) {
|
|
16807
16840
|
keys(prepared, ["schema", "kind", "metadata", "files"], "prepared export");
|
|
16808
|
-
check(prepared.schema ===
|
|
16841
|
+
check(prepared.schema === 2 && prepared.metadata?.schema === 2 && prepared.kind === "prepared-decision-learning-export", "unsupported prepared export");
|
|
16809
16842
|
text(directory, "output directory", 4096);
|
|
16810
16843
|
check(isAbsolute7(directory), "output directory must be absolute");
|
|
16811
16844
|
const allowed = ["train.jsonl", "validation.jsonl", "test.jsonl", "export-manifest.json", "train-lora.py", "training-config.example.json", "requirements-training.txt", "TRAINING.md"];
|
|
@@ -17335,7 +17368,7 @@ var RESEARCH_OPTIONAL = {
|
|
|
17335
17368
|
"run-pi": ["pi-node", "auth-path", "timeout-ms"]
|
|
17336
17369
|
};
|
|
17337
17370
|
async function researchCommand(command, flags, helpers, options) {
|
|
17338
|
-
const { cases, jsonFile: jsonFile2, textFile: textFile2, writeNew:
|
|
17371
|
+
const { cases, jsonFile: jsonFile2, textFile: textFile2, writeNew: writeNew3, readLegacyRun } = helpers;
|
|
17339
17372
|
const { emit = console.log, piRunner } = options;
|
|
17340
17373
|
if (command === "fixtures") {
|
|
17341
17374
|
const f = createLearningFixtures({});
|
|
@@ -17354,7 +17387,7 @@ async function researchCommand(command, flags, helpers, options) {
|
|
|
17354
17387
|
"label-evidence.json": refs,
|
|
17355
17388
|
"consents.json": []
|
|
17356
17389
|
}))
|
|
17357
|
-
await
|
|
17390
|
+
await writeNew3(join32(dir, name), value);
|
|
17358
17391
|
emit(
|
|
17359
17392
|
JSON.stringify({
|
|
17360
17393
|
saved: dir,
|
|
@@ -17372,7 +17405,7 @@ async function researchCommand(command, flags, helpers, options) {
|
|
|
17372
17405
|
sessionConsent: await jsonFile2(flags.consent),
|
|
17373
17406
|
experimentEntry: await jsonFile2(flags.entry)
|
|
17374
17407
|
});
|
|
17375
|
-
await
|
|
17408
|
+
await writeNew3(flags.out, result);
|
|
17376
17409
|
emit(JSON.stringify({ saved: resolve15(flags.out), trainingReady: false }));
|
|
17377
17410
|
return;
|
|
17378
17411
|
}
|
|
@@ -17469,7 +17502,7 @@ async function researchCommand(command, flags, helpers, options) {
|
|
|
17469
17502
|
}
|
|
17470
17503
|
const report = compareRuns(cases, labels, manifest, runs);
|
|
17471
17504
|
report.labelEvidenceReview = evidence;
|
|
17472
|
-
await
|
|
17505
|
+
await writeNew3(flags.out, report);
|
|
17473
17506
|
emit(
|
|
17474
17507
|
JSON.stringify({
|
|
17475
17508
|
saved: resolve15(flags.out),
|
|
@@ -17657,8 +17690,9 @@ function normalize3(payload) {
|
|
|
17657
17690
|
}
|
|
17658
17691
|
function deadline(signal) {
|
|
17659
17692
|
return new Promise((_, reject) => {
|
|
17660
|
-
|
|
17661
|
-
signal.
|
|
17693
|
+
const fail = () => reject(signal.reason instanceof Error ? signal.reason : new Error("provider request timed out"));
|
|
17694
|
+
if (signal.aborted) return fail();
|
|
17695
|
+
signal.addEventListener("abort", fail, { once: true });
|
|
17662
17696
|
});
|
|
17663
17697
|
}
|
|
17664
17698
|
async function readBounded(response, timedOut) {
|
|
@@ -17694,7 +17728,7 @@ async function readBounded(response, timedOut) {
|
|
|
17694
17728
|
return new TextDecoder("utf-8", { fatal: true }).decode(bytes);
|
|
17695
17729
|
}
|
|
17696
17730
|
function failure(startedAt, error) {
|
|
17697
|
-
const publicErrors = ["invalid provider response", "provider request timed out", "provider response exceeded limit"];
|
|
17731
|
+
const publicErrors = ["invalid provider response", "provider request timed out", "provider response exceeded limit", "provider request cancelled"];
|
|
17698
17732
|
const message = error instanceof Error && publicErrors.includes(error.message) ? error.message : "provider request failed";
|
|
17699
17733
|
return {
|
|
17700
17734
|
status: "error",
|
|
@@ -17709,6 +17743,7 @@ async function callProvider(provider, model, example, options = {}) {
|
|
|
17709
17743
|
const startedAt = performance.now();
|
|
17710
17744
|
let controller;
|
|
17711
17745
|
let timer;
|
|
17746
|
+
let cancel;
|
|
17712
17747
|
try {
|
|
17713
17748
|
const request = makeRequest(provider, model, example);
|
|
17714
17749
|
requiredString(options.apiKey, "apiKey", 16 * 1024);
|
|
@@ -17719,8 +17754,14 @@ async function callProvider(provider, model, example, options = {}) {
|
|
|
17719
17754
|
if (!Number.isSafeInteger(timeoutMs) || timeoutMs < 1 || timeoutMs > 3e5) {
|
|
17720
17755
|
throw new TypeError("invalid timeout");
|
|
17721
17756
|
}
|
|
17757
|
+
if (options.signal !== void 0 && !(options.signal instanceof AbortSignal)) {
|
|
17758
|
+
throw new TypeError("signal must be an AbortSignal");
|
|
17759
|
+
}
|
|
17760
|
+
if (options.signal?.aborted) throw new Error("provider request cancelled");
|
|
17722
17761
|
controller = new AbortController();
|
|
17723
|
-
|
|
17762
|
+
cancel = () => controller.abort(new Error("provider request cancelled"));
|
|
17763
|
+
options.signal?.addEventListener("abort", cancel, { once: true });
|
|
17764
|
+
timer = setTimeout(() => controller.abort(new Error("provider request timed out")), timeoutMs);
|
|
17724
17765
|
const timedOut = deadline(controller.signal);
|
|
17725
17766
|
const response = await Promise.race([
|
|
17726
17767
|
(options.fetchImpl ?? fetch)(request.url, {
|
|
@@ -17745,6 +17786,7 @@ async function callProvider(provider, model, example, options = {}) {
|
|
|
17745
17786
|
return failure(startedAt, error);
|
|
17746
17787
|
} finally {
|
|
17747
17788
|
if (timer !== void 0) clearTimeout(timer);
|
|
17789
|
+
if (cancel) options.signal?.removeEventListener("abort", cancel);
|
|
17748
17790
|
controller?.abort();
|
|
17749
17791
|
}
|
|
17750
17792
|
}
|
|
@@ -18597,33 +18639,264 @@ async function main(argv, {
|
|
|
18597
18639
|
);
|
|
18598
18640
|
}
|
|
18599
18641
|
|
|
18642
|
+
// packages/pi-extension/src/jev-packet.ts
|
|
18643
|
+
import { createHash as createHash19 } from "node:crypto";
|
|
18644
|
+
import { closeSync as closeSync2, fsyncSync as fsyncSync2, lstatSync, mkdirSync as mkdirSync5, mkdtempSync as mkdtempSync4, openSync as openSync2, writeFileSync as writeFileSync8 } from "node:fs";
|
|
18645
|
+
import { homedir as homedir6 } from "node:os";
|
|
18646
|
+
import { join as join34 } from "node:path";
|
|
18647
|
+
var JEV_PROVIDER = "jev";
|
|
18648
|
+
var JEV_MODEL = "typesafe/jev-1.13";
|
|
18649
|
+
var JEV_WORKFLOW_LIMIT = 3;
|
|
18650
|
+
var JEV_QUESTION = "Does this handoff account for every explicitly required acceptance check with successful evidence tied to the reported candidate?";
|
|
18651
|
+
var JEV_SOURCE_KIND = "skill-harness-selected-handoff-v1";
|
|
18652
|
+
var digest2 = (text3) => createHash19("sha256").update(text3, "utf8").digest("hex");
|
|
18653
|
+
function requireIdentity(value, name) {
|
|
18654
|
+
if (typeof value !== "string" || !value.trim() || value.length > 256)
|
|
18655
|
+
throw new Error(`${name} must be a nonempty runtime identity of at most 256 characters`);
|
|
18656
|
+
}
|
|
18657
|
+
function prepareHandoff(value) {
|
|
18658
|
+
for (const field of ["candidate", "requirements", "evidence"]) {
|
|
18659
|
+
if (typeof value[field] !== "string" || !value[field].trim() || value[field].length > 32e3)
|
|
18660
|
+
throw new Error(`${field} must be a nonempty bounded string`);
|
|
18661
|
+
if (/[\uD800-\uDBFF](?![\uDC00-\uDFFF])|(?<![\uD800-\uDBFF])[\uDC00-\uDFFF]/u.test(value[field]))
|
|
18662
|
+
throw new Error(`${field} must contain valid Unicode`);
|
|
18663
|
+
}
|
|
18664
|
+
const input = JSON.stringify({
|
|
18665
|
+
candidate: value.candidate,
|
|
18666
|
+
requirements: value.requirements,
|
|
18667
|
+
evidence: value.evidence
|
|
18668
|
+
});
|
|
18669
|
+
if (Array.from(input).length > 16e3)
|
|
18670
|
+
throw new Error("The complete selected handoff packet must not exceed 16000 characters");
|
|
18671
|
+
return Object.freeze({ input, question: JEV_QUESTION, inputSha256: digest2(input) });
|
|
18672
|
+
}
|
|
18673
|
+
function createHandoffSource(options) {
|
|
18674
|
+
requireIdentity(options.sessionId, "Pi sessionId");
|
|
18675
|
+
requireIdentity(options.toolCallId, "Pi toolCallId");
|
|
18676
|
+
const source = {
|
|
18677
|
+
schema: 1,
|
|
18678
|
+
kind: JEV_SOURCE_KIND,
|
|
18679
|
+
sessionId: options.sessionId,
|
|
18680
|
+
toolCallId: options.toolCallId,
|
|
18681
|
+
frozenAt: (/* @__PURE__ */ new Date()).toISOString(),
|
|
18682
|
+
input: options.packet.input,
|
|
18683
|
+
inputSha256: options.packet.inputSha256,
|
|
18684
|
+
question: JEV_QUESTION,
|
|
18685
|
+
provider: JEV_PROVIDER,
|
|
18686
|
+
model: JEV_MODEL,
|
|
18687
|
+
consent: options.consent,
|
|
18688
|
+
authorization: options.authorization,
|
|
18689
|
+
provenance: "tool-selected-input",
|
|
18690
|
+
candidateIdentity: "caller-claimed",
|
|
18691
|
+
sourceBinding: "unassessed",
|
|
18692
|
+
redaction: "unassessed",
|
|
18693
|
+
rights: "unassessed",
|
|
18694
|
+
labelStatus: "unlabeled",
|
|
18695
|
+
trainingEligible: false,
|
|
18696
|
+
exportEligible: false,
|
|
18697
|
+
publicCaptureVerified: false
|
|
18698
|
+
};
|
|
18699
|
+
const bytes = JSON.stringify(source, null, 2) + "\n";
|
|
18700
|
+
return Object.freeze({ source, bytes, sha256: digest2(bytes) });
|
|
18701
|
+
}
|
|
18702
|
+
function privateDirectory(path) {
|
|
18703
|
+
try {
|
|
18704
|
+
mkdirSync5(path, { mode: 448 });
|
|
18705
|
+
} catch (error) {
|
|
18706
|
+
if (error.code !== "EEXIST") throw error;
|
|
18707
|
+
}
|
|
18708
|
+
const stat4 = lstatSync(path);
|
|
18709
|
+
if (!stat4.isDirectory() || stat4.isSymbolicLink())
|
|
18710
|
+
throw new Error("JEV storage must be a private directory, not a link");
|
|
18711
|
+
if (process.platform !== "win32" && ((stat4.mode & 63) !== 0 || process.getuid && stat4.uid !== process.getuid()))
|
|
18712
|
+
throw new Error("JEV storage directory must be owned by the current user with mode 0700");
|
|
18713
|
+
}
|
|
18714
|
+
function writeNew2(path, bytes) {
|
|
18715
|
+
const fd = openSync2(path, "wx", 384);
|
|
18716
|
+
try {
|
|
18717
|
+
writeFileSync8(fd, bytes);
|
|
18718
|
+
fsyncSync2(fd);
|
|
18719
|
+
} finally {
|
|
18720
|
+
closeSync2(fd);
|
|
18721
|
+
}
|
|
18722
|
+
}
|
|
18723
|
+
function retainHandoffSource(source, assertCurrent, storageHome = join34(homedir6(), ".skill-harness")) {
|
|
18724
|
+
assertCurrent();
|
|
18725
|
+
privateDirectory(storageHome);
|
|
18726
|
+
const root = join34(storageHome, "jev-workflow");
|
|
18727
|
+
privateDirectory(root);
|
|
18728
|
+
const directory = mkdtempSync4(join34(root, "selection-"));
|
|
18729
|
+
const path = join34(directory, "selection.json");
|
|
18730
|
+
writeNew2(path, source.bytes);
|
|
18731
|
+
return {
|
|
18732
|
+
path,
|
|
18733
|
+
writeOutcome(outcome) {
|
|
18734
|
+
assertCurrent();
|
|
18735
|
+
writeNew2(join34(directory, "outcome.json"), JSON.stringify({
|
|
18736
|
+
schema: 1,
|
|
18737
|
+
kind: "skill-harness-handoff-advice-v1",
|
|
18738
|
+
source: { path: "selection.json", sha256: source.sha256 },
|
|
18739
|
+
recordedAt: (/* @__PURE__ */ new Date()).toISOString(),
|
|
18740
|
+
outcome,
|
|
18741
|
+
trainingEligible: false,
|
|
18742
|
+
exportEligible: false,
|
|
18743
|
+
labelStatus: "unlabeled"
|
|
18744
|
+
}, null, 2) + "\n");
|
|
18745
|
+
}
|
|
18746
|
+
};
|
|
18747
|
+
}
|
|
18748
|
+
|
|
18600
18749
|
// packages/pi-extension/src/jev-session.ts
|
|
18601
|
-
|
|
18750
|
+
var unmeasuredProvider = () => ({
|
|
18751
|
+
provider: JEV_PROVIDER,
|
|
18752
|
+
requestedModel: JEV_MODEL,
|
|
18753
|
+
resolvedModel: null,
|
|
18754
|
+
usage: { inputTokens: null, outputTokens: null, costUsd: null },
|
|
18755
|
+
latencyMs: null
|
|
18756
|
+
});
|
|
18757
|
+
function createJevController(pi, options = {}) {
|
|
18758
|
+
const run = options.run ?? main;
|
|
18602
18759
|
let epoch = 0;
|
|
18603
18760
|
let active = null;
|
|
18604
18761
|
const clearSession = () => {
|
|
18605
18762
|
epoch++;
|
|
18763
|
+
active?.abort.abort();
|
|
18606
18764
|
active = null;
|
|
18607
18765
|
};
|
|
18608
18766
|
pi.on("session_start", clearSession);
|
|
18609
18767
|
pi.on("session_shutdown", clearSession);
|
|
18610
|
-
|
|
18768
|
+
const currentId = (ctx) => {
|
|
18769
|
+
const id = ctx.sessionManager?.getSessionId();
|
|
18770
|
+
if (active && active.sessionId !== id) clearSession();
|
|
18771
|
+
return id;
|
|
18772
|
+
};
|
|
18773
|
+
const hasKey = () => !!(options.env ?? process.env).OPENROUTER_API_KEY?.trim();
|
|
18774
|
+
const status = (ctx) => {
|
|
18775
|
+
const id = currentId(ctx);
|
|
18776
|
+
const availability = !id ? "missing-session" : !active ? "disabled" : active.mode === "manual" ? "manual-only" : active.blocked ?? (active.busy ? "busy" : active.used >= JEV_WORKFLOW_LIMIT ? "limit-reached" : !hasKey() ? "missing-key" : "ready");
|
|
18777
|
+
return {
|
|
18778
|
+
enabled: active !== null,
|
|
18779
|
+
mode: active?.mode ?? "disabled",
|
|
18780
|
+
storage: active?.consent.decision ?? null,
|
|
18781
|
+
remaining: active?.mode === "workflow" ? JEV_WORKFLOW_LIMIT - active.used : 0,
|
|
18782
|
+
availability,
|
|
18783
|
+
advisory: true
|
|
18784
|
+
};
|
|
18785
|
+
};
|
|
18786
|
+
const evaluate = async (value, toolCallId, signal, ctx) => {
|
|
18787
|
+
requireIdentity(toolCallId, "Pi toolCallId");
|
|
18788
|
+
const packet = prepareHandoff(value);
|
|
18789
|
+
const id = currentId(ctx);
|
|
18790
|
+
const unavailable = (reason) => ({
|
|
18791
|
+
advisory: true,
|
|
18792
|
+
status: "unavailable",
|
|
18793
|
+
probability: null,
|
|
18794
|
+
reason,
|
|
18795
|
+
...unmeasuredProvider(),
|
|
18796
|
+
remaining: active?.mode === "workflow" ? JEV_WORKFLOW_LIMIT - active.used : 0
|
|
18797
|
+
});
|
|
18798
|
+
if (!id || !active || active.mode !== "workflow")
|
|
18799
|
+
return unavailable("Explicit /skill-harness jev enable workflow activation is required in this Pi session.");
|
|
18800
|
+
const bound = active;
|
|
18801
|
+
const assertCurrent = () => {
|
|
18802
|
+
if (active !== bound || ctx.sessionManager?.getSessionId() !== bound.sessionId || bound.abort.signal.aborted || signal?.aborted)
|
|
18803
|
+
throw new Error("Session, authorization or tool execution changed");
|
|
18804
|
+
};
|
|
18805
|
+
if (signal?.aborted) return unavailable("Tool execution was cancelled.");
|
|
18806
|
+
const cached = bound.cache.get(packet.inputSha256);
|
|
18807
|
+
if (cached) return { ...cached, remaining: JEV_WORKFLOW_LIMIT - bound.used, reused: true };
|
|
18808
|
+
if (bound.blocked) return unavailable("Workflow advice is suppressed after an error; explicit new activation is required.");
|
|
18809
|
+
if (bound.busy) return unavailable("An advisory request is already in flight; no additional call was made.");
|
|
18810
|
+
if (bound.used >= JEV_WORKFLOW_LIMIT) return unavailable("This activation's advisory call limit is reached.");
|
|
18811
|
+
if (!hasKey()) return unavailable("OPENROUTER_API_KEY is unavailable; no call was made.");
|
|
18812
|
+
const abort = new AbortController();
|
|
18813
|
+
const cancel = () => abort.abort();
|
|
18814
|
+
const signals = [bound.abort.signal, signal].filter((s) => s !== void 0);
|
|
18815
|
+
for (const s of signals) {
|
|
18816
|
+
if (s.aborted) cancel();
|
|
18817
|
+
else s.addEventListener("abort", cancel, { once: true });
|
|
18818
|
+
}
|
|
18819
|
+
bound.busy = true;
|
|
18820
|
+
bound.used++;
|
|
18821
|
+
let receipt;
|
|
18822
|
+
try {
|
|
18823
|
+
assertCurrent();
|
|
18824
|
+
const source = createHandoffSource({
|
|
18825
|
+
packet,
|
|
18826
|
+
sessionId: bound.sessionId,
|
|
18827
|
+
toolCallId,
|
|
18828
|
+
consent: bound.consent,
|
|
18829
|
+
authorization: bound.authorization
|
|
18830
|
+
});
|
|
18831
|
+
const retained = bound.consent.decision === "granted" ? retainHandoffSource(source, assertCurrent, options.storageHome) : void 0;
|
|
18832
|
+
receipt = {
|
|
18833
|
+
kind: JEV_SOURCE_KIND,
|
|
18834
|
+
sha256: source.sha256,
|
|
18835
|
+
sessionId: bound.sessionId,
|
|
18836
|
+
toolCallId,
|
|
18837
|
+
retained: !!retained,
|
|
18838
|
+
...retained ? { path: retained.path } : {}
|
|
18839
|
+
};
|
|
18840
|
+
assertCurrent();
|
|
18841
|
+
const result = await (options.provider ?? callProvider)(JEV_PROVIDER, JEV_MODEL, packet, {
|
|
18842
|
+
apiKey: (options.env ?? process.env).OPENROUTER_API_KEY,
|
|
18843
|
+
signal: abort.signal
|
|
18844
|
+
});
|
|
18845
|
+
assertCurrent();
|
|
18846
|
+
if (result.status !== "answered") bound.blocked = "provider-error";
|
|
18847
|
+
const advice = {
|
|
18848
|
+
advisory: true,
|
|
18849
|
+
status: result.status === "answered" ? "answered" : "unavailable",
|
|
18850
|
+
probability: result.status === "answered" ? result.probability : null,
|
|
18851
|
+
provider: JEV_PROVIDER,
|
|
18852
|
+
requestedModel: JEV_MODEL,
|
|
18853
|
+
resolvedModel: result.resolvedModel,
|
|
18854
|
+
usage: result.usage,
|
|
18855
|
+
latencyMs: result.latencyMs,
|
|
18856
|
+
...result.status === "answered" ? {} : { reason: "Provider unavailable; further workflow calls require explicit new activation." },
|
|
18857
|
+
remaining: JEV_WORKFLOW_LIMIT - bound.used,
|
|
18858
|
+
inputSha256: packet.inputSha256,
|
|
18859
|
+
source: receipt
|
|
18860
|
+
};
|
|
18861
|
+
try {
|
|
18862
|
+
retained?.writeOutcome(advice);
|
|
18863
|
+
} catch {
|
|
18864
|
+
assertCurrent();
|
|
18865
|
+
bound.blocked = "storage-error";
|
|
18866
|
+
advice.reason = "Provider outcome received, but outcome retention failed; further workflow calls require explicit new activation.";
|
|
18867
|
+
}
|
|
18868
|
+
bound.cache.set(packet.inputSha256, Object.freeze(advice));
|
|
18869
|
+
return advice;
|
|
18870
|
+
} catch {
|
|
18871
|
+
bound.blocked = "execution-error";
|
|
18872
|
+
return {
|
|
18873
|
+
advisory: true,
|
|
18874
|
+
status: "unavailable",
|
|
18875
|
+
probability: null,
|
|
18876
|
+
...unmeasuredProvider(),
|
|
18877
|
+
reason: "Advice unavailable or cancelled; no approval is implied. Explicit new activation is required for further workflow calls.",
|
|
18878
|
+
remaining: JEV_WORKFLOW_LIMIT - bound.used,
|
|
18879
|
+
inputSha256: packet.inputSha256,
|
|
18880
|
+
...receipt ? { source: receipt } : {}
|
|
18881
|
+
};
|
|
18882
|
+
} finally {
|
|
18883
|
+
bound.busy = false;
|
|
18884
|
+
for (const s of signals) s.removeEventListener("abort", cancel);
|
|
18885
|
+
}
|
|
18886
|
+
};
|
|
18887
|
+
const command = async (args, ctx) => {
|
|
18611
18888
|
const action = args.trim() || "status";
|
|
18612
|
-
const sessionId = ctx
|
|
18889
|
+
const sessionId = currentId(ctx);
|
|
18613
18890
|
if (!sessionId) throw Error("JEV requires the current Pi session identity");
|
|
18614
|
-
|
|
18615
|
-
epoch++;
|
|
18616
|
-
active = null;
|
|
18617
|
-
}
|
|
18891
|
+
requireIdentity(sessionId, "Pi sessionId");
|
|
18618
18892
|
if (action === "status") {
|
|
18619
18893
|
ctx.ui.notify(
|
|
18620
|
-
|
|
18894
|
+
JSON.stringify(status(ctx))
|
|
18621
18895
|
);
|
|
18622
18896
|
return;
|
|
18623
18897
|
}
|
|
18624
18898
|
if (action === "disable") {
|
|
18625
|
-
|
|
18626
|
-
active = null;
|
|
18899
|
+
clearSession();
|
|
18627
18900
|
ctx.ui.notify(
|
|
18628
18901
|
"JEV disabled. Existing selected records remain local; disable storage in their review before export if needed."
|
|
18629
18902
|
);
|
|
@@ -18633,15 +18906,35 @@ function createJevSessionHandler(pi, run = main) {
|
|
|
18633
18906
|
throw Error(
|
|
18634
18907
|
"Interactive JEV activation requires a per-session storage answer. Use the explicit CLI storage option in noninteractive mode."
|
|
18635
18908
|
);
|
|
18636
|
-
if (action === "enable") {
|
|
18637
|
-
|
|
18638
|
-
|
|
18909
|
+
if (action === "enable" || action === "enable workflow") {
|
|
18910
|
+
clearSession();
|
|
18911
|
+
const activation = epoch;
|
|
18912
|
+
const workflow = action === "enable workflow";
|
|
18913
|
+
let authorization = null;
|
|
18914
|
+
if (workflow) {
|
|
18915
|
+
if (!ctx.ui.confirm) throw new Error("Workflow JEV activation requires interactive paid-scope confirmation");
|
|
18916
|
+
const confirmed = await ctx.ui.confirm(
|
|
18917
|
+
"Enable optional JEV handoff advice for THIS Pi session?",
|
|
18918
|
+
`Allow up to ${JEV_WORKFLOW_LIMIT} automatic workflow paid calls to ${JEV_MODEL} through OpenRouter using OPENROUTER_API_KEY. The coordinator may send selected candidate, acceptance requirements and evidence (at most 16000 characters per packet) for the fixed handoff-readiness question when uncertainty remains after deterministic checks, before independent review. Advice never grants approval. This is separate from your Pi subscription; no retry or fallback. Manual jev run calls remain separately confirmed outside this workflow limit.`
|
|
18919
|
+
);
|
|
18920
|
+
if (activation !== epoch || ctx.sessionManager?.getSessionId() !== sessionId || confirmed !== true) return;
|
|
18921
|
+
authorization = {
|
|
18922
|
+
kind: "jev-workflow-paid-scope",
|
|
18923
|
+
interactionId: randomUUID3(),
|
|
18924
|
+
sessionId,
|
|
18925
|
+
recordedAt: (/* @__PURE__ */ new Date()).toISOString(),
|
|
18926
|
+
provider: JEV_PROVIDER,
|
|
18927
|
+
model: JEV_MODEL,
|
|
18928
|
+
question: JEV_QUESTION,
|
|
18929
|
+
maximumCalls: JEV_WORKFLOW_LIMIT
|
|
18930
|
+
};
|
|
18931
|
+
}
|
|
18639
18932
|
const choices = [
|
|
18640
18933
|
"No \u2014 use JEV without retaining data for LoRA",
|
|
18641
18934
|
"Yes \u2014 retain selected decision data for later LoRA review"
|
|
18642
18935
|
];
|
|
18643
18936
|
const answer = await ctx.ui.select(
|
|
18644
|
-
"Store selected data from THIS session for future LoRA dataset review? Storage does not grant training permission.",
|
|
18937
|
+
"Store selected data from THIS session for future LoRA dataset review? Workflow selections are saved under ~/.skill-harness/jev-workflow. Storage does not grant training permission; No does not change Pi native session history.",
|
|
18645
18938
|
choices
|
|
18646
18939
|
);
|
|
18647
18940
|
if (activation !== epoch || ctx.sessionManager?.getSessionId() !== sessionId)
|
|
@@ -18654,14 +18947,24 @@ function createJevSessionHandler(pi, run = main) {
|
|
|
18654
18947
|
recordedAt: (/* @__PURE__ */ new Date()).toISOString()
|
|
18655
18948
|
});
|
|
18656
18949
|
pi.appendEntry?.("skill-harness-jev-storage-choice", consent);
|
|
18657
|
-
active = {
|
|
18950
|
+
active = {
|
|
18951
|
+
sessionId,
|
|
18952
|
+
consent,
|
|
18953
|
+
authorization,
|
|
18954
|
+
mode: workflow ? "workflow" : "manual",
|
|
18955
|
+
abort: new AbortController(),
|
|
18956
|
+
used: 0,
|
|
18957
|
+
busy: false,
|
|
18958
|
+
blocked: null,
|
|
18959
|
+
cache: /* @__PURE__ */ new Map()
|
|
18960
|
+
};
|
|
18658
18961
|
ctx.ui.notify(
|
|
18659
|
-
"JEV enabled for explicitly selected paid calls. LoRA storage " + consent.decision + " for this session only."
|
|
18962
|
+
(workflow ? `JEV workflow advice enabled (up to ${JEV_WORKFLOW_LIMIT} paid calls). LoRA storage ` : "JEV enabled for explicitly selected paid calls. LoRA storage ") + consent.decision + " for this session only."
|
|
18660
18963
|
);
|
|
18661
18964
|
return;
|
|
18662
18965
|
}
|
|
18663
18966
|
if (action !== "run")
|
|
18664
|
-
throw Error("usage: /skill-harness jev enable | status | disable | run");
|
|
18967
|
+
throw Error("usage: /skill-harness jev enable [workflow] | status | disable | run");
|
|
18665
18968
|
if (!active)
|
|
18666
18969
|
throw Error(
|
|
18667
18970
|
"Enable JEV in this session first: /skill-harness jev enable"
|
|
@@ -18751,13 +19054,59 @@ function createJevSessionHandler(pi, run = main) {
|
|
|
18751
19054
|
}
|
|
18752
19055
|
);
|
|
18753
19056
|
};
|
|
19057
|
+
return { command, status, evaluate };
|
|
19058
|
+
}
|
|
19059
|
+
|
|
19060
|
+
// packages/pi-extension/src/jev-advice.ts
|
|
19061
|
+
import { Type as Type3 } from "typebox";
|
|
19062
|
+
function createJevAdviceTool(controller) {
|
|
19063
|
+
return {
|
|
19064
|
+
name: "jev_advice",
|
|
19065
|
+
label: "JEV handoff advice",
|
|
19066
|
+
description: "Check session-local JEV workflow status without a provider call, or evaluate selected handoff evidence against the fixed acceptance-readiness question. Evaluation requires the user's explicit /skill-harness jev enable workflow activation. Advisory probability never grants approval or replaces tests/review. Select only relevant public or already-redacted evidence; do not include secrets, transcripts, reviewer verdicts or hidden reasoning.",
|
|
19067
|
+
promptGuidelines: [
|
|
19068
|
+
"Use jev_advice action=status to discover whether JEV workflow advice is enabled.",
|
|
19069
|
+
"When enabled and useful uncertainty remains after deterministic checks, optionally evaluate a selected handoff before independent review. Do not call for every feature or retry unavailable advice.",
|
|
19070
|
+
"Keep JEV advice out of the independent reviewer's inputs; act on independently verified issues rather than treating its probability as a verdict."
|
|
19071
|
+
],
|
|
19072
|
+
parameters: Type3.Object({
|
|
19073
|
+
action: Type3.Union([Type3.Literal("status"), Type3.Literal("evaluate")]),
|
|
19074
|
+
candidate: Type3.Optional(Type3.String({ minLength: 1, maxLength: 16e3, description: "Reported candidate identity; this is a claim, not an authenticated Git binding." })),
|
|
19075
|
+
requirements: Type3.Optional(Type3.String({ minLength: 1, maxLength: 16e3, description: "Explicit acceptance checks, before seeing independent review or JEV outcomes." })),
|
|
19076
|
+
evidence: Type3.Optional(Type3.String({ minLength: 1, maxLength: 16e3, description: "Selected decision-time handoff evidence. Complete JSON packet is limited to 16000 characters." }))
|
|
19077
|
+
}, { additionalProperties: false }),
|
|
19078
|
+
async execute(id, params, signal, _onUpdate, ctx) {
|
|
19079
|
+
requireIdentity(id, "Pi toolCallId");
|
|
19080
|
+
if (!params || typeof params !== "object" || Array.isArray(params)) throw new Error("Invalid jev_advice parameters");
|
|
19081
|
+
const p = params;
|
|
19082
|
+
const expected = p.action === "status" ? ["action"] : ["action", "candidate", "requirements", "evidence"];
|
|
19083
|
+
if (Object.keys(p).length !== expected.length || expected.some((key) => !Object.hasOwn(p, key)))
|
|
19084
|
+
throw new Error("jev_advice accepts only its documented status or evaluate fields");
|
|
19085
|
+
let details;
|
|
19086
|
+
if (p.action === "status") details = controller.status(ctx);
|
|
19087
|
+
else if (p.action === "evaluate")
|
|
19088
|
+
details = await controller.evaluate(
|
|
19089
|
+
{ candidate: p.candidate, requirements: p.requirements, evidence: p.evidence },
|
|
19090
|
+
id,
|
|
19091
|
+
signal,
|
|
19092
|
+
ctx
|
|
19093
|
+
);
|
|
19094
|
+
else throw new Error("jev_advice action must be status or evaluate");
|
|
19095
|
+
return { content: [{ type: "text", text: JSON.stringify(details) }], details };
|
|
19096
|
+
}
|
|
19097
|
+
};
|
|
19098
|
+
}
|
|
19099
|
+
function registerJevAdvice(pi, controller) {
|
|
19100
|
+
pi.registerTool(createJevAdviceTool(controller));
|
|
18754
19101
|
}
|
|
18755
19102
|
|
|
18756
19103
|
// packages/pi-extension/src/index.ts
|
|
18757
19104
|
function index_default(pi) {
|
|
18758
19105
|
const moduleDir = dirname10(fileURLToPath2(import.meta.url));
|
|
18759
|
-
const assetsDir = basename4(dirname10(moduleDir)) === "skill-harness" ?
|
|
18760
|
-
|
|
19106
|
+
const assetsDir = basename4(dirname10(moduleDir)) === "skill-harness" ? join35(moduleDir, "..", "assets") : join35(moduleDir, "..", "..", "..", "assets");
|
|
19107
|
+
const jev = createJevController(pi);
|
|
19108
|
+
registerCommand(pi, assetsDir, jev.command);
|
|
19109
|
+
registerJevAdvice(pi, jev);
|
|
18761
19110
|
registerTool(pi);
|
|
18762
19111
|
pi.on("session_shutdown", async () => {
|
|
18763
19112
|
closeReview();
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "skill-harness",
|
|
3
|
-
"version": "0.
|
|
4
|
-
"description": "Test/optimize loop for agent skills
|
|
3
|
+
"version": "0.26.0",
|
|
4
|
+
"description": "Test/optimize loop for agent skills \u2014 run spec'd scenarios on pi, LLM-judge, score, review, re-run",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"keywords": [
|
|
7
7
|
"agent-skills",
|
|
@@ -31,7 +31,7 @@
|
|
|
31
31
|
"README.md"
|
|
32
32
|
],
|
|
33
33
|
"dependencies": {
|
|
34
|
-
"@skill-harness/cli": "0.
|
|
34
|
+
"@skill-harness/cli": "0.26.0"
|
|
35
35
|
},
|
|
36
36
|
"peerDependencies": {
|
|
37
37
|
"typebox": "*"
|