ll-skills 2.0.2 → 3.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +21 -0
- package/README.md +42 -20
- package/agents/ll-executor.md +1 -0
- package/assets/preamble.md +29 -35
- package/bin/install.js +4 -1
- package/hooks/ll-precompact.js +29 -1
- package/hooks/ll-skills-check-update.js +6 -6
- package/hooks/ll-state.js +30 -2
- package/package.json +3 -2
- package/scripts/evals/README.md +57 -0
- package/scripts/evals/cases/auto-dry-run/assert.sh +35 -0
- package/scripts/evals/cases/auto-dry-run/case.json +8 -0
- package/scripts/evals/cases/auto-dry-run/prompt.txt +1 -0
- package/scripts/evals/cases/auto-empty-repo/assert.sh +25 -0
- package/scripts/evals/cases/auto-empty-repo/case.json +8 -0
- package/scripts/evals/cases/auto-empty-repo/fixture/.gitkeep +0 -0
- package/scripts/evals/cases/auto-empty-repo/prompt.txt +1 -0
- package/scripts/evals/cases/decide-final-round/assert.sh +32 -0
- package/scripts/evals/cases/decide-final-round/case.json +8 -0
- package/scripts/evals/cases/decide-final-round/fixture/README.md +3 -0
- package/scripts/evals/cases/decide-final-round/prompt.txt +1 -0
- package/scripts/evals/cases/executor-block/assert.sh +33 -0
- package/scripts/evals/cases/executor-block/case.json +8 -0
- package/scripts/evals/cases/executor-block/prompt.txt +14 -0
- package/scripts/evals/cases/goal-autonomous/assert.sh +35 -0
- package/scripts/evals/cases/goal-autonomous/case.json +8 -0
- package/scripts/evals/cases/goal-autonomous/fixture/PLAN.md +42 -0
- package/scripts/evals/cases/goal-autonomous/fixture/PROGRESS.md +20 -0
- package/scripts/evals/cases/goal-autonomous/fixture/ROADMAP.md +29 -0
- package/scripts/evals/cases/goal-autonomous/fixture/package.json +8 -0
- package/scripts/evals/cases/goal-autonomous/fixture/src/money.js +6 -0
- package/scripts/evals/cases/goal-autonomous/fixture/test/reconcile.test.js +8 -0
- package/scripts/evals/cases/goal-autonomous/prompt.txt +1 -0
- package/scripts/evals/cases/implement-review-gate/assert.sh +35 -0
- package/scripts/evals/cases/implement-review-gate/case.json +8 -0
- package/scripts/evals/cases/implement-review-gate/prompt.txt +1 -0
- package/scripts/evals/cases/implement-stops-at-next/assert.sh +39 -0
- package/scripts/evals/cases/implement-stops-at-next/case.json +9 -0
- package/scripts/evals/cases/implement-stops-at-next/prompt.txt +1 -0
- package/scripts/evals/cases/preamble-no-ritual/assert.sh +17 -0
- package/scripts/evals/cases/preamble-no-ritual/case.json +8 -0
- package/scripts/evals/cases/preamble-no-ritual/fixture/README.md +3 -0
- package/scripts/evals/cases/preamble-no-ritual/fixture/src/a.ts +3 -0
- package/scripts/evals/cases/preamble-no-ritual/prompt.txt +1 -0
- package/scripts/evals/cases/router-execute/assert.sh +12 -0
- package/scripts/evals/cases/router-execute/case.json +8 -0
- package/scripts/evals/cases/router-execute/prompt.txt +1 -0
- package/scripts/evals/cases/router-research/assert.sh +11 -0
- package/scripts/evals/cases/router-research/case.json +8 -0
- package/scripts/evals/cases/router-research/fixture/README.md +3 -0
- package/scripts/evals/cases/router-research/prompt.txt +1 -0
- package/scripts/evals/cases/router-small/assert.sh +21 -0
- package/scripts/evals/cases/router-small/case.json +8 -0
- package/scripts/evals/cases/router-small/fixture/README.md +17 -0
- package/scripts/evals/cases/router-small/prompt.txt +1 -0
- package/scripts/evals/cases/scout-no-plan/assert.sh +41 -0
- package/scripts/evals/cases/scout-no-plan/case.json +8 -0
- package/scripts/evals/cases/scout-no-plan/prompt.txt +8 -0
- package/scripts/evals/cases/verifier-weakened-test/assert.sh +19 -0
- package/scripts/evals/cases/verifier-weakened-test/case.json +8 -0
- package/scripts/evals/cases/verifier-weakened-test/prompt.txt +13 -0
- package/scripts/evals/cases/verifier-weakened-test/setup.sh +19 -0
- package/scripts/evals/fixtures/manual-contract/out.json +29 -0
- package/scripts/evals/fixtures/manual-contract/out.txt +5 -0
- package/scripts/evals/fixtures/manual-contract/with-skill.json +46 -0
- package/scripts/evals/lib/assert.sh +107 -0
- package/scripts/evals/lib/extract.js +73 -0
- package/scripts/evals/run.sh +369 -0
- package/scripts/fixtures/auto-closed/PLAN.md +5 -0
- package/scripts/fixtures/auto-closed/PROGRESS.md +20 -0
- package/scripts/fixtures/auto-closed/ROADMAP.md +6 -0
- package/scripts/fixtures/auto-closed/docs/DELIVERY.md +3 -0
- package/scripts/fixtures/auto-decisions/decisions/DEC-0001-taken-alone.md +13 -0
- package/scripts/fixtures/auto-decisions/decisions/DEC-0002-owner.md +13 -0
- package/scripts/fixtures/auto-noroadmap/PLAN.md +20 -0
- package/scripts/fixtures/auto-noroadmap/PROGRESS.md +11 -0
- package/scripts/fixtures/auto-verify-next/PLAN.md +5 -0
- package/scripts/fixtures/auto-verify-next/PROGRESS.md +18 -0
- package/scripts/fixtures/auto-verify-next/ROADMAP.md +5 -0
- package/scripts/fixtures/auto-verify-next/phases/01/PLAN.md +6 -0
- package/scripts/fixtures/evals-auto/auto-dry-run/pass.txt +18 -0
- package/scripts/fixtures/evals-auto/auto-empty-repo/pass.txt +2 -0
- package/scripts/fixtures/evals-auto/goal-autonomous/pass.txt +29 -0
- package/scripts/fixtures/lint-bad/folded-description/SKILL.md +13 -0
- package/scripts/fixtures/lint-bad/model-invocation-false/SKILL.md +10 -0
- package/scripts/fixtures/next-bad/skills/ll-bad/SKILL.md +30 -0
- package/scripts/fixtures/next-good/skills/ll-good/SKILL.md +26 -0
- package/scripts/fixtures/project/PROGRESS.md +4 -0
- package/scripts/lint-contract.cjs +495 -0
- package/scripts/lint-prompts.sh +396 -0
- package/scripts/ll-tools.js +465 -447
- package/scripts/smoke-test.sh +380 -1
- package/skills/ll-auto/SKILL.md +74 -0
- package/skills/ll-auto/references/run.md +75 -0
- package/skills/ll-auto/references/stages.md +66 -0
- package/skills/ll-auto/scripts/ll-auto.js +345 -0
- package/skills/ll-brainstorm/SKILL.md +5 -4
- package/skills/ll-brainstorm/references/decision-policy.md +3 -0
- package/skills/ll-close/SKILL.md +5 -5
- package/skills/ll-close/references/delivery.md +3 -1
- package/skills/ll-decide/SKILL.md +13 -11
- package/skills/ll-decide/references/decision-policy.md +3 -0
- package/skills/ll-decide/references/interview.md +10 -0
- package/skills/ll-decide/references/plan-skeleton.md +14 -14
- package/skills/ll-decide/references/premise-gate.md +7 -0
- package/skills/ll-goal/SKILL.md +22 -4
- package/skills/ll-goal/references/goal-template.md +57 -0
- package/skills/ll-implement/SKILL.md +4 -2
- package/skills/ll-implement/references/decision-policy.md +3 -0
- package/skills/ll-oncall/SKILL.md +3 -2
- package/skills/ll-refine/SKILL.md +3 -2
- package/skills/ll-research/SKILL.md +3 -2
- package/skills/ll-resume/SKILL.md +4 -3
- package/skills/ll-update/SKILL.md +6 -1
- package/skills/ll-verify/SKILL.md +2 -1
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
[
|
|
2
|
+
{
|
|
3
|
+
"type": "system",
|
|
4
|
+
"subtype": "init",
|
|
5
|
+
"session_id": "manual-contract-fixture",
|
|
6
|
+
"model": "claude-sonnet-5"
|
|
7
|
+
},
|
|
8
|
+
{
|
|
9
|
+
"type": "assistant",
|
|
10
|
+
"message": {
|
|
11
|
+
"role": "assistant",
|
|
12
|
+
"content": [
|
|
13
|
+
{
|
|
14
|
+
"type": "text",
|
|
15
|
+
"text": "Comparing authentication providers for the signup flow is a research task, not something I can answer from what is already in the repo. Paste this to start it:\n\n/ll-research compare authentication providers for the signup flow\n\n▶ Next — `/clear` then `/ll-research compare authentication providers for the signup flow`"
|
|
16
|
+
}
|
|
17
|
+
]
|
|
18
|
+
}
|
|
19
|
+
},
|
|
20
|
+
{
|
|
21
|
+
"type": "result",
|
|
22
|
+
"subtype": "success",
|
|
23
|
+
"is_error": false,
|
|
24
|
+
"duration_ms": 1820,
|
|
25
|
+
"num_turns": 1,
|
|
26
|
+
"total_cost_usd": 0.0091,
|
|
27
|
+
"result": "Comparing authentication providers for the signup flow is a research task, not something I can answer from what is already in the repo. Paste this to start it:\n\n/ll-research compare authentication providers for the signup flow\n\n▶ Next — `/clear` then `/ll-research compare authentication providers for the signup flow`"
|
|
28
|
+
}
|
|
29
|
+
]
|
|
@@ -0,0 +1,5 @@
|
|
|
1
|
+
Comparing authentication providers for the signup flow is a research task, not something I can answer from what is already in the repo. Paste this to start it:
|
|
2
|
+
|
|
3
|
+
/ll-research compare authentication providers for the signup flow
|
|
4
|
+
|
|
5
|
+
▶ Next — `/clear` then `/ll-research compare authentication providers for the signup flow`
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
[
|
|
2
|
+
{
|
|
3
|
+
"type": "system",
|
|
4
|
+
"subtype": "init",
|
|
5
|
+
"session_id": "manual-contract-fixture-with-skill",
|
|
6
|
+
"model": "claude-sonnet-5"
|
|
7
|
+
},
|
|
8
|
+
{
|
|
9
|
+
"type": "assistant",
|
|
10
|
+
"message": {
|
|
11
|
+
"role": "assistant",
|
|
12
|
+
"content": [
|
|
13
|
+
{
|
|
14
|
+
"type": "text",
|
|
15
|
+
"text": "Comparing authentication providers for the signup flow is a research task, not something I can answer from what is already in the repo. Paste this to start it:\n\n/ll-research compare authentication providers for the signup flow\n\n▶ Next — `/clear` then `/ll-research compare authentication providers for the signup flow`"
|
|
16
|
+
}
|
|
17
|
+
]
|
|
18
|
+
}
|
|
19
|
+
},
|
|
20
|
+
{
|
|
21
|
+
"type": "assistant",
|
|
22
|
+
"message": {
|
|
23
|
+
"role": "assistant",
|
|
24
|
+
"content": [
|
|
25
|
+
{
|
|
26
|
+
"type": "tool_use",
|
|
27
|
+
"id": "toolu_01manualcontract",
|
|
28
|
+
"name": "Skill",
|
|
29
|
+
"input": {
|
|
30
|
+
"skill": "ll-research",
|
|
31
|
+
"args": "compare authentication providers for the signup flow"
|
|
32
|
+
}
|
|
33
|
+
}
|
|
34
|
+
]
|
|
35
|
+
}
|
|
36
|
+
},
|
|
37
|
+
{
|
|
38
|
+
"type": "result",
|
|
39
|
+
"subtype": "success",
|
|
40
|
+
"is_error": false,
|
|
41
|
+
"duration_ms": 2140,
|
|
42
|
+
"num_turns": 2,
|
|
43
|
+
"total_cost_usd": 0.0134,
|
|
44
|
+
"result": "Comparing authentication providers for the signup flow is a research task, not something I can answer from what is already in the repo. Paste this to start it:\n\n/ll-research compare authentication providers for the signup flow\n\n▶ Next — `/clear` then `/ll-research compare authentication providers for the signup flow`"
|
|
45
|
+
}
|
|
46
|
+
]
|
|
@@ -0,0 +1,107 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# Sourced by every cases/<id>/assert.sh.
|
|
3
|
+
#
|
|
4
|
+
# Contract of an assert script:
|
|
5
|
+
# bash assert.sh <workdir> <out.json> <out.txt> -> exit 0 pass, 1 fail
|
|
6
|
+
# It prints one line per check: "ok: <what>" or "FAIL: <what>".
|
|
7
|
+
# run.sh reads the first FAIL line into the results table.
|
|
8
|
+
#
|
|
9
|
+
# Provided helpers:
|
|
10
|
+
# ok <msg> record a passed check
|
|
11
|
+
# fail <msg> record a failed check
|
|
12
|
+
# check <cond-exit> <msg> record from an exit code already computed
|
|
13
|
+
# contains <file> <regex> <msg> grep -Eq
|
|
14
|
+
# absent <file> <regex> <msg> grep -Eq must not match
|
|
15
|
+
# no_path <path> <msg> path must not exist
|
|
16
|
+
# no_tool_use <out.json> <tool-name> <msg> no tool_use block named <tool-name> anywhere
|
|
17
|
+
# finish exit with the verdict
|
|
18
|
+
|
|
19
|
+
EVAL_FAILURES=0
|
|
20
|
+
|
|
21
|
+
ok() { printf 'ok: %s\n' "$1"; }
|
|
22
|
+
fail() { printf 'FAIL: %s\n' "$1"; EVAL_FAILURES=$((EVAL_FAILURES + 1)); }
|
|
23
|
+
|
|
24
|
+
check() { # check <exit-code> <msg>
|
|
25
|
+
if [ "$1" -eq 0 ]; then ok "$2"; else fail "$2"; fi
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
contains() { # contains <file> <extended-regex> <msg>
|
|
29
|
+
if [ -f "$1" ] && grep -Eq -- "$2" "$1"; then ok "$3"; else fail "$3"; fi
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
absent() { # absent <file> <extended-regex> <msg>
|
|
33
|
+
if [ ! -f "$1" ] || ! grep -Eq -- "$2" "$1"; then ok "$3"; else fail "$3"; fi
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
no_path() { # no_path <path> <msg>
|
|
37
|
+
if [ ! -e "$1" ]; then ok "$2"; else fail "$2"; fi
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
first_line() { # first_line <file> -> first non-empty line
|
|
41
|
+
[ -f "$1" ] || return 0
|
|
42
|
+
grep -m1 -v '^[[:space:]]*$' "$1" 2>/dev/null || true
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
# The regime is stated "before doing anything": it belongs to the first assistant
|
|
46
|
+
# message of the run, not to the final answer. This reads it out of the capture.
|
|
47
|
+
first_text() { # first_text <out.json> -> the whole first assistant message
|
|
48
|
+
node "$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)/extract.js" "$1" first_text 2>/dev/null || true
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
first_text_line() { # first_text_line <out.json> -> first non-empty line of the first assistant text
|
|
52
|
+
first_text "$1" | grep -m1 -v '^[[:space:]]*$' || true
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
first_text_contains() { # first_text_contains <out.json> <extended-regex> <msg>
|
|
56
|
+
if first_text "$1" | grep -Eq -- "$2"; then ok "$3"; else fail "$3"; fi
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
# no_tool_use <out.json> <tool-name> <msg>
|
|
60
|
+
# Scans every assistant event's message.content for a tool_use block whose
|
|
61
|
+
# `name` equals <tool-name>. Passes when none is found; fails naming the
|
|
62
|
+
# first hit (the manual contract: no skill is started by a tool call).
|
|
63
|
+
# A capture that is missing or is not readable JSON proves nothing: it fails
|
|
64
|
+
# (B-003/B-020 — a rep whose out.json never landed used to score PASS).
|
|
65
|
+
no_tool_use() {
|
|
66
|
+
local file="$1" name="$2" msg="$3" hit rc
|
|
67
|
+
if [ ! -f "$file" ]; then
|
|
68
|
+
fail "$msg: no capture at $file"
|
|
69
|
+
return
|
|
70
|
+
fi
|
|
71
|
+
hit="$(node -e '
|
|
72
|
+
const fs = require("fs");
|
|
73
|
+
let data;
|
|
74
|
+
try { data = JSON.parse(fs.readFileSync(process.argv[1], "utf8")); }
|
|
75
|
+
catch { process.exit(3); }
|
|
76
|
+
const events = Array.isArray(data) ? data : [data];
|
|
77
|
+
const wanted = process.argv[2];
|
|
78
|
+
for (const ev of events) {
|
|
79
|
+
if (!ev || ev.type !== "assistant") continue;
|
|
80
|
+
const content = ev.message && ev.message.content;
|
|
81
|
+
if (!Array.isArray(content)) continue;
|
|
82
|
+
for (const b of content) {
|
|
83
|
+
if (b && b.type === "tool_use" && b.name === wanted) {
|
|
84
|
+
process.stdout.write(b.name);
|
|
85
|
+
process.exit(0);
|
|
86
|
+
}
|
|
87
|
+
}
|
|
88
|
+
}
|
|
89
|
+
' "$file" "$name" 2>/dev/null)"
|
|
90
|
+
rc=$?
|
|
91
|
+
if [ "$rc" -ne 0 ]; then
|
|
92
|
+
fail "$msg: the capture at $file is not readable JSON"
|
|
93
|
+
elif [ -z "$hit" ]; then
|
|
94
|
+
ok "$msg"
|
|
95
|
+
else
|
|
96
|
+
fail "$msg: found a $hit tool_use call"
|
|
97
|
+
fi
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
base_sha() { # base_sha <workdir> -> the HEAD recorded before the run, or empty
|
|
101
|
+
[ -f "$1/.eval-base-sha" ] && cat "$1/.eval-base-sha"
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
finish() {
|
|
105
|
+
if [ "$EVAL_FAILURES" -eq 0 ]; then exit 0; fi
|
|
106
|
+
exit 1
|
|
107
|
+
}
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
'use strict';
|
|
3
|
+
|
|
4
|
+
// Reads one field of the `type: "result"` element of a `claude --output-format json`
|
|
5
|
+
// capture. The capture is either a single object or an array of events
|
|
6
|
+
// (system/init … assistant … result); only the result element carries the totals.
|
|
7
|
+
//
|
|
8
|
+
// node extract.js <out.json> [field] field defaults to "result"
|
|
9
|
+
//
|
|
10
|
+
// Exits 1 when the file is not JSON or has no result element, so the caller can
|
|
11
|
+
// tell "the run produced nothing" from "the run produced an empty answer".
|
|
12
|
+
|
|
13
|
+
const fs = require('fs');
|
|
14
|
+
|
|
15
|
+
const file = process.argv[2];
|
|
16
|
+
const field = process.argv[3] || 'result';
|
|
17
|
+
|
|
18
|
+
let raw;
|
|
19
|
+
try {
|
|
20
|
+
raw = fs.readFileSync(file, 'utf8');
|
|
21
|
+
} catch (e) {
|
|
22
|
+
process.stderr.write(`extract: cannot read ${file}: ${e.message}\n`);
|
|
23
|
+
process.exit(1);
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
let data;
|
|
27
|
+
try {
|
|
28
|
+
data = JSON.parse(raw);
|
|
29
|
+
} catch {
|
|
30
|
+
process.stderr.write(`extract: ${file} is not valid JSON\n`);
|
|
31
|
+
process.exit(1);
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
const events = Array.isArray(data) ? data : [data];
|
|
35
|
+
|
|
36
|
+
const assistantTexts = () => {
|
|
37
|
+
const out = [];
|
|
38
|
+
for (const ev of events) {
|
|
39
|
+
if (!ev || ev.type !== 'assistant') continue;
|
|
40
|
+
const content = ev.message && ev.message.content;
|
|
41
|
+
if (!Array.isArray(content)) continue;
|
|
42
|
+
const t = content.filter((b) => b && b.type === 'text').map((b) => b.text).join('\n');
|
|
43
|
+
if (t.trim()) out.push(t);
|
|
44
|
+
}
|
|
45
|
+
return out;
|
|
46
|
+
};
|
|
47
|
+
|
|
48
|
+
// Pseudo-field: the first non-empty assistant message. The regime line is stated
|
|
49
|
+
// "before doing anything", so it lives in the first turn, not in the final result.
|
|
50
|
+
// Needs --verbose on the run, without which the capture holds only the result element.
|
|
51
|
+
if (field === 'first_text') {
|
|
52
|
+
const texts = assistantTexts();
|
|
53
|
+
process.stdout.write(texts.length ? texts[0] : '');
|
|
54
|
+
process.exit(0);
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
const result = events.find((e) => e && e.type === 'result');
|
|
58
|
+
if (!result) {
|
|
59
|
+
process.stderr.write(`extract: no element with type "result" in ${file}\n`);
|
|
60
|
+
process.exit(1);
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
let value = result[field];
|
|
64
|
+
|
|
65
|
+
// A run stopped by --max-turns closes with subtype "error_max_turns" and no usable
|
|
66
|
+
// `result`, even though assistant text was produced. Fall back to the text blocks of the
|
|
67
|
+
// last assistant message so the case is scored on what the session actually said.
|
|
68
|
+
if (field === 'result' && (value === undefined || value === null || value === 'undefined')) {
|
|
69
|
+
const texts = assistantTexts();
|
|
70
|
+
value = texts.length ? texts[texts.length - 1] : '';
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
process.stdout.write(value === undefined || value === null ? '' : String(value));
|
|
@@ -0,0 +1,369 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# Behavioural eval harness for this package.
|
|
3
|
+
#
|
|
4
|
+
# run.sh [--all | --case <id>...] [--reps N] [--model <id>] [--dry-run]
|
|
5
|
+
#
|
|
6
|
+
# Installs the package into a throwaway CLAUDE_CONFIG_DIR, runs each case against
|
|
7
|
+
# a throwaway copy of a fixture repository with `claude -p`, and scores the answer
|
|
8
|
+
# with the case's assert.sh. Nothing is written inside the git index of this repo:
|
|
9
|
+
# results go to $LL_EVAL_RESULTS (default ~/.claude/ll-skills-evals)/<YYYY-MM-DD-HHMM>/ and the work
|
|
10
|
+
# trees to a mktemp directory outside the repo, so the session under test never
|
|
11
|
+
# discovers this project's own .claude/ or CLAUDE.md.
|
|
12
|
+
#
|
|
13
|
+
# Exit 0 when every selected case passed in at least min_pass reps (case.json,
|
|
14
|
+
# capped at the number of reps actually run).
|
|
15
|
+
|
|
16
|
+
set -uo pipefail
|
|
17
|
+
|
|
18
|
+
HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
|
19
|
+
REPO="$(cd "$HERE/../.." && pwd)"
|
|
20
|
+
CASES_DIR="$HERE/cases"
|
|
21
|
+
LIB="$HERE/lib"
|
|
22
|
+
SHARED_FIXTURE="$REPO/scripts/fixtures/project"
|
|
23
|
+
GIT_HISTORY="$REPO/scripts/fixtures/git-history.sh"
|
|
24
|
+
PREAMBLE="$REPO/assets/preamble.md"
|
|
25
|
+
INSTALLER="$REPO/bin/install.js"
|
|
26
|
+
|
|
27
|
+
REPS=3
|
|
28
|
+
MODEL=""
|
|
29
|
+
DRY_RUN=0
|
|
30
|
+
SELECTED=()
|
|
31
|
+
|
|
32
|
+
die() { printf 'run.sh: %s\n' "$1" >&2; exit 2; }
|
|
33
|
+
|
|
34
|
+
usage() {
|
|
35
|
+
sed -n '2,16p' "$HERE/run.sh" | sed 's/^# \{0,1\}//'
|
|
36
|
+
exit 0
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
# ---------------------------------------------------------------------------
|
|
40
|
+
# arguments
|
|
41
|
+
# ---------------------------------------------------------------------------
|
|
42
|
+
|
|
43
|
+
while [ $# -gt 0 ]; do
|
|
44
|
+
case "$1" in
|
|
45
|
+
--all) SELECTED=(); shift ;;
|
|
46
|
+
--case) [ $# -ge 2 ] || die "--case needs an id"; SELECTED+=("$2"); shift 2 ;;
|
|
47
|
+
--reps) [ $# -ge 2 ] || die "--reps needs a number"; REPS="$2"; shift 2 ;;
|
|
48
|
+
--model) [ $# -ge 2 ] || die "--model needs an id"; MODEL="$2"; shift 2 ;;
|
|
49
|
+
--dry-run) DRY_RUN=1; shift ;;
|
|
50
|
+
-h|--help) usage ;;
|
|
51
|
+
*) die "unknown argument: $1 (use --help)" ;;
|
|
52
|
+
esac
|
|
53
|
+
done
|
|
54
|
+
|
|
55
|
+
case "$REPS" in (''|*[!0-9]*) die "--reps must be a positive integer" ;; esac
|
|
56
|
+
[ "$REPS" -ge 1 ] || die "--reps must be at least 1"
|
|
57
|
+
|
|
58
|
+
[ -d "$CASES_DIR" ] || die "no cases directory at $CASES_DIR"
|
|
59
|
+
[ -f "$PREAMBLE" ] || die "no preamble at $PREAMBLE"
|
|
60
|
+
[ -f "$INSTALLER" ] || die "no installer at $INSTALLER"
|
|
61
|
+
command -v node >/dev/null 2>&1 || die "node is required"
|
|
62
|
+
command -v claude >/dev/null 2>&1 || [ "$DRY_RUN" -eq 1 ] || die "claude CLI is not on PATH"
|
|
63
|
+
|
|
64
|
+
all_cases() {
|
|
65
|
+
local d
|
|
66
|
+
for d in "$CASES_DIR"/*/; do
|
|
67
|
+
[ -f "${d}case.json" ] || continue
|
|
68
|
+
basename "$d"
|
|
69
|
+
done
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
if [ "${#SELECTED[@]}" -eq 0 ]; then
|
|
73
|
+
mapfile -t SELECTED < <(all_cases)
|
|
74
|
+
fi
|
|
75
|
+
[ "${#SELECTED[@]}" -gt 0 ] || die "no cases selected"
|
|
76
|
+
|
|
77
|
+
for id in "${SELECTED[@]}"; do
|
|
78
|
+
[ -f "$CASES_DIR/$id/case.json" ] || die "unknown case: $id"
|
|
79
|
+
[ -f "$CASES_DIR/$id/prompt.txt" ] || die "case $id has no prompt.txt"
|
|
80
|
+
[ -f "$CASES_DIR/$id/assert.sh" ] || die "case $id has no assert.sh"
|
|
81
|
+
done
|
|
82
|
+
|
|
83
|
+
# A case with "reuse" runs no claude call: it re-scores the work dir and the
|
|
84
|
+
# capture of the case it names. Order the selection so the source runs first.
|
|
85
|
+
reuse_of() { node -e '
|
|
86
|
+
const c = require(process.argv[1]);
|
|
87
|
+
process.stdout.write(c.reuse || "");
|
|
88
|
+
' "$CASES_DIR/$1/case.json"; }
|
|
89
|
+
|
|
90
|
+
field() { # field <case-id> <name> <default>
|
|
91
|
+
node -e '
|
|
92
|
+
const c = require(process.argv[1]);
|
|
93
|
+
const v = c[process.argv[2]];
|
|
94
|
+
process.stdout.write(v === undefined || v === null ? process.argv[3] : String(v));
|
|
95
|
+
' "$CASES_DIR/$1/case.json" "$2" "$3"
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
ORDERED=()
|
|
99
|
+
for id in "${SELECTED[@]}"; do [ -n "$(reuse_of "$id")" ] || ORDERED+=("$id"); done
|
|
100
|
+
for id in "${SELECTED[@]}"; do [ -z "$(reuse_of "$id")" ] || ORDERED+=("$id"); done
|
|
101
|
+
|
|
102
|
+
# ---------------------------------------------------------------------------
|
|
103
|
+
# results directory (outside the git index) and work root (outside the repo)
|
|
104
|
+
# ---------------------------------------------------------------------------
|
|
105
|
+
|
|
106
|
+
STAMP="$(date +%Y-%m-%d-%H%M)"
|
|
107
|
+
RESULTS="${LL_EVAL_RESULTS:-$HOME/.claude/ll-skills-evals}/$STAMP"
|
|
108
|
+
mkdir -p "$RESULTS" || die "cannot create $RESULTS"
|
|
109
|
+
|
|
110
|
+
TMP="$(mktemp -d "${TMPDIR:-/tmp}/ll-evals.XXXXXXXX")" || die "cannot create a work root"
|
|
111
|
+
CONFIG="$TMP/config"
|
|
112
|
+
|
|
113
|
+
printf 'results : %s\n' "$RESULTS"
|
|
114
|
+
printf 'work : %s\n' "$TMP"
|
|
115
|
+
printf 'cases : %s (reps %s)\n' "${#ORDERED[@]}" "$REPS"
|
|
116
|
+
printf '\n'
|
|
117
|
+
|
|
118
|
+
# ---------------------------------------------------------------------------
|
|
119
|
+
# install the package into the throwaway config dir
|
|
120
|
+
# ---------------------------------------------------------------------------
|
|
121
|
+
|
|
122
|
+
install_cmd() {
|
|
123
|
+
printf 'CLAUDE_CONFIG_DIR=%s node %s --yes --no-settings' "$CONFIG" "$INSTALLER"
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
if [ "$DRY_RUN" -eq 1 ]; then
|
|
127
|
+
printf '# install\n%s\n\n' "$(install_cmd)"
|
|
128
|
+
else
|
|
129
|
+
mkdir -p "$CONFIG"
|
|
130
|
+
# The credentials live in the real config dir; a fresh CLAUDE_CONFIG_DIR has no auth
|
|
131
|
+
# of its own. A symlink, not a copy: a copy is a snapshot that goes stale as soon as
|
|
132
|
+
# the OAuth token is refreshed, and a long run then dies with "session expired".
|
|
133
|
+
# ANTHROPIC_API_KEY, when set, covers the same ground.
|
|
134
|
+
if [ -f "$HOME/.claude/.credentials.json" ] && [ ! -e "$CONFIG/.credentials.json" ]; then
|
|
135
|
+
ln -s "$HOME/.claude/.credentials.json" "$CONFIG/.credentials.json"
|
|
136
|
+
fi
|
|
137
|
+
if ! CLAUDE_CONFIG_DIR="$CONFIG" node "$INSTALLER" --yes --no-settings > "$RESULTS/install.log" 2>&1; then
|
|
138
|
+
cat "$RESULTS/install.log" >&2
|
|
139
|
+
die "the installer failed; see $RESULTS/install.log"
|
|
140
|
+
fi
|
|
141
|
+
printf 'installed into %s (%s skills)\n\n' "$CONFIG" "$(ls -1 "$CONFIG/skills" 2>/dev/null | wc -l)"
|
|
142
|
+
fi
|
|
143
|
+
|
|
144
|
+
# ---------------------------------------------------------------------------
|
|
145
|
+
# one rep
|
|
146
|
+
# ---------------------------------------------------------------------------
|
|
147
|
+
|
|
148
|
+
prepare_workdir() { # prepare_workdir <case-id> <workdir> <history>
|
|
149
|
+
local id="$1" work="$2" history="$3" fixture="$CASES_DIR/$1/fixture"
|
|
150
|
+
|
|
151
|
+
if [ "$history" = "true" ]; then
|
|
152
|
+
# git-history.sh rebuilds <target> from the shared fixture and commits five times.
|
|
153
|
+
bash "$GIT_HISTORY" "$work" >/dev/null 2>&1 || return 1
|
|
154
|
+
else
|
|
155
|
+
[ -d "$fixture" ] || fixture="$SHARED_FIXTURE"
|
|
156
|
+
rm -rf "$work"; mkdir -p "$work"
|
|
157
|
+
cp -R "$fixture/." "$work/" 2>/dev/null || true
|
|
158
|
+
git -C "$work" init -q -b main
|
|
159
|
+
git -C "$work" add -A
|
|
160
|
+
git -C "$work" -c user.name=eval -c user.email=eval@example.com \
|
|
161
|
+
-c commit.gpgsign=false commit -q -m "chore: eval fixture" --allow-empty
|
|
162
|
+
fi
|
|
163
|
+
|
|
164
|
+
if [ -f "$CASES_DIR/$id/setup.sh" ]; then
|
|
165
|
+
bash "$CASES_DIR/$id/setup.sh" "$work" > "$work/.eval-setup.log" 2>&1 || return 1
|
|
166
|
+
fi
|
|
167
|
+
|
|
168
|
+
# The sha the run starts from, so an assert can look only at what the run added.
|
|
169
|
+
git -C "$work" rev-parse HEAD > "$work/.eval-base-sha" 2>/dev/null || echo "" > "$work/.eval-base-sha"
|
|
170
|
+
printf '.eval-base-sha\n.eval-setup.log\n' >> "$work/.git/info/exclude"
|
|
171
|
+
return 0
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
# prompt.txt may carry {{WORK}}, replaced with the absolute path of the work tree —
|
|
175
|
+
# a brief passes absolute paths and no `cd`, and the work tree is created per rep.
|
|
176
|
+
prompt_text() { # prompt_text <case-id> <workdir>
|
|
177
|
+
sed "s|{{WORK}}|$2|g" "$CASES_DIR/$1/prompt.txt"
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
claude_cmd() { # claude_cmd <case-id> <workdir> <max_turns> <agent> <permission_mode>
|
|
181
|
+
local id="$1" work="$2" turns="$3" agent="$4" perm="$5"
|
|
182
|
+
local cmd="env -u CLAUDECODE CLAUDE_CONFIG_DIR=$CONFIG claude -p \"\$(sed 's|{{WORK}}|$work|g' $CASES_DIR/$id/prompt.txt)\""
|
|
183
|
+
cmd="$cmd --max-turns $turns --output-format json"
|
|
184
|
+
cmd="$cmd --append-system-prompt \"\$(cat $PREAMBLE)\""
|
|
185
|
+
cmd="$cmd --permission-mode $perm --strict-mcp-config --verbose"
|
|
186
|
+
[ -n "$agent" ] && cmd="$cmd --agent $agent"
|
|
187
|
+
[ -n "$MODEL" ] && cmd="$cmd --model $MODEL"
|
|
188
|
+
printf '(cd %s && %s > %s/%s/rep1/out.json)' "$work" "$cmd" "$RESULTS" "$id"
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
run_claude() { # run_claude <case-id> <workdir> <turns> <agent> <perm> <out.json>
|
|
192
|
+
local id="$1" work="$2" turns="$3" agent="$4" perm="$5" outjson="$6"
|
|
193
|
+
local args=(-p "$(prompt_text "$id" "$work")"
|
|
194
|
+
--max-turns "$turns"
|
|
195
|
+
--output-format json
|
|
196
|
+
--append-system-prompt "$(cat "$PREAMBLE")"
|
|
197
|
+
--permission-mode "$perm"
|
|
198
|
+
--strict-mcp-config
|
|
199
|
+
--verbose)
|
|
200
|
+
[ -n "$agent" ] && args+=(--agent "$agent")
|
|
201
|
+
[ -n "$MODEL" ] && args+=(--model "$MODEL")
|
|
202
|
+
( cd "$work" && env -u CLAUDECODE CLAUDE_CONFIG_DIR="$CONFIG" claude "${args[@]}" ) \
|
|
203
|
+
> "$outjson" 2>"${outjson%.json}.stderr"
|
|
204
|
+
}
|
|
205
|
+
|
|
206
|
+
# ---------------------------------------------------------------------------
|
|
207
|
+
# the loop
|
|
208
|
+
# ---------------------------------------------------------------------------
|
|
209
|
+
|
|
210
|
+
ROWS=() # "case|rep|PASS/FAIL|cost|duration|turns|first failure"
|
|
211
|
+
declare -A PASSES=() MINPASS=()
|
|
212
|
+
TOTAL_COST=0
|
|
213
|
+
|
|
214
|
+
for id in "${ORDERED[@]}"; do
|
|
215
|
+
MAX_TURNS="$(field "$id" max_turns 20)"
|
|
216
|
+
HISTORY="$(field "$id" history false)"
|
|
217
|
+
MIN_PASS="$(field "$id" min_pass 2)"
|
|
218
|
+
AGENT="$(field "$id" agent '')"
|
|
219
|
+
PERM="$(field "$id" permission_mode bypassPermissions)"
|
|
220
|
+
REUSE="$(reuse_of "$id")"
|
|
221
|
+
[ "$AGENT" = "null" ] && AGENT=""
|
|
222
|
+
MINPASS["$id"]="$MIN_PASS"
|
|
223
|
+
PASSES["$id"]=0
|
|
224
|
+
|
|
225
|
+
if [ "$DRY_RUN" -eq 1 ]; then
|
|
226
|
+
work="$TMP/work-$id-1"
|
|
227
|
+
printf '# case %s (max_turns %s · history %s · min_pass %s%s)\n' \
|
|
228
|
+
"$id" "$MAX_TURNS" "$HISTORY" "$MIN_PASS" "${AGENT:+ · agent $AGENT}"
|
|
229
|
+
if [ -n "$REUSE" ]; then
|
|
230
|
+
printf '# no claude call: re-scores the work dir and capture of %s\n' "$REUSE"
|
|
231
|
+
printf 'bash %s/assert.sh %s %s %s\n\n' \
|
|
232
|
+
"$CASES_DIR/$id" "$TMP/work-$REUSE-1" "$RESULTS/$REUSE/rep1/out.json" "$RESULTS/$REUSE/rep1/out.txt"
|
|
233
|
+
continue
|
|
234
|
+
fi
|
|
235
|
+
if [ "$HISTORY" = "true" ]; then
|
|
236
|
+
printf 'bash %s %s\n' "$GIT_HISTORY" "$work"
|
|
237
|
+
else
|
|
238
|
+
src="$CASES_DIR/$id/fixture"; [ -d "$src" ] || src="$SHARED_FIXTURE"
|
|
239
|
+
printf 'cp -R %s/. %s/ && git -C %s init -q -b main && git -C %s commit -m "chore: eval fixture"\n' \
|
|
240
|
+
"$src" "$work" "$work" "$work"
|
|
241
|
+
fi
|
|
242
|
+
[ -f "$CASES_DIR/$id/setup.sh" ] && printf 'bash %s/setup.sh %s\n' "$CASES_DIR/$id" "$work"
|
|
243
|
+
rep1="$RESULTS/$id/rep1"
|
|
244
|
+
printf '%s\n' "$(claude_cmd "$id" "$work" "$MAX_TURNS" "$AGENT" "$PERM")"
|
|
245
|
+
printf 'node %s/extract.js %s/out.json result > %s/out.txt\n' "$LIB" "$rep1" "$rep1"
|
|
246
|
+
printf 'bash %s/assert.sh %s %s/out.json %s/out.txt\n\n' "$CASES_DIR/$id" "$work" "$rep1" "$rep1"
|
|
247
|
+
continue
|
|
248
|
+
fi
|
|
249
|
+
|
|
250
|
+
for rep in $(seq 1 "$REPS"); do
|
|
251
|
+
repdir="$RESULTS/$id/rep$rep"
|
|
252
|
+
mkdir -p "$repdir"
|
|
253
|
+
outjson="$repdir/out.json"
|
|
254
|
+
outtxt="$repdir/out.txt"
|
|
255
|
+
cost="-" ; dur="-" ; turns="-" ; firstfail="" ; verdict="FAIL"
|
|
256
|
+
|
|
257
|
+
if [ -n "$REUSE" ]; then
|
|
258
|
+
work="$TMP/work-$REUSE-$rep"
|
|
259
|
+
src="$RESULTS/$REUSE/rep$rep"
|
|
260
|
+
if [ ! -d "$work" ] || [ ! -f "$src/out.json" ]; then
|
|
261
|
+
firstfail="source case $REUSE was not run in this invocation"
|
|
262
|
+
ROWS+=("$id|$rep|FAIL|-|-|-|$firstfail")
|
|
263
|
+
printf ' %-26s rep %s FAIL (%s)\n' "$id" "$rep" "$firstfail"
|
|
264
|
+
continue
|
|
265
|
+
fi
|
|
266
|
+
cp "$src/out.json" "$outjson"; cp "$src/out.txt" "$outtxt"
|
|
267
|
+
else
|
|
268
|
+
work="$TMP/work-$id-$rep"
|
|
269
|
+
if ! prepare_workdir "$id" "$work" "$HISTORY"; then
|
|
270
|
+
firstfail="fixture preparation failed"
|
|
271
|
+
ROWS+=("$id|$rep|FAIL|-|-|-|$firstfail")
|
|
272
|
+
printf ' %-26s rep %s FAIL (%s)\n' "$id" "$rep" "$firstfail"
|
|
273
|
+
continue
|
|
274
|
+
fi
|
|
275
|
+
run_claude "$id" "$work" "$MAX_TURNS" "$AGENT" "$PERM" "$outjson"
|
|
276
|
+
if ! node "$LIB/extract.js" "$outjson" result > "$outtxt" 2>"$repdir/extract.err"; then
|
|
277
|
+
firstfail="no result element in out.json ($(head -c 120 "$repdir/extract.err" | tr '\n' ' '))"
|
|
278
|
+
ROWS+=("$id|$rep|FAIL|-|-|-|$firstfail")
|
|
279
|
+
printf ' %-26s rep %s FAIL (%s)\n' "$id" "$rep" "$firstfail"
|
|
280
|
+
continue
|
|
281
|
+
fi
|
|
282
|
+
fi
|
|
283
|
+
|
|
284
|
+
cost="$(node "$LIB/extract.js" "$outjson" total_cost_usd 2>/dev/null)"; [ -n "$cost" ] || cost="-"
|
|
285
|
+
ms="$(node "$LIB/extract.js" "$outjson" duration_ms 2>/dev/null)"
|
|
286
|
+
turns="$(node "$LIB/extract.js" "$outjson" num_turns 2>/dev/null)"; [ -n "$turns" ] || turns="-"
|
|
287
|
+
if [ -n "$ms" ]; then dur="$(node -e 'process.stdout.write((Number(process.argv[1])/1000).toFixed(1))' "$ms")"; fi
|
|
288
|
+
if [ "$cost" != "-" ]; then
|
|
289
|
+
TOTAL_COST="$(node -e 'process.stdout.write((Number(process.argv[1])+Number(process.argv[2])).toFixed(4))' "$TOTAL_COST" "$cost")"
|
|
290
|
+
cost="$(node -e 'process.stdout.write(Number(process.argv[1]).toFixed(4))' "$cost")"
|
|
291
|
+
fi
|
|
292
|
+
|
|
293
|
+
bash "$CASES_DIR/$id/assert.sh" "$work" "$outjson" "$outtxt" > "$repdir/assert.log" 2>&1
|
|
294
|
+
arc=$?
|
|
295
|
+
if [ "$arc" -eq 0 ]; then
|
|
296
|
+
verdict="PASS"
|
|
297
|
+
PASSES["$id"]=$(( ${PASSES["$id"]} + 1 ))
|
|
298
|
+
else
|
|
299
|
+
firstfail="$(grep -m1 '^FAIL: ' "$repdir/assert.log" | sed 's/^FAIL: //')"
|
|
300
|
+
[ -n "$firstfail" ] || firstfail="assert.sh exited non-zero with no FAIL line"
|
|
301
|
+
# A run cut short by the turn cap is a budget problem, not a behavioural one: say so.
|
|
302
|
+
sub="$(node "$LIB/extract.js" "$outjson" subtype 2>/dev/null)"
|
|
303
|
+
[ "$sub" = "success" ] || [ -z "$sub" ] || firstfail="[$sub] $firstfail"
|
|
304
|
+
fi
|
|
305
|
+
|
|
306
|
+
ROWS+=("$id|$rep|$verdict|$cost|$dur|$turns|$firstfail")
|
|
307
|
+
printf ' %-26s rep %s %s cost %s %ss turns %s%s\n' \
|
|
308
|
+
"$id" "$rep" "$verdict" "$cost" "$dur" "$turns" "${firstfail:+ — $firstfail}"
|
|
309
|
+
done
|
|
310
|
+
done
|
|
311
|
+
|
|
312
|
+
if [ "$DRY_RUN" -eq 1 ]; then
|
|
313
|
+
printf '# dry run: %s case blocks printed, no claude call made\n' "${#ORDERED[@]}"
|
|
314
|
+
exit 0
|
|
315
|
+
fi
|
|
316
|
+
|
|
317
|
+
# ---------------------------------------------------------------------------
|
|
318
|
+
# RESULTS.md and summary.json
|
|
319
|
+
# ---------------------------------------------------------------------------
|
|
320
|
+
|
|
321
|
+
EXIT=0
|
|
322
|
+
{
|
|
323
|
+
printf '# Eval results — %s\n\n' "$STAMP"
|
|
324
|
+
printf 'model: %s · reps: %s · cases: %s · total cost: USD %s\n\n' \
|
|
325
|
+
"${MODEL:-default}" "$REPS" "${#ORDERED[@]}" "$TOTAL_COST"
|
|
326
|
+
printf '| case | rep | verdict | cost USD | duration s | turns | first assert failure |\n'
|
|
327
|
+
printf '|---|---|---|---|---|---|---|\n'
|
|
328
|
+
for row in "${ROWS[@]}"; do
|
|
329
|
+
IFS='|' read -r c r v co du tu ff <<< "$row"
|
|
330
|
+
printf '| %s | %s | %s | %s | %s | %s | %s |\n' "$c" "$r" "$v" "$co" "$du" "$tu" "${ff:-—}"
|
|
331
|
+
done
|
|
332
|
+
printf '\n## Case verdicts\n\n'
|
|
333
|
+
printf '| case | passed | of reps | min_pass | verdict |\n|---|---|---|---|---|\n'
|
|
334
|
+
for id in "${ORDERED[@]}"; do
|
|
335
|
+
need="${MINPASS[$id]}"
|
|
336
|
+
[ "$need" -le "$REPS" ] || need="$REPS"
|
|
337
|
+
got="${PASSES[$id]}"
|
|
338
|
+
if [ "$got" -ge "$need" ]; then v=PASS; else v=FAIL; EXIT=1; fi
|
|
339
|
+
printf '| %s | %s | %s | %s | %s |\n' "$id" "$got" "$REPS" "$need" "$v"
|
|
340
|
+
done
|
|
341
|
+
printf '\nWork trees kept at `%s`.\n' "$TMP"
|
|
342
|
+
} > "$RESULTS/RESULTS.md"
|
|
343
|
+
|
|
344
|
+
node - "$RESULTS/summary.json" "$STAMP" "${MODEL:-default}" "$REPS" "$TMP" "$RESULTS" "$TOTAL_COST" "$EXIT" <<'NODE' "${ROWS[@]}"
|
|
345
|
+
const fs = require('fs');
|
|
346
|
+
const [out, stamp, model, reps, tmp, results, cost, exitCode, ...rows] = process.argv.slice(2);
|
|
347
|
+
const byCase = {};
|
|
348
|
+
for (const row of rows) {
|
|
349
|
+
const [id, rep, verdict, c, d, t, ff] = row.split('|');
|
|
350
|
+
(byCase[id] = byCase[id] || []).push({
|
|
351
|
+
rep: Number(rep), pass: verdict === 'PASS',
|
|
352
|
+
cost_usd: c === '-' ? null : Number(c),
|
|
353
|
+
duration_s: d === '-' ? null : Number(d),
|
|
354
|
+
turns: t === '-' ? null : Number(t),
|
|
355
|
+
first_failure: ff || null,
|
|
356
|
+
});
|
|
357
|
+
}
|
|
358
|
+
fs.writeFileSync(out, JSON.stringify({
|
|
359
|
+
stamp, model, reps: Number(reps), results_dir: results, work_dir: tmp,
|
|
360
|
+
total_cost_usd: Number(cost), exit_code: Number(exitCode),
|
|
361
|
+
cases: Object.entries(byCase).map(([id, r]) => ({
|
|
362
|
+
id, passed: r.filter((x) => x.pass).length, reps: r,
|
|
363
|
+
})),
|
|
364
|
+
}, null, 2) + '\n');
|
|
365
|
+
NODE
|
|
366
|
+
|
|
367
|
+
printf '\n%s\n' "$RESULTS/RESULTS.md"
|
|
368
|
+
printf 'total cost USD %s · exit %s\n' "$TOTAL_COST" "$EXIT"
|
|
369
|
+
exit "$EXIT"
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
# PROGRESS — fixture delivered
|
|
2
|
+
|
|
3
|
+
<!-- ll-state -->
|
|
4
|
+
phase: 02
|
|
5
|
+
milestones:
|
|
6
|
+
M1: { passes: true, commit: a1b2c3d, accepted_at: 2026-09-08T10:00:00Z }
|
|
7
|
+
M2: { passes: true, commit: b2c3d4e, accepted_at: 2026-09-08T15:30:00Z }
|
|
8
|
+
<!-- /ll-state -->
|
|
9
|
+
|
|
10
|
+
## Epilogue — phase 01 — 2026-09-07
|
|
11
|
+
|
|
12
|
+
passed: M1 · left: none
|
|
13
|
+
milestones passed 1/1 · questions asked 0 / assumptions 0 / band-1 open 0 · amendments 0 · verification: none
|
|
14
|
+
▶ Next — `/clear`, then `ll-implement 2`
|
|
15
|
+
|
|
16
|
+
## Epilogue — phase 02 — 2026-09-08
|
|
17
|
+
|
|
18
|
+
passed: M1, M2 · left: none
|
|
19
|
+
milestones passed 2/2 · questions asked 0 / assumptions 0 / band-1 open 0 · amendments 0 · verification: none
|
|
20
|
+
▶ Next — `/clear`, then `ll-close`
|