ll-skills 2.0.2 → 3.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (115) hide show
  1. package/CHANGELOG.md +21 -0
  2. package/README.md +42 -20
  3. package/agents/ll-executor.md +1 -0
  4. package/assets/preamble.md +29 -35
  5. package/bin/install.js +4 -1
  6. package/hooks/ll-precompact.js +29 -1
  7. package/hooks/ll-skills-check-update.js +6 -6
  8. package/hooks/ll-state.js +30 -2
  9. package/package.json +3 -2
  10. package/scripts/evals/README.md +57 -0
  11. package/scripts/evals/cases/auto-dry-run/assert.sh +35 -0
  12. package/scripts/evals/cases/auto-dry-run/case.json +8 -0
  13. package/scripts/evals/cases/auto-dry-run/prompt.txt +1 -0
  14. package/scripts/evals/cases/auto-empty-repo/assert.sh +25 -0
  15. package/scripts/evals/cases/auto-empty-repo/case.json +8 -0
  16. package/scripts/evals/cases/auto-empty-repo/fixture/.gitkeep +0 -0
  17. package/scripts/evals/cases/auto-empty-repo/prompt.txt +1 -0
  18. package/scripts/evals/cases/decide-final-round/assert.sh +32 -0
  19. package/scripts/evals/cases/decide-final-round/case.json +8 -0
  20. package/scripts/evals/cases/decide-final-round/fixture/README.md +3 -0
  21. package/scripts/evals/cases/decide-final-round/prompt.txt +1 -0
  22. package/scripts/evals/cases/executor-block/assert.sh +33 -0
  23. package/scripts/evals/cases/executor-block/case.json +8 -0
  24. package/scripts/evals/cases/executor-block/prompt.txt +14 -0
  25. package/scripts/evals/cases/goal-autonomous/assert.sh +35 -0
  26. package/scripts/evals/cases/goal-autonomous/case.json +8 -0
  27. package/scripts/evals/cases/goal-autonomous/fixture/PLAN.md +42 -0
  28. package/scripts/evals/cases/goal-autonomous/fixture/PROGRESS.md +20 -0
  29. package/scripts/evals/cases/goal-autonomous/fixture/ROADMAP.md +29 -0
  30. package/scripts/evals/cases/goal-autonomous/fixture/package.json +8 -0
  31. package/scripts/evals/cases/goal-autonomous/fixture/src/money.js +6 -0
  32. package/scripts/evals/cases/goal-autonomous/fixture/test/reconcile.test.js +8 -0
  33. package/scripts/evals/cases/goal-autonomous/prompt.txt +1 -0
  34. package/scripts/evals/cases/implement-review-gate/assert.sh +35 -0
  35. package/scripts/evals/cases/implement-review-gate/case.json +8 -0
  36. package/scripts/evals/cases/implement-review-gate/prompt.txt +1 -0
  37. package/scripts/evals/cases/implement-stops-at-next/assert.sh +39 -0
  38. package/scripts/evals/cases/implement-stops-at-next/case.json +9 -0
  39. package/scripts/evals/cases/implement-stops-at-next/prompt.txt +1 -0
  40. package/scripts/evals/cases/preamble-no-ritual/assert.sh +17 -0
  41. package/scripts/evals/cases/preamble-no-ritual/case.json +8 -0
  42. package/scripts/evals/cases/preamble-no-ritual/fixture/README.md +3 -0
  43. package/scripts/evals/cases/preamble-no-ritual/fixture/src/a.ts +3 -0
  44. package/scripts/evals/cases/preamble-no-ritual/prompt.txt +1 -0
  45. package/scripts/evals/cases/router-execute/assert.sh +12 -0
  46. package/scripts/evals/cases/router-execute/case.json +8 -0
  47. package/scripts/evals/cases/router-execute/prompt.txt +1 -0
  48. package/scripts/evals/cases/router-research/assert.sh +11 -0
  49. package/scripts/evals/cases/router-research/case.json +8 -0
  50. package/scripts/evals/cases/router-research/fixture/README.md +3 -0
  51. package/scripts/evals/cases/router-research/prompt.txt +1 -0
  52. package/scripts/evals/cases/router-small/assert.sh +21 -0
  53. package/scripts/evals/cases/router-small/case.json +8 -0
  54. package/scripts/evals/cases/router-small/fixture/README.md +17 -0
  55. package/scripts/evals/cases/router-small/prompt.txt +1 -0
  56. package/scripts/evals/cases/scout-no-plan/assert.sh +41 -0
  57. package/scripts/evals/cases/scout-no-plan/case.json +8 -0
  58. package/scripts/evals/cases/scout-no-plan/prompt.txt +8 -0
  59. package/scripts/evals/cases/verifier-weakened-test/assert.sh +19 -0
  60. package/scripts/evals/cases/verifier-weakened-test/case.json +8 -0
  61. package/scripts/evals/cases/verifier-weakened-test/prompt.txt +13 -0
  62. package/scripts/evals/cases/verifier-weakened-test/setup.sh +19 -0
  63. package/scripts/evals/fixtures/manual-contract/out.json +29 -0
  64. package/scripts/evals/fixtures/manual-contract/out.txt +5 -0
  65. package/scripts/evals/fixtures/manual-contract/with-skill.json +46 -0
  66. package/scripts/evals/lib/assert.sh +107 -0
  67. package/scripts/evals/lib/extract.js +73 -0
  68. package/scripts/evals/run.sh +369 -0
  69. package/scripts/fixtures/auto-closed/PLAN.md +5 -0
  70. package/scripts/fixtures/auto-closed/PROGRESS.md +20 -0
  71. package/scripts/fixtures/auto-closed/ROADMAP.md +6 -0
  72. package/scripts/fixtures/auto-closed/docs/DELIVERY.md +3 -0
  73. package/scripts/fixtures/auto-decisions/decisions/DEC-0001-taken-alone.md +13 -0
  74. package/scripts/fixtures/auto-decisions/decisions/DEC-0002-owner.md +13 -0
  75. package/scripts/fixtures/auto-noroadmap/PLAN.md +20 -0
  76. package/scripts/fixtures/auto-noroadmap/PROGRESS.md +11 -0
  77. package/scripts/fixtures/auto-verify-next/PLAN.md +5 -0
  78. package/scripts/fixtures/auto-verify-next/PROGRESS.md +18 -0
  79. package/scripts/fixtures/auto-verify-next/ROADMAP.md +5 -0
  80. package/scripts/fixtures/auto-verify-next/phases/01/PLAN.md +6 -0
  81. package/scripts/fixtures/evals-auto/auto-dry-run/pass.txt +18 -0
  82. package/scripts/fixtures/evals-auto/auto-empty-repo/pass.txt +2 -0
  83. package/scripts/fixtures/evals-auto/goal-autonomous/pass.txt +29 -0
  84. package/scripts/fixtures/lint-bad/folded-description/SKILL.md +13 -0
  85. package/scripts/fixtures/lint-bad/model-invocation-false/SKILL.md +10 -0
  86. package/scripts/fixtures/next-bad/skills/ll-bad/SKILL.md +30 -0
  87. package/scripts/fixtures/next-good/skills/ll-good/SKILL.md +26 -0
  88. package/scripts/fixtures/project/PROGRESS.md +4 -0
  89. package/scripts/lint-contract.cjs +495 -0
  90. package/scripts/lint-prompts.sh +396 -0
  91. package/scripts/ll-tools.js +465 -447
  92. package/scripts/smoke-test.sh +380 -1
  93. package/skills/ll-auto/SKILL.md +74 -0
  94. package/skills/ll-auto/references/run.md +75 -0
  95. package/skills/ll-auto/references/stages.md +66 -0
  96. package/skills/ll-auto/scripts/ll-auto.js +345 -0
  97. package/skills/ll-brainstorm/SKILL.md +5 -4
  98. package/skills/ll-brainstorm/references/decision-policy.md +3 -0
  99. package/skills/ll-close/SKILL.md +5 -5
  100. package/skills/ll-close/references/delivery.md +3 -1
  101. package/skills/ll-decide/SKILL.md +13 -11
  102. package/skills/ll-decide/references/decision-policy.md +3 -0
  103. package/skills/ll-decide/references/interview.md +10 -0
  104. package/skills/ll-decide/references/plan-skeleton.md +14 -14
  105. package/skills/ll-decide/references/premise-gate.md +7 -0
  106. package/skills/ll-goal/SKILL.md +22 -4
  107. package/skills/ll-goal/references/goal-template.md +57 -0
  108. package/skills/ll-implement/SKILL.md +4 -2
  109. package/skills/ll-implement/references/decision-policy.md +3 -0
  110. package/skills/ll-oncall/SKILL.md +3 -2
  111. package/skills/ll-refine/SKILL.md +3 -2
  112. package/skills/ll-research/SKILL.md +3 -2
  113. package/skills/ll-resume/SKILL.md +4 -3
  114. package/skills/ll-update/SKILL.md +6 -1
  115. package/skills/ll-verify/SKILL.md +2 -1
@@ -0,0 +1,29 @@
1
+ [
2
+ {
3
+ "type": "system",
4
+ "subtype": "init",
5
+ "session_id": "manual-contract-fixture",
6
+ "model": "claude-sonnet-5"
7
+ },
8
+ {
9
+ "type": "assistant",
10
+ "message": {
11
+ "role": "assistant",
12
+ "content": [
13
+ {
14
+ "type": "text",
15
+ "text": "Comparing authentication providers for the signup flow is a research task, not something I can answer from what is already in the repo. Paste this to start it:\n\n/ll-research compare authentication providers for the signup flow\n\n▶ Next — `/clear` then `/ll-research compare authentication providers for the signup flow`"
16
+ }
17
+ ]
18
+ }
19
+ },
20
+ {
21
+ "type": "result",
22
+ "subtype": "success",
23
+ "is_error": false,
24
+ "duration_ms": 1820,
25
+ "num_turns": 1,
26
+ "total_cost_usd": 0.0091,
27
+ "result": "Comparing authentication providers for the signup flow is a research task, not something I can answer from what is already in the repo. Paste this to start it:\n\n/ll-research compare authentication providers for the signup flow\n\n▶ Next — `/clear` then `/ll-research compare authentication providers for the signup flow`"
28
+ }
29
+ ]
@@ -0,0 +1,5 @@
1
+ Comparing authentication providers for the signup flow is a research task, not something I can answer from what is already in the repo. Paste this to start it:
2
+
3
+ /ll-research compare authentication providers for the signup flow
4
+
5
+ ▶ Next — `/clear` then `/ll-research compare authentication providers for the signup flow`
@@ -0,0 +1,46 @@
1
+ [
2
+ {
3
+ "type": "system",
4
+ "subtype": "init",
5
+ "session_id": "manual-contract-fixture-with-skill",
6
+ "model": "claude-sonnet-5"
7
+ },
8
+ {
9
+ "type": "assistant",
10
+ "message": {
11
+ "role": "assistant",
12
+ "content": [
13
+ {
14
+ "type": "text",
15
+ "text": "Comparing authentication providers for the signup flow is a research task, not something I can answer from what is already in the repo. Paste this to start it:\n\n/ll-research compare authentication providers for the signup flow\n\n▶ Next — `/clear` then `/ll-research compare authentication providers for the signup flow`"
16
+ }
17
+ ]
18
+ }
19
+ },
20
+ {
21
+ "type": "assistant",
22
+ "message": {
23
+ "role": "assistant",
24
+ "content": [
25
+ {
26
+ "type": "tool_use",
27
+ "id": "toolu_01manualcontract",
28
+ "name": "Skill",
29
+ "input": {
30
+ "skill": "ll-research",
31
+ "args": "compare authentication providers for the signup flow"
32
+ }
33
+ }
34
+ ]
35
+ }
36
+ },
37
+ {
38
+ "type": "result",
39
+ "subtype": "success",
40
+ "is_error": false,
41
+ "duration_ms": 2140,
42
+ "num_turns": 2,
43
+ "total_cost_usd": 0.0134,
44
+ "result": "Comparing authentication providers for the signup flow is a research task, not something I can answer from what is already in the repo. Paste this to start it:\n\n/ll-research compare authentication providers for the signup flow\n\n▶ Next — `/clear` then `/ll-research compare authentication providers for the signup flow`"
45
+ }
46
+ ]
@@ -0,0 +1,107 @@
1
+ #!/usr/bin/env bash
2
+ # Sourced by every cases/<id>/assert.sh.
3
+ #
4
+ # Contract of an assert script:
5
+ # bash assert.sh <workdir> <out.json> <out.txt> -> exit 0 pass, 1 fail
6
+ # It prints one line per check: "ok: <what>" or "FAIL: <what>".
7
+ # run.sh reads the first FAIL line into the results table.
8
+ #
9
+ # Provided helpers:
10
+ # ok <msg> record a passed check
11
+ # fail <msg> record a failed check
12
+ # check <cond-exit> <msg> record from an exit code already computed
13
+ # contains <file> <regex> <msg> grep -Eq
14
+ # absent <file> <regex> <msg> grep -Eq must not match
15
+ # no_path <path> <msg> path must not exist
16
+ # no_tool_use <out.json> <tool-name> <msg> no tool_use block named <tool-name> anywhere
17
+ # finish exit with the verdict
18
+
19
+ EVAL_FAILURES=0
20
+
21
+ ok() { printf 'ok: %s\n' "$1"; }
22
+ fail() { printf 'FAIL: %s\n' "$1"; EVAL_FAILURES=$((EVAL_FAILURES + 1)); }
23
+
24
+ check() { # check <exit-code> <msg>
25
+ if [ "$1" -eq 0 ]; then ok "$2"; else fail "$2"; fi
26
+ }
27
+
28
+ contains() { # contains <file> <extended-regex> <msg>
29
+ if [ -f "$1" ] && grep -Eq -- "$2" "$1"; then ok "$3"; else fail "$3"; fi
30
+ }
31
+
32
+ absent() { # absent <file> <extended-regex> <msg>
33
+ if [ ! -f "$1" ] || ! grep -Eq -- "$2" "$1"; then ok "$3"; else fail "$3"; fi
34
+ }
35
+
36
+ no_path() { # no_path <path> <msg>
37
+ if [ ! -e "$1" ]; then ok "$2"; else fail "$2"; fi
38
+ }
39
+
40
+ first_line() { # first_line <file> -> first non-empty line
41
+ [ -f "$1" ] || return 0
42
+ grep -m1 -v '^[[:space:]]*$' "$1" 2>/dev/null || true
43
+ }
44
+
45
+ # The regime is stated "before doing anything": it belongs to the first assistant
46
+ # message of the run, not to the final answer. This reads it out of the capture.
47
+ first_text() { # first_text <out.json> -> the whole first assistant message
48
+ node "$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)/extract.js" "$1" first_text 2>/dev/null || true
49
+ }
50
+
51
+ first_text_line() { # first_text_line <out.json> -> first non-empty line of the first assistant text
52
+ first_text "$1" | grep -m1 -v '^[[:space:]]*$' || true
53
+ }
54
+
55
+ first_text_contains() { # first_text_contains <out.json> <extended-regex> <msg>
56
+ if first_text "$1" | grep -Eq -- "$2"; then ok "$3"; else fail "$3"; fi
57
+ }
58
+
59
+ # no_tool_use <out.json> <tool-name> <msg>
60
+ # Scans every assistant event's message.content for a tool_use block whose
61
+ # `name` equals <tool-name>. Passes when none is found; fails naming the
62
+ # first hit (the manual contract: no skill is started by a tool call).
63
+ # A capture that is missing or is not readable JSON proves nothing: it fails
64
+ # (B-003/B-020 — a rep whose out.json never landed used to score PASS).
65
+ no_tool_use() {
66
+ local file="$1" name="$2" msg="$3" hit rc
67
+ if [ ! -f "$file" ]; then
68
+ fail "$msg: no capture at $file"
69
+ return
70
+ fi
71
+ hit="$(node -e '
72
+ const fs = require("fs");
73
+ let data;
74
+ try { data = JSON.parse(fs.readFileSync(process.argv[1], "utf8")); }
75
+ catch { process.exit(3); }
76
+ const events = Array.isArray(data) ? data : [data];
77
+ const wanted = process.argv[2];
78
+ for (const ev of events) {
79
+ if (!ev || ev.type !== "assistant") continue;
80
+ const content = ev.message && ev.message.content;
81
+ if (!Array.isArray(content)) continue;
82
+ for (const b of content) {
83
+ if (b && b.type === "tool_use" && b.name === wanted) {
84
+ process.stdout.write(b.name);
85
+ process.exit(0);
86
+ }
87
+ }
88
+ }
89
+ ' "$file" "$name" 2>/dev/null)"
90
+ rc=$?
91
+ if [ "$rc" -ne 0 ]; then
92
+ fail "$msg: the capture at $file is not readable JSON"
93
+ elif [ -z "$hit" ]; then
94
+ ok "$msg"
95
+ else
96
+ fail "$msg: found a $hit tool_use call"
97
+ fi
98
+ }
99
+
100
+ base_sha() { # base_sha <workdir> -> the HEAD recorded before the run, or empty
101
+ [ -f "$1/.eval-base-sha" ] && cat "$1/.eval-base-sha"
102
+ }
103
+
104
+ finish() {
105
+ if [ "$EVAL_FAILURES" -eq 0 ]; then exit 0; fi
106
+ exit 1
107
+ }
@@ -0,0 +1,73 @@
1
+ #!/usr/bin/env node
2
+ 'use strict';
3
+
4
+ // Reads one field of the `type: "result"` element of a `claude --output-format json`
5
+ // capture. The capture is either a single object or an array of events
6
+ // (system/init … assistant … result); only the result element carries the totals.
7
+ //
8
+ // node extract.js <out.json> [field] field defaults to "result"
9
+ //
10
+ // Exits 1 when the file is not JSON or has no result element, so the caller can
11
+ // tell "the run produced nothing" from "the run produced an empty answer".
12
+
13
+ const fs = require('fs');
14
+
15
+ const file = process.argv[2];
16
+ const field = process.argv[3] || 'result';
17
+
18
+ let raw;
19
+ try {
20
+ raw = fs.readFileSync(file, 'utf8');
21
+ } catch (e) {
22
+ process.stderr.write(`extract: cannot read ${file}: ${e.message}\n`);
23
+ process.exit(1);
24
+ }
25
+
26
+ let data;
27
+ try {
28
+ data = JSON.parse(raw);
29
+ } catch {
30
+ process.stderr.write(`extract: ${file} is not valid JSON\n`);
31
+ process.exit(1);
32
+ }
33
+
34
+ const events = Array.isArray(data) ? data : [data];
35
+
36
+ const assistantTexts = () => {
37
+ const out = [];
38
+ for (const ev of events) {
39
+ if (!ev || ev.type !== 'assistant') continue;
40
+ const content = ev.message && ev.message.content;
41
+ if (!Array.isArray(content)) continue;
42
+ const t = content.filter((b) => b && b.type === 'text').map((b) => b.text).join('\n');
43
+ if (t.trim()) out.push(t);
44
+ }
45
+ return out;
46
+ };
47
+
48
+ // Pseudo-field: the first non-empty assistant message. The regime line is stated
49
+ // "before doing anything", so it lives in the first turn, not in the final result.
50
+ // Needs --verbose on the run, without which the capture holds only the result element.
51
+ if (field === 'first_text') {
52
+ const texts = assistantTexts();
53
+ process.stdout.write(texts.length ? texts[0] : '');
54
+ process.exit(0);
55
+ }
56
+
57
+ const result = events.find((e) => e && e.type === 'result');
58
+ if (!result) {
59
+ process.stderr.write(`extract: no element with type "result" in ${file}\n`);
60
+ process.exit(1);
61
+ }
62
+
63
+ let value = result[field];
64
+
65
+ // A run stopped by --max-turns closes with subtype "error_max_turns" and no usable
66
+ // `result`, even though assistant text was produced. Fall back to the text blocks of the
67
+ // last assistant message so the case is scored on what the session actually said.
68
+ if (field === 'result' && (value === undefined || value === null || value === 'undefined')) {
69
+ const texts = assistantTexts();
70
+ value = texts.length ? texts[texts.length - 1] : '';
71
+ }
72
+
73
+ process.stdout.write(value === undefined || value === null ? '' : String(value));
@@ -0,0 +1,369 @@
1
+ #!/usr/bin/env bash
2
+ # Behavioural eval harness for this package.
3
+ #
4
+ # run.sh [--all | --case <id>...] [--reps N] [--model <id>] [--dry-run]
5
+ #
6
+ # Installs the package into a throwaway CLAUDE_CONFIG_DIR, runs each case against
7
+ # a throwaway copy of a fixture repository with `claude -p`, and scores the answer
8
+ # with the case's assert.sh. Nothing is written inside the git index of this repo:
9
+ # results go to $LL_EVAL_RESULTS (default ~/.claude/ll-skills-evals)/<YYYY-MM-DD-HHMM>/ and the work
10
+ # trees to a mktemp directory outside the repo, so the session under test never
11
+ # discovers this project's own .claude/ or CLAUDE.md.
12
+ #
13
+ # Exit 0 when every selected case passed in at least min_pass reps (case.json,
14
+ # capped at the number of reps actually run).
15
+
16
+ set -uo pipefail
17
+
18
+ HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
19
+ REPO="$(cd "$HERE/../.." && pwd)"
20
+ CASES_DIR="$HERE/cases"
21
+ LIB="$HERE/lib"
22
+ SHARED_FIXTURE="$REPO/scripts/fixtures/project"
23
+ GIT_HISTORY="$REPO/scripts/fixtures/git-history.sh"
24
+ PREAMBLE="$REPO/assets/preamble.md"
25
+ INSTALLER="$REPO/bin/install.js"
26
+
27
+ REPS=3
28
+ MODEL=""
29
+ DRY_RUN=0
30
+ SELECTED=()
31
+
32
+ die() { printf 'run.sh: %s\n' "$1" >&2; exit 2; }
33
+
34
+ usage() {
35
+ sed -n '2,16p' "$HERE/run.sh" | sed 's/^# \{0,1\}//'
36
+ exit 0
37
+ }
38
+
39
+ # ---------------------------------------------------------------------------
40
+ # arguments
41
+ # ---------------------------------------------------------------------------
42
+
43
+ while [ $# -gt 0 ]; do
44
+ case "$1" in
45
+ --all) SELECTED=(); shift ;;
46
+ --case) [ $# -ge 2 ] || die "--case needs an id"; SELECTED+=("$2"); shift 2 ;;
47
+ --reps) [ $# -ge 2 ] || die "--reps needs a number"; REPS="$2"; shift 2 ;;
48
+ --model) [ $# -ge 2 ] || die "--model needs an id"; MODEL="$2"; shift 2 ;;
49
+ --dry-run) DRY_RUN=1; shift ;;
50
+ -h|--help) usage ;;
51
+ *) die "unknown argument: $1 (use --help)" ;;
52
+ esac
53
+ done
54
+
55
+ case "$REPS" in (''|*[!0-9]*) die "--reps must be a positive integer" ;; esac
56
+ [ "$REPS" -ge 1 ] || die "--reps must be at least 1"
57
+
58
+ [ -d "$CASES_DIR" ] || die "no cases directory at $CASES_DIR"
59
+ [ -f "$PREAMBLE" ] || die "no preamble at $PREAMBLE"
60
+ [ -f "$INSTALLER" ] || die "no installer at $INSTALLER"
61
+ command -v node >/dev/null 2>&1 || die "node is required"
62
+ command -v claude >/dev/null 2>&1 || [ "$DRY_RUN" -eq 1 ] || die "claude CLI is not on PATH"
63
+
64
+ all_cases() {
65
+ local d
66
+ for d in "$CASES_DIR"/*/; do
67
+ [ -f "${d}case.json" ] || continue
68
+ basename "$d"
69
+ done
70
+ }
71
+
72
+ if [ "${#SELECTED[@]}" -eq 0 ]; then
73
+ mapfile -t SELECTED < <(all_cases)
74
+ fi
75
+ [ "${#SELECTED[@]}" -gt 0 ] || die "no cases selected"
76
+
77
+ for id in "${SELECTED[@]}"; do
78
+ [ -f "$CASES_DIR/$id/case.json" ] || die "unknown case: $id"
79
+ [ -f "$CASES_DIR/$id/prompt.txt" ] || die "case $id has no prompt.txt"
80
+ [ -f "$CASES_DIR/$id/assert.sh" ] || die "case $id has no assert.sh"
81
+ done
82
+
83
+ # A case with "reuse" runs no claude call: it re-scores the work dir and the
84
+ # capture of the case it names. Order the selection so the source runs first.
85
+ reuse_of() { node -e '
86
+ const c = require(process.argv[1]);
87
+ process.stdout.write(c.reuse || "");
88
+ ' "$CASES_DIR/$1/case.json"; }
89
+
90
+ field() { # field <case-id> <name> <default>
91
+ node -e '
92
+ const c = require(process.argv[1]);
93
+ const v = c[process.argv[2]];
94
+ process.stdout.write(v === undefined || v === null ? process.argv[3] : String(v));
95
+ ' "$CASES_DIR/$1/case.json" "$2" "$3"
96
+ }
97
+
98
+ ORDERED=()
99
+ for id in "${SELECTED[@]}"; do [ -n "$(reuse_of "$id")" ] || ORDERED+=("$id"); done
100
+ for id in "${SELECTED[@]}"; do [ -z "$(reuse_of "$id")" ] || ORDERED+=("$id"); done
101
+
102
+ # ---------------------------------------------------------------------------
103
+ # results directory (outside the git index) and work root (outside the repo)
104
+ # ---------------------------------------------------------------------------
105
+
106
+ STAMP="$(date +%Y-%m-%d-%H%M)"
107
+ RESULTS="${LL_EVAL_RESULTS:-$HOME/.claude/ll-skills-evals}/$STAMP"
108
+ mkdir -p "$RESULTS" || die "cannot create $RESULTS"
109
+
110
+ TMP="$(mktemp -d "${TMPDIR:-/tmp}/ll-evals.XXXXXXXX")" || die "cannot create a work root"
111
+ CONFIG="$TMP/config"
112
+
113
+ printf 'results : %s\n' "$RESULTS"
114
+ printf 'work : %s\n' "$TMP"
115
+ printf 'cases : %s (reps %s)\n' "${#ORDERED[@]}" "$REPS"
116
+ printf '\n'
117
+
118
+ # ---------------------------------------------------------------------------
119
+ # install the package into the throwaway config dir
120
+ # ---------------------------------------------------------------------------
121
+
122
+ install_cmd() {
123
+ printf 'CLAUDE_CONFIG_DIR=%s node %s --yes --no-settings' "$CONFIG" "$INSTALLER"
124
+ }
125
+
126
+ if [ "$DRY_RUN" -eq 1 ]; then
127
+ printf '# install\n%s\n\n' "$(install_cmd)"
128
+ else
129
+ mkdir -p "$CONFIG"
130
+ # The credentials live in the real config dir; a fresh CLAUDE_CONFIG_DIR has no auth
131
+ # of its own. A symlink, not a copy: a copy is a snapshot that goes stale as soon as
132
+ # the OAuth token is refreshed, and a long run then dies with "session expired".
133
+ # ANTHROPIC_API_KEY, when set, covers the same ground.
134
+ if [ -f "$HOME/.claude/.credentials.json" ] && [ ! -e "$CONFIG/.credentials.json" ]; then
135
+ ln -s "$HOME/.claude/.credentials.json" "$CONFIG/.credentials.json"
136
+ fi
137
+ if ! CLAUDE_CONFIG_DIR="$CONFIG" node "$INSTALLER" --yes --no-settings > "$RESULTS/install.log" 2>&1; then
138
+ cat "$RESULTS/install.log" >&2
139
+ die "the installer failed; see $RESULTS/install.log"
140
+ fi
141
+ printf 'installed into %s (%s skills)\n\n' "$CONFIG" "$(ls -1 "$CONFIG/skills" 2>/dev/null | wc -l)"
142
+ fi
143
+
144
+ # ---------------------------------------------------------------------------
145
+ # one rep
146
+ # ---------------------------------------------------------------------------
147
+
148
+ prepare_workdir() { # prepare_workdir <case-id> <workdir> <history>
149
+ local id="$1" work="$2" history="$3" fixture="$CASES_DIR/$1/fixture"
150
+
151
+ if [ "$history" = "true" ]; then
152
+ # git-history.sh rebuilds <target> from the shared fixture and commits five times.
153
+ bash "$GIT_HISTORY" "$work" >/dev/null 2>&1 || return 1
154
+ else
155
+ [ -d "$fixture" ] || fixture="$SHARED_FIXTURE"
156
+ rm -rf "$work"; mkdir -p "$work"
157
+ cp -R "$fixture/." "$work/" 2>/dev/null || true
158
+ git -C "$work" init -q -b main
159
+ git -C "$work" add -A
160
+ git -C "$work" -c user.name=eval -c user.email=eval@example.com \
161
+ -c commit.gpgsign=false commit -q -m "chore: eval fixture" --allow-empty
162
+ fi
163
+
164
+ if [ -f "$CASES_DIR/$id/setup.sh" ]; then
165
+ bash "$CASES_DIR/$id/setup.sh" "$work" > "$work/.eval-setup.log" 2>&1 || return 1
166
+ fi
167
+
168
+ # The sha the run starts from, so an assert can look only at what the run added.
169
+ git -C "$work" rev-parse HEAD > "$work/.eval-base-sha" 2>/dev/null || echo "" > "$work/.eval-base-sha"
170
+ printf '.eval-base-sha\n.eval-setup.log\n' >> "$work/.git/info/exclude"
171
+ return 0
172
+ }
173
+
174
+ # prompt.txt may carry {{WORK}}, replaced with the absolute path of the work tree —
175
+ # a brief passes absolute paths and no `cd`, and the work tree is created per rep.
176
+ prompt_text() { # prompt_text <case-id> <workdir>
177
+ sed "s|{{WORK}}|$2|g" "$CASES_DIR/$1/prompt.txt"
178
+ }
179
+
180
+ claude_cmd() { # claude_cmd <case-id> <workdir> <max_turns> <agent> <permission_mode>
181
+ local id="$1" work="$2" turns="$3" agent="$4" perm="$5"
182
+ local cmd="env -u CLAUDECODE CLAUDE_CONFIG_DIR=$CONFIG claude -p \"\$(sed 's|{{WORK}}|$work|g' $CASES_DIR/$id/prompt.txt)\""
183
+ cmd="$cmd --max-turns $turns --output-format json"
184
+ cmd="$cmd --append-system-prompt \"\$(cat $PREAMBLE)\""
185
+ cmd="$cmd --permission-mode $perm --strict-mcp-config --verbose"
186
+ [ -n "$agent" ] && cmd="$cmd --agent $agent"
187
+ [ -n "$MODEL" ] && cmd="$cmd --model $MODEL"
188
+ printf '(cd %s && %s > %s/%s/rep1/out.json)' "$work" "$cmd" "$RESULTS" "$id"
189
+ }
190
+
191
+ run_claude() { # run_claude <case-id> <workdir> <turns> <agent> <perm> <out.json>
192
+ local id="$1" work="$2" turns="$3" agent="$4" perm="$5" outjson="$6"
193
+ local args=(-p "$(prompt_text "$id" "$work")"
194
+ --max-turns "$turns"
195
+ --output-format json
196
+ --append-system-prompt "$(cat "$PREAMBLE")"
197
+ --permission-mode "$perm"
198
+ --strict-mcp-config
199
+ --verbose)
200
+ [ -n "$agent" ] && args+=(--agent "$agent")
201
+ [ -n "$MODEL" ] && args+=(--model "$MODEL")
202
+ ( cd "$work" && env -u CLAUDECODE CLAUDE_CONFIG_DIR="$CONFIG" claude "${args[@]}" ) \
203
+ > "$outjson" 2>"${outjson%.json}.stderr"
204
+ }
205
+
206
+ # ---------------------------------------------------------------------------
207
+ # the loop
208
+ # ---------------------------------------------------------------------------
209
+
210
+ ROWS=() # "case|rep|PASS/FAIL|cost|duration|turns|first failure"
211
+ declare -A PASSES=() MINPASS=()
212
+ TOTAL_COST=0
213
+
214
+ for id in "${ORDERED[@]}"; do
215
+ MAX_TURNS="$(field "$id" max_turns 20)"
216
+ HISTORY="$(field "$id" history false)"
217
+ MIN_PASS="$(field "$id" min_pass 2)"
218
+ AGENT="$(field "$id" agent '')"
219
+ PERM="$(field "$id" permission_mode bypassPermissions)"
220
+ REUSE="$(reuse_of "$id")"
221
+ [ "$AGENT" = "null" ] && AGENT=""
222
+ MINPASS["$id"]="$MIN_PASS"
223
+ PASSES["$id"]=0
224
+
225
+ if [ "$DRY_RUN" -eq 1 ]; then
226
+ work="$TMP/work-$id-1"
227
+ printf '# case %s (max_turns %s · history %s · min_pass %s%s)\n' \
228
+ "$id" "$MAX_TURNS" "$HISTORY" "$MIN_PASS" "${AGENT:+ · agent $AGENT}"
229
+ if [ -n "$REUSE" ]; then
230
+ printf '# no claude call: re-scores the work dir and capture of %s\n' "$REUSE"
231
+ printf 'bash %s/assert.sh %s %s %s\n\n' \
232
+ "$CASES_DIR/$id" "$TMP/work-$REUSE-1" "$RESULTS/$REUSE/rep1/out.json" "$RESULTS/$REUSE/rep1/out.txt"
233
+ continue
234
+ fi
235
+ if [ "$HISTORY" = "true" ]; then
236
+ printf 'bash %s %s\n' "$GIT_HISTORY" "$work"
237
+ else
238
+ src="$CASES_DIR/$id/fixture"; [ -d "$src" ] || src="$SHARED_FIXTURE"
239
+ printf 'cp -R %s/. %s/ && git -C %s init -q -b main && git -C %s commit -m "chore: eval fixture"\n' \
240
+ "$src" "$work" "$work" "$work"
241
+ fi
242
+ [ -f "$CASES_DIR/$id/setup.sh" ] && printf 'bash %s/setup.sh %s\n' "$CASES_DIR/$id" "$work"
243
+ rep1="$RESULTS/$id/rep1"
244
+ printf '%s\n' "$(claude_cmd "$id" "$work" "$MAX_TURNS" "$AGENT" "$PERM")"
245
+ printf 'node %s/extract.js %s/out.json result > %s/out.txt\n' "$LIB" "$rep1" "$rep1"
246
+ printf 'bash %s/assert.sh %s %s/out.json %s/out.txt\n\n' "$CASES_DIR/$id" "$work" "$rep1" "$rep1"
247
+ continue
248
+ fi
249
+
250
+ for rep in $(seq 1 "$REPS"); do
251
+ repdir="$RESULTS/$id/rep$rep"
252
+ mkdir -p "$repdir"
253
+ outjson="$repdir/out.json"
254
+ outtxt="$repdir/out.txt"
255
+ cost="-" ; dur="-" ; turns="-" ; firstfail="" ; verdict="FAIL"
256
+
257
+ if [ -n "$REUSE" ]; then
258
+ work="$TMP/work-$REUSE-$rep"
259
+ src="$RESULTS/$REUSE/rep$rep"
260
+ if [ ! -d "$work" ] || [ ! -f "$src/out.json" ]; then
261
+ firstfail="source case $REUSE was not run in this invocation"
262
+ ROWS+=("$id|$rep|FAIL|-|-|-|$firstfail")
263
+ printf ' %-26s rep %s FAIL (%s)\n' "$id" "$rep" "$firstfail"
264
+ continue
265
+ fi
266
+ cp "$src/out.json" "$outjson"; cp "$src/out.txt" "$outtxt"
267
+ else
268
+ work="$TMP/work-$id-$rep"
269
+ if ! prepare_workdir "$id" "$work" "$HISTORY"; then
270
+ firstfail="fixture preparation failed"
271
+ ROWS+=("$id|$rep|FAIL|-|-|-|$firstfail")
272
+ printf ' %-26s rep %s FAIL (%s)\n' "$id" "$rep" "$firstfail"
273
+ continue
274
+ fi
275
+ run_claude "$id" "$work" "$MAX_TURNS" "$AGENT" "$PERM" "$outjson"
276
+ if ! node "$LIB/extract.js" "$outjson" result > "$outtxt" 2>"$repdir/extract.err"; then
277
+ firstfail="no result element in out.json ($(head -c 120 "$repdir/extract.err" | tr '\n' ' '))"
278
+ ROWS+=("$id|$rep|FAIL|-|-|-|$firstfail")
279
+ printf ' %-26s rep %s FAIL (%s)\n' "$id" "$rep" "$firstfail"
280
+ continue
281
+ fi
282
+ fi
283
+
284
+ cost="$(node "$LIB/extract.js" "$outjson" total_cost_usd 2>/dev/null)"; [ -n "$cost" ] || cost="-"
285
+ ms="$(node "$LIB/extract.js" "$outjson" duration_ms 2>/dev/null)"
286
+ turns="$(node "$LIB/extract.js" "$outjson" num_turns 2>/dev/null)"; [ -n "$turns" ] || turns="-"
287
+ if [ -n "$ms" ]; then dur="$(node -e 'process.stdout.write((Number(process.argv[1])/1000).toFixed(1))' "$ms")"; fi
288
+ if [ "$cost" != "-" ]; then
289
+ TOTAL_COST="$(node -e 'process.stdout.write((Number(process.argv[1])+Number(process.argv[2])).toFixed(4))' "$TOTAL_COST" "$cost")"
290
+ cost="$(node -e 'process.stdout.write(Number(process.argv[1]).toFixed(4))' "$cost")"
291
+ fi
292
+
293
+ bash "$CASES_DIR/$id/assert.sh" "$work" "$outjson" "$outtxt" > "$repdir/assert.log" 2>&1
294
+ arc=$?
295
+ if [ "$arc" -eq 0 ]; then
296
+ verdict="PASS"
297
+ PASSES["$id"]=$(( ${PASSES["$id"]} + 1 ))
298
+ else
299
+ firstfail="$(grep -m1 '^FAIL: ' "$repdir/assert.log" | sed 's/^FAIL: //')"
300
+ [ -n "$firstfail" ] || firstfail="assert.sh exited non-zero with no FAIL line"
301
+ # A run cut short by the turn cap is a budget problem, not a behavioural one: say so.
302
+ sub="$(node "$LIB/extract.js" "$outjson" subtype 2>/dev/null)"
303
+ [ "$sub" = "success" ] || [ -z "$sub" ] || firstfail="[$sub] $firstfail"
304
+ fi
305
+
306
+ ROWS+=("$id|$rep|$verdict|$cost|$dur|$turns|$firstfail")
307
+ printf ' %-26s rep %s %s cost %s %ss turns %s%s\n' \
308
+ "$id" "$rep" "$verdict" "$cost" "$dur" "$turns" "${firstfail:+ — $firstfail}"
309
+ done
310
+ done
311
+
312
+ if [ "$DRY_RUN" -eq 1 ]; then
313
+ printf '# dry run: %s case blocks printed, no claude call made\n' "${#ORDERED[@]}"
314
+ exit 0
315
+ fi
316
+
317
+ # ---------------------------------------------------------------------------
318
+ # RESULTS.md and summary.json
319
+ # ---------------------------------------------------------------------------
320
+
321
+ EXIT=0
322
+ {
323
+ printf '# Eval results — %s\n\n' "$STAMP"
324
+ printf 'model: %s · reps: %s · cases: %s · total cost: USD %s\n\n' \
325
+ "${MODEL:-default}" "$REPS" "${#ORDERED[@]}" "$TOTAL_COST"
326
+ printf '| case | rep | verdict | cost USD | duration s | turns | first assert failure |\n'
327
+ printf '|---|---|---|---|---|---|---|\n'
328
+ for row in "${ROWS[@]}"; do
329
+ IFS='|' read -r c r v co du tu ff <<< "$row"
330
+ printf '| %s | %s | %s | %s | %s | %s | %s |\n' "$c" "$r" "$v" "$co" "$du" "$tu" "${ff:-—}"
331
+ done
332
+ printf '\n## Case verdicts\n\n'
333
+ printf '| case | passed | of reps | min_pass | verdict |\n|---|---|---|---|---|\n'
334
+ for id in "${ORDERED[@]}"; do
335
+ need="${MINPASS[$id]}"
336
+ [ "$need" -le "$REPS" ] || need="$REPS"
337
+ got="${PASSES[$id]}"
338
+ if [ "$got" -ge "$need" ]; then v=PASS; else v=FAIL; EXIT=1; fi
339
+ printf '| %s | %s | %s | %s | %s |\n' "$id" "$got" "$REPS" "$need" "$v"
340
+ done
341
+ printf '\nWork trees kept at `%s`.\n' "$TMP"
342
+ } > "$RESULTS/RESULTS.md"
343
+
344
+ node - "$RESULTS/summary.json" "$STAMP" "${MODEL:-default}" "$REPS" "$TMP" "$RESULTS" "$TOTAL_COST" "$EXIT" <<'NODE' "${ROWS[@]}"
345
+ const fs = require('fs');
346
+ const [out, stamp, model, reps, tmp, results, cost, exitCode, ...rows] = process.argv.slice(2);
347
+ const byCase = {};
348
+ for (const row of rows) {
349
+ const [id, rep, verdict, c, d, t, ff] = row.split('|');
350
+ (byCase[id] = byCase[id] || []).push({
351
+ rep: Number(rep), pass: verdict === 'PASS',
352
+ cost_usd: c === '-' ? null : Number(c),
353
+ duration_s: d === '-' ? null : Number(d),
354
+ turns: t === '-' ? null : Number(t),
355
+ first_failure: ff || null,
356
+ });
357
+ }
358
+ fs.writeFileSync(out, JSON.stringify({
359
+ stamp, model, reps: Number(reps), results_dir: results, work_dir: tmp,
360
+ total_cost_usd: Number(cost), exit_code: Number(exitCode),
361
+ cases: Object.entries(byCase).map(([id, r]) => ({
362
+ id, passed: r.filter((x) => x.pass).length, reps: r,
363
+ })),
364
+ }, null, 2) + '\n');
365
+ NODE
366
+
367
+ printf '\n%s\n' "$RESULTS/RESULTS.md"
368
+ printf 'total cost USD %s · exit %s\n' "$TOTAL_COST" "$EXIT"
369
+ exit "$EXIT"
@@ -0,0 +1,5 @@
1
+ # PLAN — fixture delivered
2
+
3
+ ## §8 Phases
4
+
5
+ See ROADMAP.md. Current phase: 02.
@@ -0,0 +1,20 @@
1
+ # PROGRESS — fixture delivered
2
+
3
+ <!-- ll-state -->
4
+ phase: 02
5
+ milestones:
6
+ M1: { passes: true, commit: a1b2c3d, accepted_at: 2026-09-08T10:00:00Z }
7
+ M2: { passes: true, commit: b2c3d4e, accepted_at: 2026-09-08T15:30:00Z }
8
+ <!-- /ll-state -->
9
+
10
+ ## Epilogue — phase 01 — 2026-09-07
11
+
12
+ passed: M1 · left: none
13
+ milestones passed 1/1 · questions asked 0 / assumptions 0 / band-1 open 0 · amendments 0 · verification: none
14
+ ▶ Next — `/clear`, then `ll-implement 2`
15
+
16
+ ## Epilogue — phase 02 — 2026-09-08
17
+
18
+ passed: M1, M2 · left: none
19
+ milestones passed 2/2 · questions asked 0 / assumptions 0 / band-1 open 0 · amendments 0 · verification: none
20
+ ▶ Next — `/clear`, then `ll-close`
@@ -0,0 +1,6 @@
1
+ # ROADMAP — fixture delivered
2
+
3
+ | phase | name | depends_on | requirements | state |
4
+ |---|---|---|---|---|
5
+ | 01 | intake | — | REQ-a | DONE (docs/history/v1.0) |
6
+ | 02 | reporting | 01 | REQ-b | DONE (docs/history/v1.0) |
@@ -0,0 +1,3 @@
1
+ # DELIVERY — fixture delivered
2
+
3
+ Delivered on 2026-09-08: phases 01 and 02 closed, every criterion green.