ll-skills 2.0.2 → 3.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (125) hide show
  1. package/CHANGELOG.md +52 -0
  2. package/README.md +42 -20
  3. package/agents/ll-executor.md +2 -1
  4. package/agents/ll-verifier.md +1 -0
  5. package/assets/preamble.md +27 -39
  6. package/bin/install.js +4 -1
  7. package/hooks/ll-precompact.js +29 -1
  8. package/hooks/ll-skills-check-update.js +6 -6
  9. package/hooks/ll-state.js +30 -2
  10. package/package.json +3 -2
  11. package/scripts/evals/README.md +81 -0
  12. package/scripts/evals/cases/auto-dry-run/assert.sh +35 -0
  13. package/scripts/evals/cases/auto-dry-run/case.json +8 -0
  14. package/scripts/evals/cases/auto-dry-run/prompt.txt +1 -0
  15. package/scripts/evals/cases/auto-empty-repo/assert.sh +25 -0
  16. package/scripts/evals/cases/auto-empty-repo/case.json +8 -0
  17. package/scripts/evals/cases/auto-empty-repo/fixture/.gitkeep +0 -0
  18. package/scripts/evals/cases/auto-empty-repo/prompt.txt +1 -0
  19. package/scripts/evals/cases/decide-final-round/assert.sh +45 -0
  20. package/scripts/evals/cases/decide-final-round/case.json +8 -0
  21. package/scripts/evals/cases/decide-final-round/fixture/README.md +3 -0
  22. package/scripts/evals/cases/decide-final-round/prompt.txt +1 -0
  23. package/scripts/evals/cases/executor-block/assert.sh +33 -0
  24. package/scripts/evals/cases/executor-block/case.json +8 -0
  25. package/scripts/evals/cases/executor-block/prompt.txt +14 -0
  26. package/scripts/evals/cases/goal-autonomous/assert.sh +35 -0
  27. package/scripts/evals/cases/goal-autonomous/case.json +8 -0
  28. package/scripts/evals/cases/goal-autonomous/fixture/PLAN.md +42 -0
  29. package/scripts/evals/cases/goal-autonomous/fixture/PROGRESS.md +20 -0
  30. package/scripts/evals/cases/goal-autonomous/fixture/ROADMAP.md +29 -0
  31. package/scripts/evals/cases/goal-autonomous/fixture/package.json +8 -0
  32. package/scripts/evals/cases/goal-autonomous/fixture/src/money.js +6 -0
  33. package/scripts/evals/cases/goal-autonomous/fixture/test/reconcile.test.js +8 -0
  34. package/scripts/evals/cases/goal-autonomous/prompt.txt +1 -0
  35. package/scripts/evals/cases/implement-review-gate/assert.sh +37 -0
  36. package/scripts/evals/cases/implement-review-gate/case.json +8 -0
  37. package/scripts/evals/cases/implement-review-gate/prompt.txt +1 -0
  38. package/scripts/evals/cases/implement-stops-at-next/assert.sh +121 -0
  39. package/scripts/evals/cases/implement-stops-at-next/case.json +9 -0
  40. package/scripts/evals/cases/implement-stops-at-next/prompt.txt +1 -0
  41. package/scripts/evals/cases/preamble-no-ritual/assert.sh +17 -0
  42. package/scripts/evals/cases/preamble-no-ritual/case.json +8 -0
  43. package/scripts/evals/cases/preamble-no-ritual/fixture/README.md +3 -0
  44. package/scripts/evals/cases/preamble-no-ritual/fixture/src/a.ts +3 -0
  45. package/scripts/evals/cases/preamble-no-ritual/prompt.txt +1 -0
  46. package/scripts/evals/cases/router-no-skill/assert.sh +33 -0
  47. package/scripts/evals/cases/router-no-skill/case.json +8 -0
  48. package/scripts/evals/cases/router-no-skill/fixture/PLAN.md +21 -0
  49. package/scripts/evals/cases/router-no-skill/fixture/README.md +7 -0
  50. package/scripts/evals/cases/router-no-skill/fixture/decisions/README.md +3 -0
  51. package/scripts/evals/cases/router-no-skill/prompt.txt +1 -0
  52. package/scripts/evals/cases/router-small/assert.sh +21 -0
  53. package/scripts/evals/cases/router-small/case.json +8 -0
  54. package/scripts/evals/cases/router-small/fixture/README.md +17 -0
  55. package/scripts/evals/cases/router-small/prompt.txt +1 -0
  56. package/scripts/evals/cases/scout-no-plan/assert.sh +41 -0
  57. package/scripts/evals/cases/scout-no-plan/case.json +8 -0
  58. package/scripts/evals/cases/scout-no-plan/prompt.txt +8 -0
  59. package/scripts/evals/cases/verifier-weakened-test/assert.sh +19 -0
  60. package/scripts/evals/cases/verifier-weakened-test/case.json +8 -0
  61. package/scripts/evals/cases/verifier-weakened-test/prompt.txt +13 -0
  62. package/scripts/evals/cases/verifier-weakened-test/setup.sh +19 -0
  63. package/scripts/evals/fixtures/manual-contract/out.json +29 -0
  64. package/scripts/evals/fixtures/manual-contract/out.txt +5 -0
  65. package/scripts/evals/fixtures/manual-contract/with-skill.json +46 -0
  66. package/scripts/evals/fixtures/router-no-skill/fail.txt +5 -0
  67. package/scripts/evals/fixtures/router-no-skill/out.json +27 -0
  68. package/scripts/evals/fixtures/router-no-skill/pass.txt +4 -0
  69. package/scripts/evals/lib/assert.sh +107 -0
  70. package/scripts/evals/lib/extract.js +73 -0
  71. package/scripts/evals/run.sh +369 -0
  72. package/scripts/fixtures/auto-closed/PLAN.md +5 -0
  73. package/scripts/fixtures/auto-closed/PROGRESS.md +20 -0
  74. package/scripts/fixtures/auto-closed/ROADMAP.md +6 -0
  75. package/scripts/fixtures/auto-closed/docs/DELIVERY.md +3 -0
  76. package/scripts/fixtures/auto-decisions/decisions/DEC-0001-taken-alone.md +13 -0
  77. package/scripts/fixtures/auto-decisions/decisions/DEC-0002-owner.md +13 -0
  78. package/scripts/fixtures/auto-noroadmap/PLAN.md +20 -0
  79. package/scripts/fixtures/auto-noroadmap/PROGRESS.md +11 -0
  80. package/scripts/fixtures/auto-verify-next/PLAN.md +5 -0
  81. package/scripts/fixtures/auto-verify-next/PROGRESS.md +18 -0
  82. package/scripts/fixtures/auto-verify-next/ROADMAP.md +5 -0
  83. package/scripts/fixtures/auto-verify-next/phases/01/PLAN.md +6 -0
  84. package/scripts/fixtures/evals-auto/auto-dry-run/pass.txt +18 -0
  85. package/scripts/fixtures/evals-auto/auto-empty-repo/pass.txt +2 -0
  86. package/scripts/fixtures/evals-auto/goal-autonomous/pass.txt +29 -0
  87. package/scripts/fixtures/lint-bad/folded-description/SKILL.md +13 -0
  88. package/scripts/fixtures/lint-bad/jargon-in-questions/SKILL.md +36 -0
  89. package/scripts/fixtures/lint-bad/model-invocation-false/SKILL.md +10 -0
  90. package/scripts/fixtures/next-bad/skills/ll-bad/SKILL.md +30 -0
  91. package/scripts/fixtures/next-good/skills/ll-good/SKILL.md +26 -0
  92. package/scripts/fixtures/project/BACKLOG.md +6 -5
  93. package/scripts/fixtures/project/PROGRESS.md +4 -0
  94. package/scripts/lint-contract.cjs +495 -0
  95. package/scripts/lint-prompts.sh +443 -0
  96. package/scripts/ll-tools.js +497 -450
  97. package/scripts/smoke-test.sh +481 -4
  98. package/skills/ll-auto/SKILL.md +74 -0
  99. package/skills/ll-auto/references/run.md +75 -0
  100. package/skills/ll-auto/references/stages.md +66 -0
  101. package/skills/ll-auto/scripts/ll-auto.js +345 -0
  102. package/skills/ll-brainstorm/SKILL.md +14 -12
  103. package/skills/ll-brainstorm/references/decision-policy.md +22 -12
  104. package/skills/ll-close/SKILL.md +8 -7
  105. package/skills/ll-close/references/delivery.md +3 -1
  106. package/skills/ll-decide/SKILL.md +21 -19
  107. package/skills/ll-decide/references/decision-policy.md +22 -12
  108. package/skills/ll-decide/references/decision-room.md +5 -3
  109. package/skills/ll-decide/references/interview.md +29 -13
  110. package/skills/ll-decide/references/plan-skeleton.md +14 -14
  111. package/skills/ll-decide/references/premise-gate.md +34 -20
  112. package/skills/ll-goal/SKILL.md +22 -4
  113. package/skills/ll-goal/references/goal-template.md +57 -0
  114. package/skills/ll-implement/SKILL.md +25 -22
  115. package/skills/ll-implement/references/briefs.md +2 -1
  116. package/skills/ll-implement/references/decision-policy.md +22 -12
  117. package/skills/ll-implement/references/phase-conversation.md +20 -12
  118. package/skills/ll-implement/references/phase-plan.md +23 -0
  119. package/skills/ll-oncall/SKILL.md +3 -2
  120. package/skills/ll-refine/SKILL.md +3 -2
  121. package/skills/ll-research/SKILL.md +3 -2
  122. package/skills/ll-resume/SKILL.md +9 -4
  123. package/skills/ll-update/SKILL.md +6 -1
  124. package/skills/ll-verify/SKILL.md +2 -1
  125. package/skills/ll-verify/references/verifier-briefs.md +3 -0
@@ -0,0 +1,369 @@
1
+ #!/usr/bin/env bash
2
+ # Behavioural eval harness for this package.
3
+ #
4
+ # run.sh [--all | --case <id>...] [--reps N] [--model <id>] [--dry-run]
5
+ #
6
+ # Installs the package into a throwaway CLAUDE_CONFIG_DIR, runs each case against
7
+ # a throwaway copy of a fixture repository with `claude -p`, and scores the answer
8
+ # with the case's assert.sh. Nothing is written inside the git index of this repo:
9
+ # results go to $LL_EVAL_RESULTS (default ~/.claude/ll-skills-evals)/<YYYY-MM-DD-HHMM>/ and the work
10
+ # trees to a mktemp directory outside the repo, so the session under test never
11
+ # discovers this project's own .claude/ or CLAUDE.md.
12
+ #
13
+ # Exit 0 when every selected case passed in at least min_pass reps (case.json,
14
+ # capped at the number of reps actually run).
15
+
16
+ set -uo pipefail
17
+
18
+ HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
19
+ REPO="$(cd "$HERE/../.." && pwd)"
20
+ CASES_DIR="$HERE/cases"
21
+ LIB="$HERE/lib"
22
+ SHARED_FIXTURE="$REPO/scripts/fixtures/project"
23
+ GIT_HISTORY="$REPO/scripts/fixtures/git-history.sh"
24
+ PREAMBLE="$REPO/assets/preamble.md"
25
+ INSTALLER="$REPO/bin/install.js"
26
+
27
+ REPS=3
28
+ MODEL=""
29
+ DRY_RUN=0
30
+ SELECTED=()
31
+
32
+ die() { printf 'run.sh: %s\n' "$1" >&2; exit 2; }
33
+
34
+ usage() {
35
+ sed -n '2,16p' "$HERE/run.sh" | sed 's/^# \{0,1\}//'
36
+ exit 0
37
+ }
38
+
39
+ # ---------------------------------------------------------------------------
40
+ # arguments
41
+ # ---------------------------------------------------------------------------
42
+
43
+ while [ $# -gt 0 ]; do
44
+ case "$1" in
45
+ --all) SELECTED=(); shift ;;
46
+ --case) [ $# -ge 2 ] || die "--case needs an id"; SELECTED+=("$2"); shift 2 ;;
47
+ --reps) [ $# -ge 2 ] || die "--reps needs a number"; REPS="$2"; shift 2 ;;
48
+ --model) [ $# -ge 2 ] || die "--model needs an id"; MODEL="$2"; shift 2 ;;
49
+ --dry-run) DRY_RUN=1; shift ;;
50
+ -h|--help) usage ;;
51
+ *) die "unknown argument: $1 (use --help)" ;;
52
+ esac
53
+ done
54
+
55
+ case "$REPS" in (''|*[!0-9]*) die "--reps must be a positive integer" ;; esac
56
+ [ "$REPS" -ge 1 ] || die "--reps must be at least 1"
57
+
58
+ [ -d "$CASES_DIR" ] || die "no cases directory at $CASES_DIR"
59
+ [ -f "$PREAMBLE" ] || die "no preamble at $PREAMBLE"
60
+ [ -f "$INSTALLER" ] || die "no installer at $INSTALLER"
61
+ command -v node >/dev/null 2>&1 || die "node is required"
62
+ command -v claude >/dev/null 2>&1 || [ "$DRY_RUN" -eq 1 ] || die "claude CLI is not on PATH"
63
+
64
+ all_cases() {
65
+ local d
66
+ for d in "$CASES_DIR"/*/; do
67
+ [ -f "${d}case.json" ] || continue
68
+ basename "$d"
69
+ done
70
+ }
71
+
72
+ if [ "${#SELECTED[@]}" -eq 0 ]; then
73
+ mapfile -t SELECTED < <(all_cases)
74
+ fi
75
+ [ "${#SELECTED[@]}" -gt 0 ] || die "no cases selected"
76
+
77
+ for id in "${SELECTED[@]}"; do
78
+ [ -f "$CASES_DIR/$id/case.json" ] || die "unknown case: $id"
79
+ [ -f "$CASES_DIR/$id/prompt.txt" ] || die "case $id has no prompt.txt"
80
+ [ -f "$CASES_DIR/$id/assert.sh" ] || die "case $id has no assert.sh"
81
+ done
82
+
83
+ # A case with "reuse" runs no claude call: it re-scores the work dir and the
84
+ # capture of the case it names. Order the selection so the source runs first.
85
+ reuse_of() { node -e '
86
+ const c = require(process.argv[1]);
87
+ process.stdout.write(c.reuse || "");
88
+ ' "$CASES_DIR/$1/case.json"; }
89
+
90
+ field() { # field <case-id> <name> <default>
91
+ node -e '
92
+ const c = require(process.argv[1]);
93
+ const v = c[process.argv[2]];
94
+ process.stdout.write(v === undefined || v === null ? process.argv[3] : String(v));
95
+ ' "$CASES_DIR/$1/case.json" "$2" "$3"
96
+ }
97
+
98
+ ORDERED=()
99
+ for id in "${SELECTED[@]}"; do [ -n "$(reuse_of "$id")" ] || ORDERED+=("$id"); done
100
+ for id in "${SELECTED[@]}"; do [ -z "$(reuse_of "$id")" ] || ORDERED+=("$id"); done
101
+
102
+ # ---------------------------------------------------------------------------
103
+ # results directory (outside the git index) and work root (outside the repo)
104
+ # ---------------------------------------------------------------------------
105
+
106
+ STAMP="$(date +%Y-%m-%d-%H%M)"
107
+ RESULTS="${LL_EVAL_RESULTS:-$HOME/.claude/ll-skills-evals}/$STAMP"
108
+ mkdir -p "$RESULTS" || die "cannot create $RESULTS"
109
+
110
+ TMP="$(mktemp -d "${TMPDIR:-/tmp}/ll-evals.XXXXXXXX")" || die "cannot create a work root"
111
+ CONFIG="$TMP/config"
112
+
113
+ printf 'results : %s\n' "$RESULTS"
114
+ printf 'work : %s\n' "$TMP"
115
+ printf 'cases : %s (reps %s)\n' "${#ORDERED[@]}" "$REPS"
116
+ printf '\n'
117
+
118
+ # ---------------------------------------------------------------------------
119
+ # install the package into the throwaway config dir
120
+ # ---------------------------------------------------------------------------
121
+
122
+ install_cmd() {
123
+ printf 'CLAUDE_CONFIG_DIR=%s node %s --yes --no-settings' "$CONFIG" "$INSTALLER"
124
+ }
125
+
126
+ if [ "$DRY_RUN" -eq 1 ]; then
127
+ printf '# install\n%s\n\n' "$(install_cmd)"
128
+ else
129
+ mkdir -p "$CONFIG"
130
+ # The credentials live in the real config dir; a fresh CLAUDE_CONFIG_DIR has no auth
131
+ # of its own. A symlink, not a copy: a copy is a snapshot that goes stale as soon as
132
+ # the OAuth token is refreshed, and a long run then dies with "session expired".
133
+ # ANTHROPIC_API_KEY, when set, covers the same ground.
134
+ if [ -f "$HOME/.claude/.credentials.json" ] && [ ! -e "$CONFIG/.credentials.json" ]; then
135
+ ln -s "$HOME/.claude/.credentials.json" "$CONFIG/.credentials.json"
136
+ fi
137
+ if ! CLAUDE_CONFIG_DIR="$CONFIG" node "$INSTALLER" --yes --no-settings > "$RESULTS/install.log" 2>&1; then
138
+ cat "$RESULTS/install.log" >&2
139
+ die "the installer failed; see $RESULTS/install.log"
140
+ fi
141
+ printf 'installed into %s (%s skills)\n\n' "$CONFIG" "$(ls -1 "$CONFIG/skills" 2>/dev/null | wc -l)"
142
+ fi
143
+
144
+ # ---------------------------------------------------------------------------
145
+ # one rep
146
+ # ---------------------------------------------------------------------------
147
+
148
+ prepare_workdir() { # prepare_workdir <case-id> <workdir> <history>
149
+ local id="$1" work="$2" history="$3" fixture="$CASES_DIR/$1/fixture"
150
+
151
+ if [ "$history" = "true" ]; then
152
+ # git-history.sh rebuilds <target> from the shared fixture and commits five times.
153
+ bash "$GIT_HISTORY" "$work" >/dev/null 2>&1 || return 1
154
+ else
155
+ [ -d "$fixture" ] || fixture="$SHARED_FIXTURE"
156
+ rm -rf "$work"; mkdir -p "$work"
157
+ cp -R "$fixture/." "$work/" 2>/dev/null || true
158
+ git -C "$work" init -q -b main
159
+ git -C "$work" add -A
160
+ git -C "$work" -c user.name=eval -c user.email=eval@example.com \
161
+ -c commit.gpgsign=false commit -q -m "chore: eval fixture" --allow-empty
162
+ fi
163
+
164
+ if [ -f "$CASES_DIR/$id/setup.sh" ]; then
165
+ bash "$CASES_DIR/$id/setup.sh" "$work" > "$work/.eval-setup.log" 2>&1 || return 1
166
+ fi
167
+
168
+ # The sha the run starts from, so an assert can look only at what the run added.
169
+ git -C "$work" rev-parse HEAD > "$work/.eval-base-sha" 2>/dev/null || echo "" > "$work/.eval-base-sha"
170
+ printf '.eval-base-sha\n.eval-setup.log\n' >> "$work/.git/info/exclude"
171
+ return 0
172
+ }
173
+
174
+ # prompt.txt may carry {{WORK}}, replaced with the absolute path of the work tree —
175
+ # a brief passes absolute paths and no `cd`, and the work tree is created per rep.
176
+ prompt_text() { # prompt_text <case-id> <workdir>
177
+ sed "s|{{WORK}}|$2|g" "$CASES_DIR/$1/prompt.txt"
178
+ }
179
+
180
+ claude_cmd() { # claude_cmd <case-id> <workdir> <max_turns> <agent> <permission_mode>
181
+ local id="$1" work="$2" turns="$3" agent="$4" perm="$5"
182
+ local cmd="env -u CLAUDECODE CLAUDE_CONFIG_DIR=$CONFIG claude -p \"\$(sed 's|{{WORK}}|$work|g' $CASES_DIR/$id/prompt.txt)\""
183
+ cmd="$cmd --max-turns $turns --output-format json"
184
+ cmd="$cmd --append-system-prompt \"\$(cat $PREAMBLE)\""
185
+ cmd="$cmd --permission-mode $perm --strict-mcp-config --verbose"
186
+ [ -n "$agent" ] && cmd="$cmd --agent $agent"
187
+ [ -n "$MODEL" ] && cmd="$cmd --model $MODEL"
188
+ printf '(cd %s && %s > %s/%s/rep1/out.json)' "$work" "$cmd" "$RESULTS" "$id"
189
+ }
190
+
191
+ run_claude() { # run_claude <case-id> <workdir> <turns> <agent> <perm> <out.json>
192
+ local id="$1" work="$2" turns="$3" agent="$4" perm="$5" outjson="$6"
193
+ local args=(-p "$(prompt_text "$id" "$work")"
194
+ --max-turns "$turns"
195
+ --output-format json
196
+ --append-system-prompt "$(cat "$PREAMBLE")"
197
+ --permission-mode "$perm"
198
+ --strict-mcp-config
199
+ --verbose)
200
+ [ -n "$agent" ] && args+=(--agent "$agent")
201
+ [ -n "$MODEL" ] && args+=(--model "$MODEL")
202
+ ( cd "$work" && env -u CLAUDECODE CLAUDE_CONFIG_DIR="$CONFIG" claude "${args[@]}" ) \
203
+ > "$outjson" 2>"${outjson%.json}.stderr"
204
+ }
205
+
206
+ # ---------------------------------------------------------------------------
207
+ # the loop
208
+ # ---------------------------------------------------------------------------
209
+
210
+ ROWS=() # "case|rep|PASS/FAIL|cost|duration|turns|first failure"
211
+ declare -A PASSES=() MINPASS=()
212
+ TOTAL_COST=0
213
+
214
+ for id in "${ORDERED[@]}"; do
215
+ MAX_TURNS="$(field "$id" max_turns 20)"
216
+ HISTORY="$(field "$id" history false)"
217
+ MIN_PASS="$(field "$id" min_pass 2)"
218
+ AGENT="$(field "$id" agent '')"
219
+ PERM="$(field "$id" permission_mode bypassPermissions)"
220
+ REUSE="$(reuse_of "$id")"
221
+ [ "$AGENT" = "null" ] && AGENT=""
222
+ MINPASS["$id"]="$MIN_PASS"
223
+ PASSES["$id"]=0
224
+
225
+ if [ "$DRY_RUN" -eq 1 ]; then
226
+ work="$TMP/work-$id-1"
227
+ printf '# case %s (max_turns %s · history %s · min_pass %s%s)\n' \
228
+ "$id" "$MAX_TURNS" "$HISTORY" "$MIN_PASS" "${AGENT:+ · agent $AGENT}"
229
+ if [ -n "$REUSE" ]; then
230
+ printf '# no claude call: re-scores the work dir and capture of %s\n' "$REUSE"
231
+ printf 'bash %s/assert.sh %s %s %s\n\n' \
232
+ "$CASES_DIR/$id" "$TMP/work-$REUSE-1" "$RESULTS/$REUSE/rep1/out.json" "$RESULTS/$REUSE/rep1/out.txt"
233
+ continue
234
+ fi
235
+ if [ "$HISTORY" = "true" ]; then
236
+ printf 'bash %s %s\n' "$GIT_HISTORY" "$work"
237
+ else
238
+ src="$CASES_DIR/$id/fixture"; [ -d "$src" ] || src="$SHARED_FIXTURE"
239
+ printf 'cp -R %s/. %s/ && git -C %s init -q -b main && git -C %s commit -m "chore: eval fixture"\n' \
240
+ "$src" "$work" "$work" "$work"
241
+ fi
242
+ [ -f "$CASES_DIR/$id/setup.sh" ] && printf 'bash %s/setup.sh %s\n' "$CASES_DIR/$id" "$work"
243
+ rep1="$RESULTS/$id/rep1"
244
+ printf '%s\n' "$(claude_cmd "$id" "$work" "$MAX_TURNS" "$AGENT" "$PERM")"
245
+ printf 'node %s/extract.js %s/out.json result > %s/out.txt\n' "$LIB" "$rep1" "$rep1"
246
+ printf 'bash %s/assert.sh %s %s/out.json %s/out.txt\n\n' "$CASES_DIR/$id" "$work" "$rep1" "$rep1"
247
+ continue
248
+ fi
249
+
250
+ for rep in $(seq 1 "$REPS"); do
251
+ repdir="$RESULTS/$id/rep$rep"
252
+ mkdir -p "$repdir"
253
+ outjson="$repdir/out.json"
254
+ outtxt="$repdir/out.txt"
255
+ cost="-" ; dur="-" ; turns="-" ; firstfail="" ; verdict="FAIL"
256
+
257
+ if [ -n "$REUSE" ]; then
258
+ work="$TMP/work-$REUSE-$rep"
259
+ src="$RESULTS/$REUSE/rep$rep"
260
+ if [ ! -d "$work" ] || [ ! -f "$src/out.json" ]; then
261
+ firstfail="source case $REUSE was not run in this invocation"
262
+ ROWS+=("$id|$rep|FAIL|-|-|-|$firstfail")
263
+ printf ' %-26s rep %s FAIL (%s)\n' "$id" "$rep" "$firstfail"
264
+ continue
265
+ fi
266
+ cp "$src/out.json" "$outjson"; cp "$src/out.txt" "$outtxt"
267
+ else
268
+ work="$TMP/work-$id-$rep"
269
+ if ! prepare_workdir "$id" "$work" "$HISTORY"; then
270
+ firstfail="fixture preparation failed"
271
+ ROWS+=("$id|$rep|FAIL|-|-|-|$firstfail")
272
+ printf ' %-26s rep %s FAIL (%s)\n' "$id" "$rep" "$firstfail"
273
+ continue
274
+ fi
275
+ run_claude "$id" "$work" "$MAX_TURNS" "$AGENT" "$PERM" "$outjson"
276
+ if ! node "$LIB/extract.js" "$outjson" result > "$outtxt" 2>"$repdir/extract.err"; then
277
+ firstfail="no result element in out.json ($(head -c 120 "$repdir/extract.err" | tr '\n' ' '))"
278
+ ROWS+=("$id|$rep|FAIL|-|-|-|$firstfail")
279
+ printf ' %-26s rep %s FAIL (%s)\n' "$id" "$rep" "$firstfail"
280
+ continue
281
+ fi
282
+ fi
283
+
284
+ cost="$(node "$LIB/extract.js" "$outjson" total_cost_usd 2>/dev/null)"; [ -n "$cost" ] || cost="-"
285
+ ms="$(node "$LIB/extract.js" "$outjson" duration_ms 2>/dev/null)"
286
+ turns="$(node "$LIB/extract.js" "$outjson" num_turns 2>/dev/null)"; [ -n "$turns" ] || turns="-"
287
+ if [ -n "$ms" ]; then dur="$(node -e 'process.stdout.write((Number(process.argv[1])/1000).toFixed(1))' "$ms")"; fi
288
+ if [ "$cost" != "-" ]; then
289
+ TOTAL_COST="$(node -e 'process.stdout.write((Number(process.argv[1])+Number(process.argv[2])).toFixed(4))' "$TOTAL_COST" "$cost")"
290
+ cost="$(node -e 'process.stdout.write(Number(process.argv[1]).toFixed(4))' "$cost")"
291
+ fi
292
+
293
+ bash "$CASES_DIR/$id/assert.sh" "$work" "$outjson" "$outtxt" > "$repdir/assert.log" 2>&1
294
+ arc=$?
295
+ if [ "$arc" -eq 0 ]; then
296
+ verdict="PASS"
297
+ PASSES["$id"]=$(( ${PASSES["$id"]} + 1 ))
298
+ else
299
+ firstfail="$(grep -m1 '^FAIL: ' "$repdir/assert.log" | sed 's/^FAIL: //')"
300
+ [ -n "$firstfail" ] || firstfail="assert.sh exited non-zero with no FAIL line"
301
+ # A run cut short by the turn cap is a budget problem, not a behavioural one: say so.
302
+ sub="$(node "$LIB/extract.js" "$outjson" subtype 2>/dev/null)"
303
+ [ "$sub" = "success" ] || [ -z "$sub" ] || firstfail="[$sub] $firstfail"
304
+ fi
305
+
306
+ ROWS+=("$id|$rep|$verdict|$cost|$dur|$turns|$firstfail")
307
+ printf ' %-26s rep %s %s cost %s %ss turns %s%s\n' \
308
+ "$id" "$rep" "$verdict" "$cost" "$dur" "$turns" "${firstfail:+ — $firstfail}"
309
+ done
310
+ done
311
+
312
+ if [ "$DRY_RUN" -eq 1 ]; then
313
+ printf '# dry run: %s case blocks printed, no claude call made\n' "${#ORDERED[@]}"
314
+ exit 0
315
+ fi
316
+
317
+ # ---------------------------------------------------------------------------
318
+ # RESULTS.md and summary.json
319
+ # ---------------------------------------------------------------------------
320
+
321
+ EXIT=0
322
+ {
323
+ printf '# Eval results — %s\n\n' "$STAMP"
324
+ printf 'model: %s · reps: %s · cases: %s · total cost: USD %s\n\n' \
325
+ "${MODEL:-default}" "$REPS" "${#ORDERED[@]}" "$TOTAL_COST"
326
+ printf '| case | rep | verdict | cost USD | duration s | turns | first assert failure |\n'
327
+ printf '|---|---|---|---|---|---|---|\n'
328
+ for row in "${ROWS[@]}"; do
329
+ IFS='|' read -r c r v co du tu ff <<< "$row"
330
+ printf '| %s | %s | %s | %s | %s | %s | %s |\n' "$c" "$r" "$v" "$co" "$du" "$tu" "${ff:-—}"
331
+ done
332
+ printf '\n## Case verdicts\n\n'
333
+ printf '| case | passed | of reps | min_pass | verdict |\n|---|---|---|---|---|\n'
334
+ for id in "${ORDERED[@]}"; do
335
+ need="${MINPASS[$id]}"
336
+ [ "$need" -le "$REPS" ] || need="$REPS"
337
+ got="${PASSES[$id]}"
338
+ if [ "$got" -ge "$need" ]; then v=PASS; else v=FAIL; EXIT=1; fi
339
+ printf '| %s | %s | %s | %s | %s |\n' "$id" "$got" "$REPS" "$need" "$v"
340
+ done
341
+ printf '\nWork trees kept at `%s`.\n' "$TMP"
342
+ } > "$RESULTS/RESULTS.md"
343
+
344
+ node - "$RESULTS/summary.json" "$STAMP" "${MODEL:-default}" "$REPS" "$TMP" "$RESULTS" "$TOTAL_COST" "$EXIT" <<'NODE' "${ROWS[@]}"
345
+ const fs = require('fs');
346
+ const [out, stamp, model, reps, tmp, results, cost, exitCode, ...rows] = process.argv.slice(2);
347
+ const byCase = {};
348
+ for (const row of rows) {
349
+ const [id, rep, verdict, c, d, t, ff] = row.split('|');
350
+ (byCase[id] = byCase[id] || []).push({
351
+ rep: Number(rep), pass: verdict === 'PASS',
352
+ cost_usd: c === '-' ? null : Number(c),
353
+ duration_s: d === '-' ? null : Number(d),
354
+ turns: t === '-' ? null : Number(t),
355
+ first_failure: ff || null,
356
+ });
357
+ }
358
+ fs.writeFileSync(out, JSON.stringify({
359
+ stamp, model, reps: Number(reps), results_dir: results, work_dir: tmp,
360
+ total_cost_usd: Number(cost), exit_code: Number(exitCode),
361
+ cases: Object.entries(byCase).map(([id, r]) => ({
362
+ id, passed: r.filter((x) => x.pass).length, reps: r,
363
+ })),
364
+ }, null, 2) + '\n');
365
+ NODE
366
+
367
+ printf '\n%s\n' "$RESULTS/RESULTS.md"
368
+ printf 'total cost USD %s · exit %s\n' "$TOTAL_COST" "$EXIT"
369
+ exit "$EXIT"
@@ -0,0 +1,5 @@
1
+ # PLAN — fixture delivered
2
+
3
+ ## §8 Phases
4
+
5
+ See ROADMAP.md. Current phase: 02.
@@ -0,0 +1,20 @@
1
+ # PROGRESS — fixture delivered
2
+
3
+ <!-- ll-state -->
4
+ phase: 02
5
+ milestones:
6
+ M1: { passes: true, commit: a1b2c3d, accepted_at: 2026-09-08T10:00:00Z }
7
+ M2: { passes: true, commit: b2c3d4e, accepted_at: 2026-09-08T15:30:00Z }
8
+ <!-- /ll-state -->
9
+
10
+ ## Epilogue — phase 01 — 2026-09-07
11
+
12
+ passed: M1 · left: none
13
+ milestones passed 1/1 · questions asked 0 / assumptions 0 / band-1 open 0 · amendments 0 · verification: none
14
+ ▶ Next — `/clear`, then `ll-implement 2`
15
+
16
+ ## Epilogue — phase 02 — 2026-09-08
17
+
18
+ passed: M1, M2 · left: none
19
+ milestones passed 2/2 · questions asked 0 / assumptions 0 / band-1 open 0 · amendments 0 · verification: none
20
+ ▶ Next — `/clear`, then `ll-close`
@@ -0,0 +1,6 @@
1
+ # ROADMAP — fixture delivered
2
+
3
+ | phase | name | depends_on | requirements | state |
4
+ |---|---|---|---|---|
5
+ | 01 | intake | — | REQ-a | DONE (docs/history/v1.0) |
6
+ | 02 | reporting | 01 | REQ-b | DONE (docs/history/v1.0) |
@@ -0,0 +1,3 @@
1
+ # DELIVERY — fixture delivered
2
+
3
+ Delivered on 2026-09-08: phases 01 and 02 closed, every criterion green.
@@ -0,0 +1,13 @@
1
+ # DEC-0001 — retry policy for the provider adapter
2
+
3
+ - Date: 2026-09-10
4
+ - Decided by: the run, under `--auto-decision`
5
+ - status: DECIDED — three retries with exponential backoff [decided by absence — revisable]
6
+
7
+ ## Decision
8
+
9
+ The adapter retries a failed provider call three times with exponential backoff.
10
+
11
+ ## Why
12
+
13
+ The recommended option of the open question; nobody answered before the run reached the stage.
@@ -0,0 +1,13 @@
1
+ # DEC-0002 — the reconciliation window is 24 hours
2
+
3
+ - Date: 2026-09-10
4
+ - Decided by: the owner, in free text
5
+ - status: DECIDED — 24 hours
6
+
7
+ ## Decision
8
+
9
+ The reconciliation job compares a 24-hour window, not a calendar day.
10
+
11
+ ## Why
12
+
13
+ The owner answered the question directly, so no marker is carried here.
@@ -0,0 +1,20 @@
1
+ # PLAN — fixture without ROADMAP
2
+
3
+ Three phases or fewer: the phase table lives inline in §8, there is no ROADMAP.md.
4
+
5
+ ## §7 Execution protocol
6
+
7
+ Session fable/high; contract executor opus/high.
8
+
9
+ ## §8 Phases
10
+
11
+ | phase | name | requirements | state |
12
+ |---|---|---|---|
13
+ | 01 | intake | REQ-a | DONE (docs/history/v1.0) |
14
+ | 02 | reporting | REQ-b | ACTIVE |
15
+
16
+ Current phase: 02.
17
+
18
+ ## §9 Environment
19
+
20
+ `FIXTURE_KEY` (name only, never the value).
@@ -0,0 +1,11 @@
1
+ # PROGRESS — fixture without ROADMAP
2
+
3
+ <!-- ll-state -->
4
+ phase: 02
5
+ milestones:
6
+ M1: { passes: false, reason: "not started" }
7
+ <!-- /ll-state -->
8
+
9
+ ## Phase 02
10
+
11
+ - [2026-09-09T09:00Z] phase 02 opened — reporting
@@ -0,0 +1,5 @@
1
+ # PLAN — fixture whose epilogue asks for the audit
2
+
3
+ ## §8 Phases
4
+
5
+ See ROADMAP.md. Current phase: 01.
@@ -0,0 +1,18 @@
1
+ # PROGRESS — fixture whose epilogue asks for the audit
2
+
3
+ <!-- ll-state -->
4
+ phase: 01
5
+ milestones:
6
+ M1: { passes: true, commit: a1b2c3d, accepted_at: 2026-09-09T10:00:00Z }
7
+ M2: { passes: false, reason: "acceptance red: 1 test fails" }
8
+ <!-- /ll-state -->
9
+
10
+ ## Phase 01
11
+
12
+ - [2026-09-09T09:00Z] phase 01 opened — intake
13
+
14
+ ## Epilogue — phase 01 — 2026-09-09
15
+
16
+ passed: M1 · left: M2 (acceptance red)
17
+ milestones passed 1/2 · questions asked 0 / assumptions 0 / band-1 open 0 · amendments 0 · verification: pending
18
+ ▶ Next — `/clear`, then `ll-verify 01`
@@ -0,0 +1,5 @@
1
+ # ROADMAP — fixture whose epilogue asks for the audit
2
+
3
+ | phase | name | depends_on | requirements | state |
4
+ |---|---|---|---|---|
5
+ | 01 | intake | — | REQ-a | ACTIVE |
@@ -0,0 +1,6 @@
1
+ # PLAN — phase 01
2
+
3
+ ## Milestones
4
+
5
+ M1 — intake · acceptance: `npm test -- intake.test.ts` · tdd: yes
6
+ M2 — reporting · acceptance: `npm test -- report.test.ts` · tdd: yes
@@ -0,0 +1,18 @@
1
+ research todo no docs/research-*/SUMMARY.md
2
+ brainstorm todo no docs/decide/OPENING.md
3
+ decide done PLAN.md + PROGRESS.md ll-state block
4
+ phase-05 done ROADMAP row: DONE (docs/history/v1.0)
5
+ phase-06 done ROADMAP row: DONE (docs/history/v1.0)
6
+ phase-07 half epilogue over a board with M2, M3 not passing
7
+ phase-08 todo ROADMAP row: PLANNED
8
+ verify-05 todo no phases/05/VERIFICATION.md
9
+ verify-06 todo no phases/06/VERIFICATION.md
10
+ verify-07 todo no phases/07/VERIFICATION.md
11
+ verify-08 todo no phases/08/VERIFICATION.md
12
+ close todo no docs/DELIVERY.md
13
+
14
+ 1. phase-07 ll-implement 07 --no-talk
15
+ 2. phase-08 ll-implement 08 --no-talk
16
+ 3. close ll-close --no-talk
17
+
18
+ Dry run: the roteiro above is the plan; nothing was written.
@@ -0,0 +1,2 @@
1
+ Nada encontrado neste repositório: sem pesquisa, OPENING.md nem PLAN.md.
2
+ /ll-auto "<objetivo>" [--research] [--brainstorm]
@@ -0,0 +1,29 @@
1
+ docs/GOAL.md written and committed (mode: autonomous, phase: all). Paste this text into the
2
+ unattended run:
3
+
4
+ /goal Deliver what ROADMAP.md still owes — phases 07 and 08 — to main, tested, per PLAN.md.
5
+
6
+ DONE WHEN: docs/DELIVERY.md exists; PROGRESS.md carries ## Epilogue — phase 08; phases/07 and
7
+ phases/08 VERIFICATION.md say APPROVED or APPROVED_WITH_RESERVATIONS; docs/AUTO.md has every roteiro
8
+ row done or skipped and ## Decisions taken alone filled; `npm test` exit 0; `git status --porcelain`
9
+ empty — each pasted here as its last output line. Or stop after 80 turns, saying what is missing.
10
+
11
+ READ FIRST: PLAN.md (§0, §2, §3, §7), ROADMAP.md phases 07 and 08, docs/AUTO.md when it exists.
12
+
13
+ INVALIDATING: deleting, weakening, skipping or marking as skip any test, acceptance or threshold;
14
+ self-validation instead of a clean-context verifier; a TODO or "phase 2" left in a milestone marked
15
+ done; partial delivery counted as done. Impossibility without an implemented alternative does not
16
+ complete the goal.
17
+
18
+ EXECUTION: skill ll-auto --auto-decision --verify all, started again from this text whenever the
19
+ session stops before DONE WHEN. Models per role from PLAN.md §7.
20
+
21
+ DECISIONS: everything that stays inside the repository is decided, recorded [decided by absence —
22
+ revisable] and listed at the end; money, production data and credentials never proceed alone.
23
+
24
+ STATE: docs/AUTO.md one log line per stage; PROGRESS.md one block per wave.
25
+
26
+ STOP: external block (key, host, rate limit, an answer only the owner has) = record the state in
27
+ PROGRESS.md and STOP without marking done.
28
+
29
+ ▶ Next — `/clear` then paste the text above into the unattended run.
@@ -0,0 +1,13 @@
1
+ ---
2
+ name: ll-fake
3
+ description: >
4
+ Pretends to describe a skill in a folded scalar.
5
+
6
+ The blank line above keeps a newline inside the value, which rule 1 must reject.
7
+ argument-hint: "[none]"
8
+ disable-model-invocation: true
9
+ ---
10
+
11
+ # Folded description
12
+
13
+ Fixture for lint-prompts rule 1: the description is not one line.
@@ -0,0 +1,36 @@
1
+ ---
2
+ name: ll-fake
3
+ description: Prints the owner's questions with the internal vocabulary still in them, so lint rule 9 has something to catch.
4
+ argument-hint: "[--no-talk]"
5
+ disable-model-invocation: true
6
+ ---
7
+
8
+ # ll-fake
9
+
10
+ ## Flow
11
+
12
+ 1. Print the round:
13
+
14
+ **Pergunta 1/2 — limite de título [DEC-0001]** (impacto ALTO · desfazer: barato)
15
+
16
+ **Question 2/2 — retenção [ASM-3]** (impact LOW)
17
+
18
+ **[PG-1] Pergunta 1/5 — janela de retenção** (impacto ALTO · desfazer: caro)
19
+
20
+ 2. Print the counter: `questions asked 2 / assumptions 1 / band-1 open 1`
21
+
22
+ 3. The Portuguese line: `perguntas 2 / assunções 1 · decisões [D-07-01] em aberto 1`
23
+
24
+ 4. Print the label the owner never reads: esta é uma banda 1, só o dono decide.
25
+
26
+ ## Deliverables
27
+
28
+ | File | Role | Mutability |
29
+ | --- | --- | --- |
30
+ | `docs/fake.md` | nothing | rewritten |
31
+
32
+ ## Completion criterion
33
+
34
+ Done when the file exists.
35
+
36
+ ▶ Next — `/clear`, then `/ll-resume`
@@ -0,0 +1,10 @@
1
+ ---
2
+ name: ll-fake
3
+ description: Pretends to be a manual skill while the frontmatter leaves the model free to invoke it, which rule 1 must reject.
4
+ argument-hint: "[none]"
5
+ disable-model-invocation: false
6
+ ---
7
+
8
+ # Model invocation left open
9
+
10
+ Fixture for lint-prompts rule 1: disable-model-invocation is declared false.
@@ -0,0 +1,30 @@
1
+ ---
2
+ name: ll-bad
3
+ description: Fixture for lint-contract rule 6 — handoff lines that the Next grammar must reject, one break per line.
4
+ argument-hint: "[none]"
5
+ disable-model-invocation: true
6
+ ---
7
+
8
+ # Bad handoffs
9
+
10
+ Each line below breaks the grammar in one way, so rule 6 must print one FAIL for each.
11
+
12
+ Old shape, backticks around /clear and around a skill that does exist:
13
+
14
+ ▶ Next — `/clear` then `ll-bad`
15
+
16
+ Bare command, no `/clear, then` opening:
17
+
18
+ ▶ Next — ll-bad
19
+
20
+ Right grammar, command that names no skill in this tree:
21
+
22
+ ▶ Next — /clear, then ll-nope
23
+
24
+ Right grammar, but the same skill twice outside a parenthetical:
25
+
26
+ ▶ Next — /clear, then ll-bad or ll-bad --resume
27
+
28
+ Same grammar, two different skills to choose between, still outside a parenthetical:
29
+
30
+ ▶ Next — /clear, then ll-bad or ll-nope --resume