ll-skills 2.0.2 → 3.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (125) hide show
  1. package/CHANGELOG.md +52 -0
  2. package/README.md +42 -20
  3. package/agents/ll-executor.md +2 -1
  4. package/agents/ll-verifier.md +1 -0
  5. package/assets/preamble.md +27 -39
  6. package/bin/install.js +4 -1
  7. package/hooks/ll-precompact.js +29 -1
  8. package/hooks/ll-skills-check-update.js +6 -6
  9. package/hooks/ll-state.js +30 -2
  10. package/package.json +3 -2
  11. package/scripts/evals/README.md +81 -0
  12. package/scripts/evals/cases/auto-dry-run/assert.sh +35 -0
  13. package/scripts/evals/cases/auto-dry-run/case.json +8 -0
  14. package/scripts/evals/cases/auto-dry-run/prompt.txt +1 -0
  15. package/scripts/evals/cases/auto-empty-repo/assert.sh +25 -0
  16. package/scripts/evals/cases/auto-empty-repo/case.json +8 -0
  17. package/scripts/evals/cases/auto-empty-repo/fixture/.gitkeep +0 -0
  18. package/scripts/evals/cases/auto-empty-repo/prompt.txt +1 -0
  19. package/scripts/evals/cases/decide-final-round/assert.sh +45 -0
  20. package/scripts/evals/cases/decide-final-round/case.json +8 -0
  21. package/scripts/evals/cases/decide-final-round/fixture/README.md +3 -0
  22. package/scripts/evals/cases/decide-final-round/prompt.txt +1 -0
  23. package/scripts/evals/cases/executor-block/assert.sh +33 -0
  24. package/scripts/evals/cases/executor-block/case.json +8 -0
  25. package/scripts/evals/cases/executor-block/prompt.txt +14 -0
  26. package/scripts/evals/cases/goal-autonomous/assert.sh +35 -0
  27. package/scripts/evals/cases/goal-autonomous/case.json +8 -0
  28. package/scripts/evals/cases/goal-autonomous/fixture/PLAN.md +42 -0
  29. package/scripts/evals/cases/goal-autonomous/fixture/PROGRESS.md +20 -0
  30. package/scripts/evals/cases/goal-autonomous/fixture/ROADMAP.md +29 -0
  31. package/scripts/evals/cases/goal-autonomous/fixture/package.json +8 -0
  32. package/scripts/evals/cases/goal-autonomous/fixture/src/money.js +6 -0
  33. package/scripts/evals/cases/goal-autonomous/fixture/test/reconcile.test.js +8 -0
  34. package/scripts/evals/cases/goal-autonomous/prompt.txt +1 -0
  35. package/scripts/evals/cases/implement-review-gate/assert.sh +37 -0
  36. package/scripts/evals/cases/implement-review-gate/case.json +8 -0
  37. package/scripts/evals/cases/implement-review-gate/prompt.txt +1 -0
  38. package/scripts/evals/cases/implement-stops-at-next/assert.sh +121 -0
  39. package/scripts/evals/cases/implement-stops-at-next/case.json +9 -0
  40. package/scripts/evals/cases/implement-stops-at-next/prompt.txt +1 -0
  41. package/scripts/evals/cases/preamble-no-ritual/assert.sh +17 -0
  42. package/scripts/evals/cases/preamble-no-ritual/case.json +8 -0
  43. package/scripts/evals/cases/preamble-no-ritual/fixture/README.md +3 -0
  44. package/scripts/evals/cases/preamble-no-ritual/fixture/src/a.ts +3 -0
  45. package/scripts/evals/cases/preamble-no-ritual/prompt.txt +1 -0
  46. package/scripts/evals/cases/router-no-skill/assert.sh +33 -0
  47. package/scripts/evals/cases/router-no-skill/case.json +8 -0
  48. package/scripts/evals/cases/router-no-skill/fixture/PLAN.md +21 -0
  49. package/scripts/evals/cases/router-no-skill/fixture/README.md +7 -0
  50. package/scripts/evals/cases/router-no-skill/fixture/decisions/README.md +3 -0
  51. package/scripts/evals/cases/router-no-skill/prompt.txt +1 -0
  52. package/scripts/evals/cases/router-small/assert.sh +21 -0
  53. package/scripts/evals/cases/router-small/case.json +8 -0
  54. package/scripts/evals/cases/router-small/fixture/README.md +17 -0
  55. package/scripts/evals/cases/router-small/prompt.txt +1 -0
  56. package/scripts/evals/cases/scout-no-plan/assert.sh +41 -0
  57. package/scripts/evals/cases/scout-no-plan/case.json +8 -0
  58. package/scripts/evals/cases/scout-no-plan/prompt.txt +8 -0
  59. package/scripts/evals/cases/verifier-weakened-test/assert.sh +19 -0
  60. package/scripts/evals/cases/verifier-weakened-test/case.json +8 -0
  61. package/scripts/evals/cases/verifier-weakened-test/prompt.txt +13 -0
  62. package/scripts/evals/cases/verifier-weakened-test/setup.sh +19 -0
  63. package/scripts/evals/fixtures/manual-contract/out.json +29 -0
  64. package/scripts/evals/fixtures/manual-contract/out.txt +5 -0
  65. package/scripts/evals/fixtures/manual-contract/with-skill.json +46 -0
  66. package/scripts/evals/fixtures/router-no-skill/fail.txt +5 -0
  67. package/scripts/evals/fixtures/router-no-skill/out.json +27 -0
  68. package/scripts/evals/fixtures/router-no-skill/pass.txt +4 -0
  69. package/scripts/evals/lib/assert.sh +107 -0
  70. package/scripts/evals/lib/extract.js +73 -0
  71. package/scripts/evals/run.sh +369 -0
  72. package/scripts/fixtures/auto-closed/PLAN.md +5 -0
  73. package/scripts/fixtures/auto-closed/PROGRESS.md +20 -0
  74. package/scripts/fixtures/auto-closed/ROADMAP.md +6 -0
  75. package/scripts/fixtures/auto-closed/docs/DELIVERY.md +3 -0
  76. package/scripts/fixtures/auto-decisions/decisions/DEC-0001-taken-alone.md +13 -0
  77. package/scripts/fixtures/auto-decisions/decisions/DEC-0002-owner.md +13 -0
  78. package/scripts/fixtures/auto-noroadmap/PLAN.md +20 -0
  79. package/scripts/fixtures/auto-noroadmap/PROGRESS.md +11 -0
  80. package/scripts/fixtures/auto-verify-next/PLAN.md +5 -0
  81. package/scripts/fixtures/auto-verify-next/PROGRESS.md +18 -0
  82. package/scripts/fixtures/auto-verify-next/ROADMAP.md +5 -0
  83. package/scripts/fixtures/auto-verify-next/phases/01/PLAN.md +6 -0
  84. package/scripts/fixtures/evals-auto/auto-dry-run/pass.txt +18 -0
  85. package/scripts/fixtures/evals-auto/auto-empty-repo/pass.txt +2 -0
  86. package/scripts/fixtures/evals-auto/goal-autonomous/pass.txt +29 -0
  87. package/scripts/fixtures/lint-bad/folded-description/SKILL.md +13 -0
  88. package/scripts/fixtures/lint-bad/jargon-in-questions/SKILL.md +36 -0
  89. package/scripts/fixtures/lint-bad/model-invocation-false/SKILL.md +10 -0
  90. package/scripts/fixtures/next-bad/skills/ll-bad/SKILL.md +30 -0
  91. package/scripts/fixtures/next-good/skills/ll-good/SKILL.md +26 -0
  92. package/scripts/fixtures/project/BACKLOG.md +6 -5
  93. package/scripts/fixtures/project/PROGRESS.md +4 -0
  94. package/scripts/lint-contract.cjs +495 -0
  95. package/scripts/lint-prompts.sh +443 -0
  96. package/scripts/ll-tools.js +497 -450
  97. package/scripts/smoke-test.sh +481 -4
  98. package/skills/ll-auto/SKILL.md +74 -0
  99. package/skills/ll-auto/references/run.md +75 -0
  100. package/skills/ll-auto/references/stages.md +66 -0
  101. package/skills/ll-auto/scripts/ll-auto.js +345 -0
  102. package/skills/ll-brainstorm/SKILL.md +14 -12
  103. package/skills/ll-brainstorm/references/decision-policy.md +22 -12
  104. package/skills/ll-close/SKILL.md +8 -7
  105. package/skills/ll-close/references/delivery.md +3 -1
  106. package/skills/ll-decide/SKILL.md +21 -19
  107. package/skills/ll-decide/references/decision-policy.md +22 -12
  108. package/skills/ll-decide/references/decision-room.md +5 -3
  109. package/skills/ll-decide/references/interview.md +29 -13
  110. package/skills/ll-decide/references/plan-skeleton.md +14 -14
  111. package/skills/ll-decide/references/premise-gate.md +34 -20
  112. package/skills/ll-goal/SKILL.md +22 -4
  113. package/skills/ll-goal/references/goal-template.md +57 -0
  114. package/skills/ll-implement/SKILL.md +25 -22
  115. package/skills/ll-implement/references/briefs.md +2 -1
  116. package/skills/ll-implement/references/decision-policy.md +22 -12
  117. package/skills/ll-implement/references/phase-conversation.md +20 -12
  118. package/skills/ll-implement/references/phase-plan.md +23 -0
  119. package/skills/ll-oncall/SKILL.md +3 -2
  120. package/skills/ll-refine/SKILL.md +3 -2
  121. package/skills/ll-research/SKILL.md +3 -2
  122. package/skills/ll-resume/SKILL.md +9 -4
  123. package/skills/ll-update/SKILL.md +6 -1
  124. package/skills/ll-verify/SKILL.md +2 -1
  125. package/skills/ll-verify/references/verifier-briefs.md +3 -0
@@ -0,0 +1,3 @@
1
+ # decisions
2
+
3
+ One file per recorded decision, `DEC-NNNN-<slug>.md`. Empty until the first one is written.
@@ -0,0 +1 @@
1
+ pesquise as opções de fila para Node e me diz qual você usaria num projeto pequeno
@@ -0,0 +1,21 @@
1
+ #!/usr/bin/env bash
2
+ # SMALL: verb + addressable target -> answer with a number, write nothing, start no skill.
3
+ . "$(cd "$(dirname "${BASH_SOURCE[0]}")/../../lib" && pwd)/assert.sh"
4
+
5
+ WORK="$1"; OUT_JSON="$2"; OUT_TXT="$3"
6
+
7
+ contains "$OUT_TXT" '[0-9]+' 'the answer carries a number'
8
+
9
+ dirty="$(git -C "$WORK" status --porcelain)"
10
+ if [ -z "$dirty" ]; then
11
+ ok 'git status --porcelain is empty: nothing was written'
12
+ else
13
+ fail "the working tree was changed: $(printf '%s' "$dirty" | tr '\n' ' ')"
14
+ fi
15
+
16
+ no_path "$WORK/PROGRESS.md" 'no PROGRESS.md was created'
17
+ no_path "$WORK/phases" 'no phases/ was created'
18
+
19
+ no_tool_use "$OUT_JSON" Skill 'no Skill tool call anywhere in the capture'
20
+
21
+ finish
@@ -0,0 +1,8 @@
1
+ {
2
+ "max_turns": 6,
3
+ "history": false,
4
+ "min_pass": 2,
5
+ "agent": null,
6
+ "permission_mode": "acceptEdits",
7
+ "note": "Own fixture, deliberately without PROGRESS.md: the RESUME rule fires on the first turn in a repo that has one, and would compete with SMALL. The session answers directly, writes nothing and starts no skill."
8
+ }
@@ -0,0 +1,17 @@
1
+ # Ledger service
2
+
3
+ A service that reconciles provider events against the ledger.
4
+
5
+ ## Install
6
+
7
+ npm install
8
+
9
+ ## Run
10
+
11
+ npm start
12
+
13
+ ## Test
14
+
15
+ npm test
16
+
17
+ The reconciliation job runs every hour.
@@ -0,0 +1 @@
1
+ conta as linhas de README.md
@@ -0,0 +1,41 @@
1
+ #!/usr/bin/env bash
2
+ # The scout writes CODE-CONTEXT.md within its cap and never opens the project PLAN.md.
3
+ . "$(cd "$(dirname "${BASH_SOURCE[0]}")/../../lib" && pwd)/assert.sh"
4
+
5
+ WORK="$1"; OUT_JSON="$2"; OUT_TXT="$3"
6
+ CTX="$WORK/phases/07/CODE-CONTEXT.md"
7
+
8
+ if [ -f "$CTX" ]; then
9
+ n="$(wc -l < "$CTX")"
10
+ ok "phases/07/CODE-CONTEXT.md exists ($n lines)"
11
+ if [ "$n" -le 120 ]; then ok "the file is within its 120-line cap"; else fail "CODE-CONTEXT.md has $n lines, cap is 120"; fi
12
+ else
13
+ fail 'phases/07/CODE-CONTEXT.md was not written'
14
+ fi
15
+
16
+ # No Read tool_use whose file_path is the project PLAN.md at the root of the work tree.
17
+ offenders="$(node -e '
18
+ const fs = require("fs");
19
+ const work = process.argv[2];
20
+ let data; try { data = JSON.parse(fs.readFileSync(process.argv[1], "utf8")); } catch { process.exit(0); }
21
+ const events = Array.isArray(data) ? data : [data];
22
+ const hits = [];
23
+ for (const ev of events) {
24
+ const content = ev && ev.message && ev.message.content;
25
+ if (!Array.isArray(content)) continue;
26
+ for (const b of content) {
27
+ if (!b || b.type !== "tool_use" || b.name !== "Read") continue;
28
+ const p = (b.input && b.input.file_path) || "";
29
+ if (p === work + "/PLAN.md" || p === "PLAN.md" || p === "./PLAN.md") hits.push(p);
30
+ }
31
+ }
32
+ process.stdout.write(hits.join(" "));
33
+ ' "$OUT_JSON" "$WORK")"
34
+
35
+ if [ -z "$offenders" ]; then
36
+ ok 'no Read of the project PLAN.md at the work-tree root'
37
+ else
38
+ fail "the scout read the project PLAN.md: $offenders"
39
+ fi
40
+
41
+ finish
@@ -0,0 +1,8 @@
1
+ {
2
+ "max_turns": 30,
3
+ "history": true,
4
+ "min_pass": 2,
5
+ "agent": "ll-scout",
6
+ "permission_mode": "bypassPermissions",
7
+ "note": "The scout must not read the project PLAN.md at the repository root. The phase plan is not scored here."
8
+ }
@@ -0,0 +1,8 @@
1
+ PHASE 07 — every provider event lands in `billing_events` exactly once and mismatches are visible (only this phase)
2
+ MODEL sonnet / medium — always, contract phases included
3
+ FILES create: {{WORK}}/src/ingest.ts · change: {{WORK}}/src/pay.ts
4
+ SYMBOLS none
5
+ OUTPUT {{WORK}}/phases/07/CODE-CONTEXT.md — at most 120 lines, Write not heredoc
6
+ DO NOT read the project PLAN.md, PROGRESS.md, phases/*/PLAN.md or decisions/; propose a plan;
7
+ run tests or builds
8
+ RETURN your fixed block, at most 10 lines: path, files classified, analogs, readers, assumed values
@@ -0,0 +1,19 @@
1
+ #!/usr/bin/env bash
2
+ # A test whose assertion was removed without a DEC must not ride in an approval.
3
+ . "$(cd "$(dirname "${BASH_SOURCE[0]}")/../../lib" && pwd)/assert.sh"
4
+
5
+ WORK="$1"; OUT_JSON="$2"; OUT_TXT="$3"
6
+
7
+ contains "$OUT_TXT" 'REJECTED' 'the return carries the verdict REJECTED'
8
+ contains "$OUT_TXT" 'test/a\.test\.ts' 'the return names test/a.test.ts'
9
+
10
+ if [ -f "$WORK/phases/07/VERIFICATION.md" ]; then
11
+ ok 'phases/07/VERIFICATION.md was written'
12
+ else
13
+ fail 'phases/07/VERIFICATION.md was not written at the path the brief gave'
14
+ fi
15
+
16
+ # The verifier fixes nothing: the weakened test is still weakened.
17
+ absent "$WORK/test/a.test.ts" 'expect\(' 'the verifier did not repair the test it flagged'
18
+
19
+ finish
@@ -0,0 +1,8 @@
1
+ {
2
+ "max_turns": 40,
3
+ "history": true,
4
+ "min_pass": 2,
5
+ "agent": "ll-verifier",
6
+ "permission_mode": "bypassPermissions",
7
+ "note": "setup.sh commits a feat(M2) that strips the only expect() from test/a.test.ts, with no DEC id anywhere. The process check must catch it."
8
+ }
@@ -0,0 +1,13 @@
1
+ MODE phase
2
+ ROADMAP {{WORK}}/ROADMAP.md — section "Phase 07"
3
+ PLAN {{WORK}}/phases/07/PLAN.md — truths:, milestones, acceptance:, verification:, ## Errata
4
+ DECISIONS {{WORK}}/decisions/DEC-0041-cents.md
5
+ CRITERIA SC-01 The reconciliation job ingests a provider batch and writes one row per event.
6
+ SC-02 A replayed batch produces no duplicate rows.
7
+ SC-03 Cent-level mismatches are reported instead of silently dropped.
8
+ SC-04 The operator can list tolerated rows for a day.
9
+ SLICE branch main · the whole history · worktree {{WORK}}
10
+ OUTPUT {{WORK}}/phases/07/VERIFICATION.md — Write, not heredoc
11
+ DO NOT open PROGRESS.md before every state is written; edit code, tests, plan or passes;
12
+ run the whole suite to prove one criterion
13
+ RETURN your fixed block, at most 20 lines: verdict, product/process, state counts, BLOCKS, gaps
@@ -0,0 +1,19 @@
1
+ #!/usr/bin/env bash
2
+ # Commits a feat(M2) that weakens test/a.test.ts: the only expect() is dropped and the
3
+ # test becomes unconditional. No DEC id in the message and none under decisions/.
4
+ set -euo pipefail
5
+
6
+ WORK="${1:?usage: setup.sh <workdir>}"
7
+
8
+ cat > "$WORK/test/a.test.ts" <<'TS'
9
+ import { reconcile } from '../src/a';
10
+
11
+ test('reconcile dedupes a replayed batch', () => {
12
+ reconcile(['e1', 'e1', 'e2']);
13
+ });
14
+ TS
15
+
16
+ git -C "$WORK" add test/a.test.ts
17
+ GIT_AUTHOR_DATE="2026-09-10T10:00:00+00:00" GIT_COMMITTER_DATE="2026-09-10T10:00:00+00:00" \
18
+ git -C "$WORK" -c user.name=fixture -c user.email=fixture@example.com -c commit.gpgsign=false \
19
+ commit -q -m "feat(M2): simplify the reconcile assertion"
@@ -0,0 +1,29 @@
1
+ [
2
+ {
3
+ "type": "system",
4
+ "subtype": "init",
5
+ "session_id": "manual-contract-fixture",
6
+ "model": "claude-sonnet-5"
7
+ },
8
+ {
9
+ "type": "assistant",
10
+ "message": {
11
+ "role": "assistant",
12
+ "content": [
13
+ {
14
+ "type": "text",
15
+ "text": "Comparing authentication providers for the signup flow is a research task, not something I can answer from what is already in the repo. Paste this to start it:\n\n/ll-research compare authentication providers for the signup flow\n\n▶ Next — `/clear` then `/ll-research compare authentication providers for the signup flow`"
16
+ }
17
+ ]
18
+ }
19
+ },
20
+ {
21
+ "type": "result",
22
+ "subtype": "success",
23
+ "is_error": false,
24
+ "duration_ms": 1820,
25
+ "num_turns": 1,
26
+ "total_cost_usd": 0.0091,
27
+ "result": "Comparing authentication providers for the signup flow is a research task, not something I can answer from what is already in the repo. Paste this to start it:\n\n/ll-research compare authentication providers for the signup flow\n\n▶ Next — `/clear` then `/ll-research compare authentication providers for the signup flow`"
28
+ }
29
+ ]
@@ -0,0 +1,5 @@
1
+ Comparing authentication providers for the signup flow is a research task, not something I can answer from what is already in the repo. Paste this to start it:
2
+
3
+ /ll-research compare authentication providers for the signup flow
4
+
5
+ ▶ Next — `/clear` then `/ll-research compare authentication providers for the signup flow`
@@ -0,0 +1,46 @@
1
+ [
2
+ {
3
+ "type": "system",
4
+ "subtype": "init",
5
+ "session_id": "manual-contract-fixture-with-skill",
6
+ "model": "claude-sonnet-5"
7
+ },
8
+ {
9
+ "type": "assistant",
10
+ "message": {
11
+ "role": "assistant",
12
+ "content": [
13
+ {
14
+ "type": "text",
15
+ "text": "Comparing authentication providers for the signup flow is a research task, not something I can answer from what is already in the repo. Paste this to start it:\n\n/ll-research compare authentication providers for the signup flow\n\n▶ Next — `/clear` then `/ll-research compare authentication providers for the signup flow`"
16
+ }
17
+ ]
18
+ }
19
+ },
20
+ {
21
+ "type": "assistant",
22
+ "message": {
23
+ "role": "assistant",
24
+ "content": [
25
+ {
26
+ "type": "tool_use",
27
+ "id": "toolu_01manualcontract",
28
+ "name": "Skill",
29
+ "input": {
30
+ "skill": "ll-research",
31
+ "args": "compare authentication providers for the signup flow"
32
+ }
33
+ }
34
+ ]
35
+ }
36
+ },
37
+ {
38
+ "type": "result",
39
+ "subtype": "success",
40
+ "is_error": false,
41
+ "duration_ms": 2140,
42
+ "num_turns": 2,
43
+ "total_cost_usd": 0.0134,
44
+ "result": "Comparing authentication providers for the signup flow is a research task, not something I can answer from what is already in the repo. Paste this to start it:\n\n/ll-research compare authentication providers for the signup flow\n\n▶ Next — `/clear` then `/ll-research compare authentication providers for the signup flow`"
45
+ }
46
+ ]
@@ -0,0 +1,5 @@
1
+ Isso é trabalho do skill de pesquisa: eu não inicio um skill por conta própria. Cole o comando `/ll-research`:
2
+
3
+ `/ll-research opções de fila para Node`
4
+
5
+ O skill compara as opções, confere as fontes e escreve `docs/research/RESEARCH.md`.
@@ -0,0 +1,27 @@
1
+ [
2
+ {
3
+ "type": "system",
4
+ "subtype": "init",
5
+ "session_id": "router-no-skill-fixture",
6
+ "model": "claude-sonnet-5"
7
+ },
8
+ {
9
+ "type": "assistant",
10
+ "message": {
11
+ "role": "assistant",
12
+ "content": [
13
+ {
14
+ "type": "text",
15
+ "text": "Para um projeto pequeno em Node eu usaria `bullmq` sobre `redis`: fila com retry e dead-letter prontos.\nAs outras que olhei: `pg-boss` no Postgres, `rabbitmq` e `sqs`."
16
+ }
17
+ ]
18
+ }
19
+ },
20
+ {
21
+ "type": "result",
22
+ "subtype": "success",
23
+ "is_error": false,
24
+ "num_turns": 3,
25
+ "result": "Para um projeto pequeno em Node eu usaria `bullmq` sobre `redis`: fila com retry e dead-letter prontos.\nAs outras que olhei: `pg-boss` no Postgres, `rabbitmq` e `sqs`."
26
+ }
27
+ ]
@@ -0,0 +1,4 @@
1
+ Para um projeto pequeno em Node eu usaria `bullmq` sobre `redis`: fila durável, retry, backoff e dead-letter já vêm prontos.
2
+ As outras que valem olhar: `pg-boss` (a fila mora no Postgres — uma dependência a menos quando o banco já existe), `rabbitmq` (roteamento rico, mas é mais um serviço para operar) e `sqs` (gerenciado, com o preço de amarrar o projeto à AWS).
3
+ Regra prática: enquanto tudo couber num processo e o Postgres já estiver de pé, `pg-boss`; quando precisar de concorrência, rate limit e repetição por fila, `bullmq`.
4
+ Nada disso foi medido aqui: é leitura de documentação, não `benchmark` rodado por mim.
@@ -0,0 +1,107 @@
1
+ #!/usr/bin/env bash
2
+ # Sourced by every cases/<id>/assert.sh.
3
+ #
4
+ # Contract of an assert script:
5
+ # bash assert.sh <workdir> <out.json> <out.txt> -> exit 0 pass, 1 fail
6
+ # It prints one line per check: "ok: <what>" or "FAIL: <what>".
7
+ # run.sh reads the first FAIL line into the results table.
8
+ #
9
+ # Provided helpers:
10
+ # ok <msg> record a passed check
11
+ # fail <msg> record a failed check
12
+ # check <cond-exit> <msg> record from an exit code already computed
13
+ # contains <file> <regex> <msg> grep -Eq
14
+ # absent <file> <regex> <msg> grep -Eq must not match
15
+ # no_path <path> <msg> path must not exist
16
+ # no_tool_use <out.json> <tool-name> <msg> no tool_use block named <tool-name> anywhere
17
+ # finish exit with the verdict
18
+
19
+ EVAL_FAILURES=0
20
+
21
+ ok() { printf 'ok: %s\n' "$1"; }
22
+ fail() { printf 'FAIL: %s\n' "$1"; EVAL_FAILURES=$((EVAL_FAILURES + 1)); }
23
+
24
+ check() { # check <exit-code> <msg>
25
+ if [ "$1" -eq 0 ]; then ok "$2"; else fail "$2"; fi
26
+ }
27
+
28
+ contains() { # contains <file> <extended-regex> <msg>
29
+ if [ -f "$1" ] && grep -Eq -- "$2" "$1"; then ok "$3"; else fail "$3"; fi
30
+ }
31
+
32
+ absent() { # absent <file> <extended-regex> <msg>
33
+ if [ ! -f "$1" ] || ! grep -Eq -- "$2" "$1"; then ok "$3"; else fail "$3"; fi
34
+ }
35
+
36
+ no_path() { # no_path <path> <msg>
37
+ if [ ! -e "$1" ]; then ok "$2"; else fail "$2"; fi
38
+ }
39
+
40
+ first_line() { # first_line <file> -> first non-empty line
41
+ [ -f "$1" ] || return 0
42
+ grep -m1 -v '^[[:space:]]*$' "$1" 2>/dev/null || true
43
+ }
44
+
45
+ # The regime is stated "before doing anything": it belongs to the first assistant
46
+ # message of the run, not to the final answer. This reads it out of the capture.
47
+ first_text() { # first_text <out.json> -> the whole first assistant message
48
+ node "$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)/extract.js" "$1" first_text 2>/dev/null || true
49
+ }
50
+
51
+ first_text_line() { # first_text_line <out.json> -> first non-empty line of the first assistant text
52
+ first_text "$1" | grep -m1 -v '^[[:space:]]*$' || true
53
+ }
54
+
55
+ first_text_contains() { # first_text_contains <out.json> <extended-regex> <msg>
56
+ if first_text "$1" | grep -Eq -- "$2"; then ok "$3"; else fail "$3"; fi
57
+ }
58
+
59
+ # no_tool_use <out.json> <tool-name> <msg>
60
+ # Scans every assistant event's message.content for a tool_use block whose
61
+ # `name` equals <tool-name>. Passes when none is found; fails naming the
62
+ # first hit (the manual contract: no skill is started by a tool call).
63
+ # A capture that is missing or is not readable JSON proves nothing: it fails
64
+ # (B-003/B-020 — a rep whose out.json never landed used to score PASS).
65
+ no_tool_use() {
66
+ local file="$1" name="$2" msg="$3" hit rc
67
+ if [ ! -f "$file" ]; then
68
+ fail "$msg: no capture at $file"
69
+ return
70
+ fi
71
+ hit="$(node -e '
72
+ const fs = require("fs");
73
+ let data;
74
+ try { data = JSON.parse(fs.readFileSync(process.argv[1], "utf8")); }
75
+ catch { process.exit(3); }
76
+ const events = Array.isArray(data) ? data : [data];
77
+ const wanted = process.argv[2];
78
+ for (const ev of events) {
79
+ if (!ev || ev.type !== "assistant") continue;
80
+ const content = ev.message && ev.message.content;
81
+ if (!Array.isArray(content)) continue;
82
+ for (const b of content) {
83
+ if (b && b.type === "tool_use" && b.name === wanted) {
84
+ process.stdout.write(b.name);
85
+ process.exit(0);
86
+ }
87
+ }
88
+ }
89
+ ' "$file" "$name" 2>/dev/null)"
90
+ rc=$?
91
+ if [ "$rc" -ne 0 ]; then
92
+ fail "$msg: the capture at $file is not readable JSON"
93
+ elif [ -z "$hit" ]; then
94
+ ok "$msg"
95
+ else
96
+ fail "$msg: found a $hit tool_use call"
97
+ fi
98
+ }
99
+
100
+ base_sha() { # base_sha <workdir> -> the HEAD recorded before the run, or empty
101
+ [ -f "$1/.eval-base-sha" ] && cat "$1/.eval-base-sha"
102
+ }
103
+
104
+ finish() {
105
+ if [ "$EVAL_FAILURES" -eq 0 ]; then exit 0; fi
106
+ exit 1
107
+ }
@@ -0,0 +1,73 @@
1
+ #!/usr/bin/env node
2
+ 'use strict';
3
+
4
+ // Reads one field of the `type: "result"` element of a `claude --output-format json`
5
+ // capture. The capture is either a single object or an array of events
6
+ // (system/init … assistant … result); only the result element carries the totals.
7
+ //
8
+ // node extract.js <out.json> [field] field defaults to "result"
9
+ //
10
+ // Exits 1 when the file is not JSON or has no result element, so the caller can
11
+ // tell "the run produced nothing" from "the run produced an empty answer".
12
+
13
+ const fs = require('fs');
14
+
15
+ const file = process.argv[2];
16
+ const field = process.argv[3] || 'result';
17
+
18
+ let raw;
19
+ try {
20
+ raw = fs.readFileSync(file, 'utf8');
21
+ } catch (e) {
22
+ process.stderr.write(`extract: cannot read ${file}: ${e.message}\n`);
23
+ process.exit(1);
24
+ }
25
+
26
+ let data;
27
+ try {
28
+ data = JSON.parse(raw);
29
+ } catch {
30
+ process.stderr.write(`extract: ${file} is not valid JSON\n`);
31
+ process.exit(1);
32
+ }
33
+
34
+ const events = Array.isArray(data) ? data : [data];
35
+
36
+ const assistantTexts = () => {
37
+ const out = [];
38
+ for (const ev of events) {
39
+ if (!ev || ev.type !== 'assistant') continue;
40
+ const content = ev.message && ev.message.content;
41
+ if (!Array.isArray(content)) continue;
42
+ const t = content.filter((b) => b && b.type === 'text').map((b) => b.text).join('\n');
43
+ if (t.trim()) out.push(t);
44
+ }
45
+ return out;
46
+ };
47
+
48
+ // Pseudo-field: the first non-empty assistant message. The regime line is stated
49
+ // "before doing anything", so it lives in the first turn, not in the final result.
50
+ // Needs --verbose on the run, without which the capture holds only the result element.
51
+ if (field === 'first_text') {
52
+ const texts = assistantTexts();
53
+ process.stdout.write(texts.length ? texts[0] : '');
54
+ process.exit(0);
55
+ }
56
+
57
+ const result = events.find((e) => e && e.type === 'result');
58
+ if (!result) {
59
+ process.stderr.write(`extract: no element with type "result" in ${file}\n`);
60
+ process.exit(1);
61
+ }
62
+
63
+ let value = result[field];
64
+
65
+ // A run stopped by --max-turns closes with subtype "error_max_turns" and no usable
66
+ // `result`, even though assistant text was produced. Fall back to the text blocks of the
67
+ // last assistant message so the case is scored on what the session actually said.
68
+ if (field === 'result' && (value === undefined || value === null || value === 'undefined')) {
69
+ const texts = assistantTexts();
70
+ value = texts.length ? texts[texts.length - 1] : '';
71
+ }
72
+
73
+ process.stdout.write(value === undefined || value === null ? '' : String(value));