ll-skills 2.0.2 → 3.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (125) hide show
  1. package/CHANGELOG.md +52 -0
  2. package/README.md +42 -20
  3. package/agents/ll-executor.md +2 -1
  4. package/agents/ll-verifier.md +1 -0
  5. package/assets/preamble.md +27 -39
  6. package/bin/install.js +4 -1
  7. package/hooks/ll-precompact.js +29 -1
  8. package/hooks/ll-skills-check-update.js +6 -6
  9. package/hooks/ll-state.js +30 -2
  10. package/package.json +3 -2
  11. package/scripts/evals/README.md +81 -0
  12. package/scripts/evals/cases/auto-dry-run/assert.sh +35 -0
  13. package/scripts/evals/cases/auto-dry-run/case.json +8 -0
  14. package/scripts/evals/cases/auto-dry-run/prompt.txt +1 -0
  15. package/scripts/evals/cases/auto-empty-repo/assert.sh +25 -0
  16. package/scripts/evals/cases/auto-empty-repo/case.json +8 -0
  17. package/scripts/evals/cases/auto-empty-repo/fixture/.gitkeep +0 -0
  18. package/scripts/evals/cases/auto-empty-repo/prompt.txt +1 -0
  19. package/scripts/evals/cases/decide-final-round/assert.sh +45 -0
  20. package/scripts/evals/cases/decide-final-round/case.json +8 -0
  21. package/scripts/evals/cases/decide-final-round/fixture/README.md +3 -0
  22. package/scripts/evals/cases/decide-final-round/prompt.txt +1 -0
  23. package/scripts/evals/cases/executor-block/assert.sh +33 -0
  24. package/scripts/evals/cases/executor-block/case.json +8 -0
  25. package/scripts/evals/cases/executor-block/prompt.txt +14 -0
  26. package/scripts/evals/cases/goal-autonomous/assert.sh +35 -0
  27. package/scripts/evals/cases/goal-autonomous/case.json +8 -0
  28. package/scripts/evals/cases/goal-autonomous/fixture/PLAN.md +42 -0
  29. package/scripts/evals/cases/goal-autonomous/fixture/PROGRESS.md +20 -0
  30. package/scripts/evals/cases/goal-autonomous/fixture/ROADMAP.md +29 -0
  31. package/scripts/evals/cases/goal-autonomous/fixture/package.json +8 -0
  32. package/scripts/evals/cases/goal-autonomous/fixture/src/money.js +6 -0
  33. package/scripts/evals/cases/goal-autonomous/fixture/test/reconcile.test.js +8 -0
  34. package/scripts/evals/cases/goal-autonomous/prompt.txt +1 -0
  35. package/scripts/evals/cases/implement-review-gate/assert.sh +37 -0
  36. package/scripts/evals/cases/implement-review-gate/case.json +8 -0
  37. package/scripts/evals/cases/implement-review-gate/prompt.txt +1 -0
  38. package/scripts/evals/cases/implement-stops-at-next/assert.sh +121 -0
  39. package/scripts/evals/cases/implement-stops-at-next/case.json +9 -0
  40. package/scripts/evals/cases/implement-stops-at-next/prompt.txt +1 -0
  41. package/scripts/evals/cases/preamble-no-ritual/assert.sh +17 -0
  42. package/scripts/evals/cases/preamble-no-ritual/case.json +8 -0
  43. package/scripts/evals/cases/preamble-no-ritual/fixture/README.md +3 -0
  44. package/scripts/evals/cases/preamble-no-ritual/fixture/src/a.ts +3 -0
  45. package/scripts/evals/cases/preamble-no-ritual/prompt.txt +1 -0
  46. package/scripts/evals/cases/router-no-skill/assert.sh +33 -0
  47. package/scripts/evals/cases/router-no-skill/case.json +8 -0
  48. package/scripts/evals/cases/router-no-skill/fixture/PLAN.md +21 -0
  49. package/scripts/evals/cases/router-no-skill/fixture/README.md +7 -0
  50. package/scripts/evals/cases/router-no-skill/fixture/decisions/README.md +3 -0
  51. package/scripts/evals/cases/router-no-skill/prompt.txt +1 -0
  52. package/scripts/evals/cases/router-small/assert.sh +21 -0
  53. package/scripts/evals/cases/router-small/case.json +8 -0
  54. package/scripts/evals/cases/router-small/fixture/README.md +17 -0
  55. package/scripts/evals/cases/router-small/prompt.txt +1 -0
  56. package/scripts/evals/cases/scout-no-plan/assert.sh +41 -0
  57. package/scripts/evals/cases/scout-no-plan/case.json +8 -0
  58. package/scripts/evals/cases/scout-no-plan/prompt.txt +8 -0
  59. package/scripts/evals/cases/verifier-weakened-test/assert.sh +19 -0
  60. package/scripts/evals/cases/verifier-weakened-test/case.json +8 -0
  61. package/scripts/evals/cases/verifier-weakened-test/prompt.txt +13 -0
  62. package/scripts/evals/cases/verifier-weakened-test/setup.sh +19 -0
  63. package/scripts/evals/fixtures/manual-contract/out.json +29 -0
  64. package/scripts/evals/fixtures/manual-contract/out.txt +5 -0
  65. package/scripts/evals/fixtures/manual-contract/with-skill.json +46 -0
  66. package/scripts/evals/fixtures/router-no-skill/fail.txt +5 -0
  67. package/scripts/evals/fixtures/router-no-skill/out.json +27 -0
  68. package/scripts/evals/fixtures/router-no-skill/pass.txt +4 -0
  69. package/scripts/evals/lib/assert.sh +107 -0
  70. package/scripts/evals/lib/extract.js +73 -0
  71. package/scripts/evals/run.sh +369 -0
  72. package/scripts/fixtures/auto-closed/PLAN.md +5 -0
  73. package/scripts/fixtures/auto-closed/PROGRESS.md +20 -0
  74. package/scripts/fixtures/auto-closed/ROADMAP.md +6 -0
  75. package/scripts/fixtures/auto-closed/docs/DELIVERY.md +3 -0
  76. package/scripts/fixtures/auto-decisions/decisions/DEC-0001-taken-alone.md +13 -0
  77. package/scripts/fixtures/auto-decisions/decisions/DEC-0002-owner.md +13 -0
  78. package/scripts/fixtures/auto-noroadmap/PLAN.md +20 -0
  79. package/scripts/fixtures/auto-noroadmap/PROGRESS.md +11 -0
  80. package/scripts/fixtures/auto-verify-next/PLAN.md +5 -0
  81. package/scripts/fixtures/auto-verify-next/PROGRESS.md +18 -0
  82. package/scripts/fixtures/auto-verify-next/ROADMAP.md +5 -0
  83. package/scripts/fixtures/auto-verify-next/phases/01/PLAN.md +6 -0
  84. package/scripts/fixtures/evals-auto/auto-dry-run/pass.txt +18 -0
  85. package/scripts/fixtures/evals-auto/auto-empty-repo/pass.txt +2 -0
  86. package/scripts/fixtures/evals-auto/goal-autonomous/pass.txt +29 -0
  87. package/scripts/fixtures/lint-bad/folded-description/SKILL.md +13 -0
  88. package/scripts/fixtures/lint-bad/jargon-in-questions/SKILL.md +36 -0
  89. package/scripts/fixtures/lint-bad/model-invocation-false/SKILL.md +10 -0
  90. package/scripts/fixtures/next-bad/skills/ll-bad/SKILL.md +30 -0
  91. package/scripts/fixtures/next-good/skills/ll-good/SKILL.md +26 -0
  92. package/scripts/fixtures/project/BACKLOG.md +6 -5
  93. package/scripts/fixtures/project/PROGRESS.md +4 -0
  94. package/scripts/lint-contract.cjs +495 -0
  95. package/scripts/lint-prompts.sh +443 -0
  96. package/scripts/ll-tools.js +497 -450
  97. package/scripts/smoke-test.sh +481 -4
  98. package/skills/ll-auto/SKILL.md +74 -0
  99. package/skills/ll-auto/references/run.md +75 -0
  100. package/skills/ll-auto/references/stages.md +66 -0
  101. package/skills/ll-auto/scripts/ll-auto.js +345 -0
  102. package/skills/ll-brainstorm/SKILL.md +14 -12
  103. package/skills/ll-brainstorm/references/decision-policy.md +22 -12
  104. package/skills/ll-close/SKILL.md +8 -7
  105. package/skills/ll-close/references/delivery.md +3 -1
  106. package/skills/ll-decide/SKILL.md +21 -19
  107. package/skills/ll-decide/references/decision-policy.md +22 -12
  108. package/skills/ll-decide/references/decision-room.md +5 -3
  109. package/skills/ll-decide/references/interview.md +29 -13
  110. package/skills/ll-decide/references/plan-skeleton.md +14 -14
  111. package/skills/ll-decide/references/premise-gate.md +34 -20
  112. package/skills/ll-goal/SKILL.md +22 -4
  113. package/skills/ll-goal/references/goal-template.md +57 -0
  114. package/skills/ll-implement/SKILL.md +25 -22
  115. package/skills/ll-implement/references/briefs.md +2 -1
  116. package/skills/ll-implement/references/decision-policy.md +22 -12
  117. package/skills/ll-implement/references/phase-conversation.md +20 -12
  118. package/skills/ll-implement/references/phase-plan.md +23 -0
  119. package/skills/ll-oncall/SKILL.md +3 -2
  120. package/skills/ll-refine/SKILL.md +3 -2
  121. package/skills/ll-research/SKILL.md +3 -2
  122. package/skills/ll-resume/SKILL.md +9 -4
  123. package/skills/ll-update/SKILL.md +6 -1
  124. package/skills/ll-verify/SKILL.md +2 -1
  125. package/skills/ll-verify/references/verifier-briefs.md +3 -0
@@ -0,0 +1,25 @@
1
+ #!/usr/bin/env bash
2
+ # "/ll-auto" on an empty repository -> the two-line stop, no question, nothing written.
3
+ . "$(cd "$(dirname "${BASH_SOURCE[0]}")/../../lib" && pwd)/assert.sh"
4
+
5
+ WORK="$1"; OUT_JSON="$2"; OUT_TXT="$3"
6
+
7
+ contains "$OUT_TXT" 'Nada encontrado neste reposit' 'the answer carries the empty-repository line'
8
+ contains "$OUT_TXT" '/ll-auto "<objetivo>"' 'the answer carries the command to complete'
9
+
10
+ no_path "$WORK/docs" 'no docs/ was created'
11
+ no_path "$WORK/PLAN.md" 'no PLAN.md was created'
12
+
13
+ # OUT_JSON may sit inside WORK for an offline check; exclude it by name so the
14
+ # capture file itself is not read as a change to the work tree.
15
+ dirty="$(git -C "$WORK" status --porcelain 2>/dev/null | grep -v -- " $(basename "$OUT_JSON")\$")"
16
+ if [ -z "$dirty" ]; then
17
+ ok 'git status --porcelain is empty: nothing was written'
18
+ else
19
+ fail "the working tree was changed: $(printf '%s' "$dirty" | tr '\n' ' ')"
20
+ fi
21
+
22
+ no_tool_use "$OUT_JSON" AskUserQuestion 'no AskUserQuestion tool call anywhere in the capture'
23
+ no_tool_use "$OUT_JSON" Skill 'no Skill tool call anywhere in the capture'
24
+
25
+ finish
@@ -0,0 +1,8 @@
1
+ {
2
+ "max_turns": 10,
3
+ "history": false,
4
+ "min_pass": 1,
5
+ "agent": null,
6
+ "permission_mode": "bypassPermissions",
7
+ "note": "Own fixture holding only .gitkeep (an empty repository). Proves the empty-repo stop: two exact lines, no question, nothing written. max_turns 10 (B-018): the `num_turns` the result reports counts the skill's `!` preprocessor Bash calls and the model's own tool calls together — real reps of this case reported 3 and 7 while the cap was 4 and the run still ended `success`, so the two numbers are not the same unit. The cap is pinned above the highest count seen, so a rep that really runs long is cut instead of scoring."
8
+ }
@@ -0,0 +1 @@
1
+ /ll-auto
@@ -0,0 +1,45 @@
1
+ #!/usr/bin/env bash
2
+ # The final round is always sent, with the counter, before anything is frozen.
3
+ . "$(cd "$(dirname "${BASH_SOURCE[0]}")/../../lib" && pwd)/assert.sh"
4
+
5
+ WORK="$1"; OUT_JSON="$2"; OUT_TXT="$3"
6
+
7
+ # a one-shot run has no owner to answer: either the final round (counter) or a blocking gate block (≤ 4 questions) is the correct stop
8
+ if grep -qE 'questions asked' "$OUT_TXT"; then
9
+ contains "$OUT_TXT" 'owner decisions open' 'the final round carries "owner decisions open"'
10
+ else
11
+ contains "$OUT_TXT" '\[PG-1\]|Pergunta 1/|Question 1/' 'the gate asks its first question instead of assuming'
12
+ n="$(grep -cE '^\*\*\[PG-[0-9]+\]|^\*\*Pergunta [0-9]+/|^\*\*Question [0-9]+/' "$OUT_TXT")"
13
+ if [ "$n" -le 4 ]; then ok "the block carries $n questions (≤ 4)"; else fail "the block carries $n questions (> 4)"; fi
14
+ no_path "$WORK/PLAN.md" 'nothing frozen while the gate is open'
15
+ fi
16
+
17
+ # The decision room is handed over before the first question, never after it (F-5).
18
+ if grep -qE 'OPTIONS\.html' "$OUT_TXT"; then
19
+ opt="$(grep -nE 'OPTIONS\.html' "$OUT_TXT" | head -1 | cut -d: -f1)"
20
+ q="$(grep -nE 'Pergunta 1/|Question 1/' "$OUT_TXT" | head -1 | cut -d: -f1)"
21
+ if [ -z "$q" ] || [ "$opt" -lt "$q" ]; then
22
+ ok "the decision-room path (line $opt) comes before the first question"
23
+ else
24
+ fail "the decision-room path (line $opt) comes after the first question (line $q)"
25
+ fi
26
+ else
27
+ ok 'no decision-room path in this answer: the room was not opened, so the order is not scored'
28
+ fi
29
+
30
+ # The contract is only scored when it was written: with an owner-only item open, nothing freezes.
31
+ if [ -f "$WORK/PLAN.md" ]; then
32
+ contains "$WORK/PLAN.md" '^## ' 'PLAN.md carries ## sections'
33
+ if [ -d "$WORK/decisions" ]; then
34
+ ok 'decisions/ exists beside the frozen PLAN.md'
35
+ else
36
+ fail 'PLAN.md was written but decisions/ does not exist'
37
+ fi
38
+ else
39
+ ok 'no PLAN.md written — nothing was frozen, so the contract is not scored'
40
+ fi
41
+
42
+ # The phase plan is never written by this skill.
43
+ no_path "$WORK/phases/01/PLAN.md" 'no phases/NN/PLAN.md was written here'
44
+
45
+ finish
@@ -0,0 +1,8 @@
1
+ {
2
+ "max_turns": 40,
3
+ "history": false,
4
+ "min_pass": 2,
5
+ "agent": null,
6
+ "permission_mode": "bypassPermissions",
7
+ "note": "Prompt is the slash command itself: since 3.0.0 the session never starts a skill from prose (preamble), so a prose prompt only hands the command back. Empty fixture so the LARGE route has nowhere to resume from. The final round is always sent, even when every item was band 2/3, so the counter must appear whether or not PLAN.md was frozen."
8
+ }
@@ -0,0 +1,3 @@
1
+ # reminders
2
+
3
+ An empty repository. Nothing is planned yet: no PLAN.md, no ROADMAP.md, no PROGRESS.md.
@@ -0,0 +1 @@
1
+ /ll-decide project serviço que envia lembretes por e-mail
@@ -0,0 +1,33 @@
1
+ #!/usr/bin/env bash
2
+ # The executor returns exactly one block with the seven fields, and the TDD order holds.
3
+ . "$(cd "$(dirname "${BASH_SOURCE[0]}")/../../lib" && pwd)/assert.sh"
4
+
5
+ WORK="$1"; OUT_JSON="$2"; OUT_TXT="$3"
6
+
7
+ contains "$OUT_TXT" '^### M1' 'the return opens with the ### M1 heading'
8
+ for f in built commits commands deviations questions backlog not_verified; do
9
+ contains "$OUT_TXT" "^${f}:" "the return carries the field ${f}:"
10
+ done
11
+
12
+ # TDD order over the whole history: a test(M1) commit exists and precedes the first feat(M1).
13
+ log="$(git -C "$WORK" log --format=%s --reverse)"
14
+ t="$(printf '%s\n' "$log" | grep -n '^test(M1): ' | head -1 | cut -d: -f1)"
15
+ f="$(printf '%s\n' "$log" | grep -n '^feat(M1): ' | head -1 | cut -d: -f1)"
16
+ if [ -n "$t" ] && [ -n "$f" ] && [ "$t" -lt "$f" ]; then
17
+ ok "git log shows test(M1) at #$t before feat(M1) at #$f"
18
+ else
19
+ fail "git log does not show a test(M1) commit before a feat(M1) commit (test=${t:-none} feat=${f:-none})"
20
+ fi
21
+
22
+ # The executor owns no state file.
23
+ base="$(base_sha "$WORK")"
24
+ if [ -n "$base" ]; then
25
+ touched="$(git -C "$WORK" diff --name-only "$base" HEAD -- PROGRESS.md PLAN.md ROADMAP.md BACKLOG.md decisions)"
26
+ if [ -z "$touched" ]; then
27
+ ok 'no state file was committed by the executor'
28
+ else
29
+ fail "the executor committed a state file: $(printf '%s' "$touched" | tr '\n' ' ')"
30
+ fi
31
+ fi
32
+
33
+ finish
@@ -0,0 +1,8 @@
1
+ {
2
+ "max_turns": 40,
3
+ "history": true,
4
+ "min_pass": 2,
5
+ "agent": "ll-executor",
6
+ "permission_mode": "bypassPermissions",
7
+ "note": "The fixture history already builds M1 (test(M1) then feat(M1)), so the red test passes on its first run and the contract says the executor returns BLOCKED. The case scores the return block, not the code."
8
+ }
@@ -0,0 +1,14 @@
1
+ MILESTONE M1 — end-to-end tracer
2
+ PLAN {{WORK}}/phases/07/PLAN.md (read whole before acting; your entry is the contract)
3
+ CONTEXT none — no CODE-CONTEXT.md for this phase; use read_first in the plan
4
+ FILES {{WORK}}/src/a.ts, {{WORK}}/test/a.test.ts
5
+ WAVE 1 of 3 · alone in this wave · previous blocks: {{WORK}}/PROGRESS.md "## Phase 07"
6
+ TDD yes — behavior cases in PLAN, milestone M1 · test first, commit test(M1) before feat(M1)
7
+ ACCEPTANCE npm test -- a.test.ts — run from {{WORK}} after the last commit
8
+ MODEL contract milestone → opus/high
9
+ DEC RESERVED DEC-0042, DEC-0043 (cite only these ids in questions:; never create a decision)
10
+ INPUTS {{WORK}}/src/a.ts, {{WORK}}/test/a.test.ts (checked: exist)
11
+ DO NOT write PROGRESS/PLAN/ROADMAP/BACKLOG/decisions; commit outside FILES; push; weaken a test;
12
+ spawn an agent; cd; relative paths; grep a directory that holds a .env
13
+ RETURN the `### M1` block in your fixed format (built, commits, commands, deviations, questions,
14
+ backlog, not_verified), at most 1,500 tokens. Nothing before or after it.
@@ -0,0 +1,35 @@
1
+ #!/usr/bin/env bash
2
+ # `/ll-goal --autonomous "<objective>"`: one pasted /goal text for the whole delivery,
3
+ # docs/GOAL.md written with `mode: autonomous` and `phase: all`, nothing asked, no skill started.
4
+ . "$(cd "$(dirname "${BASH_SOURCE[0]}")/../../lib" && pwd)/assert.sh"
5
+
6
+ WORK="$1"; OUT_JSON="$2"; OUT_TXT="$3"
7
+
8
+ contains "$OUT_TXT" '/goal' 'the answer carries the /goal text to paste'
9
+ # EXECUTION carries `ll-auto` with `--auto-decision`; the objective and `--verify all` may sit
10
+ # between them and the text wraps, so the answer is read flattened.
11
+ if tr '\n' ' ' < "$OUT_TXT" 2>/dev/null | grep -Eq 'll-auto.{0,80}--auto-decision'; then
12
+ ok 'EXECUTION runs the delivery with ll-auto --auto-decision'
13
+ else
14
+ fail 'EXECUTION runs the delivery with ll-auto --auto-decision'
15
+ fi
16
+
17
+ # The pasted block is what the owner copies: from the /goal line to the `▶ Next` line, or to the
18
+ # end when there is none. The goal text is a pointer, never a copy of PLAN.md: it stays small.
19
+ block="$(awk '/\/goal/ { on = 1 } on && /▶ Next/ { exit } on { print }' "$OUT_TXT" 2>/dev/null)"
20
+ n="$(printf '%s' "$block" | wc -c)"
21
+ if [ "$n" -ge 400 ] && [ "$n" -le 4000 ]; then
22
+ ok "the pasted /goal text measures $n chars (400..4000)"
23
+ else
24
+ fail "the pasted /goal text measures $n chars, outside 400..4000"
25
+ fi
26
+
27
+ if [ -f "$WORK/docs/GOAL.md" ]; then ok 'docs/GOAL.md exists in the work tree'
28
+ else fail 'docs/GOAL.md exists in the work tree'; fi
29
+ contains "$WORK/docs/GOAL.md" '^mode: autonomous$' 'docs/GOAL.md frontmatter carries mode: autonomous'
30
+ contains "$WORK/docs/GOAL.md" '^phase: all$' 'docs/GOAL.md frontmatter carries phase: all (the whole delivery)'
31
+
32
+ no_tool_use "$OUT_JSON" Skill 'no Skill tool call anywhere in the capture'
33
+ no_tool_use "$OUT_JSON" AskUserQuestion 'no AskUserQuestion tool call anywhere in the capture'
34
+
35
+ finish
@@ -0,0 +1,8 @@
1
+ {
2
+ "max_turns": 20,
3
+ "history": false,
4
+ "min_pass": 1,
5
+ "agent": null,
6
+ "permission_mode": "bypassPermissions",
7
+ "note": "Own fixture, not the shared scripts/fixtures/project: a healthy repo with PLAN.md, ROADMAP.md (07 and 08 PLANNED), a consistent PROGRESS.md board, no phases/NN/PLAN.md yet and `npm test` exit 0. The shared fixture is broken on purpose (plan-lint verdict fail, milestones marked passes: true with no commit, no test runner) and ll-goal step 1 is right to stop there — `a plan that fails there does not become a goal` — so it can never prove the autonomous branch. Proves: one pasted /goal text for the whole delivery with EXECUTION on ll-auto --auto-decision, docs/GOAL.md with `mode: autonomous` and `phase: all` committed, nothing asked and no skill started. max_turns 20 (B-018 rule: above the highest count real reps show — 17 here): the run reads PLAN/ROADMAP/PROGRESS, calls the ll-auto detect helper, writes and commits docs/GOAL.md, and the reported num_turns counts the skill's `!` preprocessor Bash calls together with the model's own tool calls."
8
+ }
@@ -0,0 +1,42 @@
1
+ # PLAN — reconciliation service
2
+
3
+ ## §0 Precedence
4
+
5
+ This file is self-contained and the ONLY entry of the work. §2 > §3 > §6 > §7. Decisions in §3 are a
6
+ contract — do not re-litigate. Project CLAUDE.md > this file > phase plans > briefs.
7
+
8
+ ## §1 Objective and truths
9
+
10
+ Every provider event is reconciled exactly once and every mismatch is visible to the operator.
11
+
12
+ - T1 A provider batch produces one row per event. — `npm test` exit 0
13
+ - T2 A replayed batch adds no rows. — `npm test` exit 0
14
+
15
+ Requirements: REQ-k (phase 07), REQ-m (phase 08).
16
+
17
+ ## §2 Invariants
18
+
19
+ I-01 Never delete, disable or weaken a test or an acceptance criterion. [owner, 2026-08-20]
20
+ I-02 Do not fill a gap with a plausible interpretation: report and ask. [owner, CLAUDE.md]
21
+
22
+ ## §3 Owner decisions
23
+
24
+ | id | question | decision | by | date | against recommendation? | reversible? |
25
+ |---|---|---|---|---|---|---|
26
+ | DEC-0041 | mismatch unit | cents, integer | owner | 2026-08-22 | no | 1 commit |
27
+
28
+ ## §6 Global acceptance
29
+
30
+ CA-01 — WHEN a batch is ingested THE SYSTEM SHALL write one row per event · `npm test` exit 0
31
+ CA-02 — WHEN a mismatch is found THE SYSTEM SHALL report it · `npm test` exit 0
32
+
33
+ ## §7 Execution protocol
34
+
35
+ Models per role: session fable/high · scout sonnet/medium · contract executor opus/high · mechanical
36
+ executor sonnet/medium · verifier opus/high · reviewer opus/medium. Max 3 executors per wave on
37
+ disjoint files. Commit per path after each milestone; push only with everything green (push is not a
38
+ deploy here). TDD on by default; exceptions: config and glue.
39
+
40
+ ## §8 Phases
41
+
42
+ Phases live in ROADMAP.md. Current phase: 07.
@@ -0,0 +1,20 @@
1
+ # PROGRESS — reconciliation service
2
+
3
+ <!-- ll-state -->
4
+ phase: 07
5
+ milestones:
6
+ <!-- /ll-state -->
7
+
8
+ ## Rules for all agents
9
+
10
+ - Only the main session writes this file. Executors return blocks; the session appends them.
11
+ - Never delete, disable or weaken a test or an acceptance criterion.
12
+ - Commit per milestone with `type(Mn): what`.
13
+
14
+ ## Epilogue — phase 06 — 2026-08-30
15
+
16
+ passed: M1, M2 (2/2) — provider adapters shipped, `npm test` green.
17
+ left: none. waiting: none.
18
+ verification: APPROVED.
19
+
20
+ ## Phase 07
@@ -0,0 +1,29 @@
1
+ # ROADMAP — reconciliation service
2
+
3
+ | phase | name | depends_on | requirements | state |
4
+ |---|---|---|---|---|
5
+ | 05 | payment intake | — | REQ-a | DONE (docs/history/v1.0) |
6
+ | 06 | provider adapters | 05 | REQ-b | DONE (docs/history/v1.0) |
7
+ | 07 | billing reconciliation | 05, 06 | REQ-k | PLANNED |
8
+ | 08 | reconciliation reporting | 07 | REQ-m | PLANNED |
9
+
10
+ ## Phase 07 — billing reconciliation
11
+
12
+ Objective: every provider event lands in `billing_events` exactly once and mismatches are visible.
13
+
14
+ Success criteria:
15
+ - SC-01 The reconciliation job ingests a provider batch and writes one row per event.
16
+ - SC-02 A replayed batch produces no duplicate rows.
17
+ - SC-03 Cent-level mismatches are reported instead of silently dropped.
18
+
19
+ Deferred ideas: customer portal; accounting export.
20
+
21
+ ## Phase 08 — reconciliation reporting
22
+
23
+ Objective: the operator reads yesterday's reconciliation without opening the database.
24
+
25
+ Success criteria:
26
+ - SC-01 A daily report lists ingested, tolerated and rejected rows.
27
+ - SC-02 The report is reachable from the operator console.
28
+
29
+ Deferred ideas: CSV export.
@@ -0,0 +1,8 @@
1
+ {
2
+ "name": "reconciliation-service",
3
+ "version": "0.6.0",
4
+ "private": true,
5
+ "scripts": {
6
+ "test": "node --test test/*.test.js"
7
+ }
8
+ }
@@ -0,0 +1,6 @@
1
+ // Cent-level arithmetic: DEC-0041 — money is an integer number of cents.
2
+ function toCents(amount) {
3
+ return Math.round(Number(amount) * 100);
4
+ }
5
+
6
+ module.exports = { toCents };
@@ -0,0 +1,8 @@
1
+ const test = require('node:test');
2
+ const assert = require('node:assert');
3
+
4
+ const { toCents } = require('../src/money.js');
5
+
6
+ test('amounts are compared in integer cents', () => {
7
+ assert.strictEqual(toCents('10.05'), 1005);
8
+ });
@@ -0,0 +1 @@
1
+ /ll-goal --autonomous "Deliver phases 07 and 08"
@@ -0,0 +1,37 @@
1
+ #!/usr/bin/env bash
2
+ # The review gate opens wave 1: PLAN-REVIEW.md exists, and no feat( commit of this run
3
+ # predates it. Only phase 07 is planned.
4
+ . "$(cd "$(dirname "${BASH_SOURCE[0]}")/../../lib" && pwd)/assert.sh"
5
+
6
+ WORK="$1"; OUT_JSON="$2"; OUT_TXT="$3"
7
+ REVIEW="phases/07/PLAN-REVIEW.md"
8
+
9
+ if [ -f "$WORK/$REVIEW" ]; then
10
+ ok "$REVIEW exists"
11
+ else
12
+ fail "$REVIEW does not exist: the gate before wave 1 was not written"
13
+ fi
14
+
15
+ base="$(base_sha "$WORK")"
16
+ # Only commits this run added count; the fixture history already carries feat( commits.
17
+ range="${base:+$base..HEAD}"
18
+ first_feat="$(git -C "$WORK" log ${range:+"$range"} --reverse --format='%ct %s' | grep -m1 ' feat(' | cut -d' ' -f1)"
19
+
20
+ if [ -z "$first_feat" ]; then
21
+ ok 'this run committed no feat(: nothing could precede the gate'
22
+ else
23
+ # The review is written once by the verifier and committed later with the phase state, so the
24
+ # earliest evidence dates it: the file mtime, or its commit when that is older.
25
+ review_ct="$(stat -c %Y "$WORK/$REVIEW" 2>/dev/null)"
26
+ review_commit="$(git -C "$WORK" log --diff-filter=A --format=%ct -- "$REVIEW" | tail -1)"
27
+ if [ -n "$review_commit" ] && { [ -z "$review_ct" ] || [ "$review_commit" -lt "$review_ct" ]; }; then review_ct="$review_commit"; fi
28
+ if [ -n "$review_ct" ] && [ "$review_ct" -le "$first_feat" ]; then
29
+ ok "the gate ($review_ct) precedes the first feat( of this run ($first_feat)"
30
+ else
31
+ fail "a feat( commit at $first_feat precedes the review gate at ${review_ct:-none}"
32
+ fi
33
+ fi
34
+
35
+ no_path "$WORK/phases/08" 'nothing was written under phases/08/'
36
+
37
+ finish
@@ -0,0 +1,8 @@
1
+ {
2
+ "max_turns": 45,
3
+ "history": true,
4
+ "min_pass": 2,
5
+ "agent": null,
6
+ "permission_mode": "bypassPermissions",
7
+ "note": "--no-talk so a one-shot run never waits on an owner-only question: the cents item (DEC-0041) becomes WAITING, M5/M6/M7 get stop: owner, and the review gate must still be written for M1/M2/M4. No executor is dispatched until phases/07/PLAN-REVIEW.md is on disk; and one invocation plans one phase, so phases/08/ must stay untouched."
8
+ }
@@ -0,0 +1 @@
1
+ /ll-implement 7 --no-talk
@@ -0,0 +1,121 @@
1
+ #!/usr/bin/env bash
2
+ # "▶ Next" ends the turn: no tool call follows it.
3
+ . "$(cd "$(dirname "${BASH_SOURCE[0]}")/../../lib" && pwd)/assert.sh"
4
+
5
+ WORK="$1"; OUT_JSON="$2"; OUT_TXT="$3"
6
+
7
+ verdict="$(node -e '
8
+ const fs = require("fs");
9
+ let data; try { data = JSON.parse(fs.readFileSync(process.argv[1], "utf8")); }
10
+ catch { process.stdout.write("unparseable out.json"); process.exit(0); }
11
+ const events = Array.isArray(data) ? data : [data];
12
+
13
+ let nextAt = -1; // index of the assistant text block carrying the marker
14
+ let toolAfter = null; // the first tool_use seen after it
15
+ events.forEach((ev, i) => {
16
+ if (!ev || ev.type !== "assistant") return; // hook context and tool results are not assistant text
17
+ const content = ev.message && ev.message.content;
18
+ if (!Array.isArray(content)) return;
19
+ for (const b of content) {
20
+ if (!b) continue;
21
+ if (b.type === "text" && /▶ Next/.test(b.text || "")) { if (nextAt < 0) nextAt = i; }
22
+ else if (b.type === "tool_use" && nextAt >= 0 && !toolAfter) toolAfter = b.name;
23
+ }
24
+ });
25
+
26
+ if (nextAt < 0) { process.stdout.write("no assistant text carries the ▶ Next marker"); process.exit(0); }
27
+ if (toolAfter) { process.stdout.write("a " + toolAfter + " tool call follows the ▶ Next line"); process.exit(0); }
28
+ process.stdout.write("");
29
+ ' "$OUT_JSON")"
30
+
31
+ if [ -z "$verdict" ]; then
32
+ ok 'the ▶ Next line is the last thing the invocation does'
33
+ else
34
+ fail "$verdict"
35
+ fi
36
+
37
+ contains "$OUT_TXT" '/clear' 'the next command is handed over with /clear'
38
+
39
+ # The waves are visible while they run, not only in the epilogue (F-4).
40
+ # The wave line is printed mid-run, so it lives in an earlier assistant text block, not in the final
41
+ # answer: scan the capture in order and map it onto the same scale as the epilogue check.
42
+ onda="$(node -e '
43
+ const fs = require("fs");
44
+ let data; try { data = JSON.parse(fs.readFileSync(process.argv[1], "utf8")); } catch { process.exit(0); }
45
+ const events = Array.isArray(data) ? data : [data];
46
+ let n = 0, hit = 0, epi = 0;
47
+ for (const ev of events) {
48
+ if (!ev || ev.type !== "assistant") continue;
49
+ const content = ev.message && ev.message.content;
50
+ if (!Array.isArray(content)) continue;
51
+ for (const b of content) {
52
+ if (!b) continue;
53
+ // a one-shot run prints no mid-run text: the wave heartbeat is the proxy; the screen line is checked by the lab rubric
54
+ if (b.type === "tool_use" && !hit && /heartbeat\s+\\?"(wave|onda) 1\//.test(JSON.stringify(b.input || {}))) hit = n + 1;
55
+ if (b.type !== "text") continue;
56
+ n++;
57
+ if (!hit && /(^|\n)onda 1\//.test(b.text || "")) hit = n;
58
+ if (!epi && /## Epilogue|▶ Next/.test(b.text || "")) epi = n;
59
+ }
60
+ }
61
+ if (hit) process.stdout.write(String(hit));
62
+ ' "$OUT_JSON")"
63
+ close_block="$(node -e '
64
+ const fs = require("fs");
65
+ let data; try { data = JSON.parse(fs.readFileSync(process.argv[1], "utf8")); } catch { process.exit(0); }
66
+ const events = Array.isArray(data) ? data : [data];
67
+ let n = 0;
68
+ for (const ev of events) {
69
+ if (!ev || ev.type !== "assistant") continue;
70
+ const content = ev.message && ev.message.content;
71
+ if (!Array.isArray(content)) continue;
72
+ for (const b of content) {
73
+ if (!b || b.type !== "text") continue;
74
+ n++;
75
+ if (/## Epilogue|▶ Next/.test(b.text || "")) { process.stdout.write(String(n)); process.exit(0); }
76
+ }
77
+ }
78
+ ' "$OUT_JSON")"
79
+ close="$(grep -nE '^## Epilogue|▶ Next' "$OUT_TXT" | head -1 | cut -d: -f1)"
80
+ if [ -z "$onda" ]; then
81
+ fail 'no "onda 1/M" line: the first wave ran with no visible progress'
82
+ elif [ -n "$close_block" ] && [ "$onda" -gt "$close_block" ]; then
83
+ fail "the onda 1/M line (text block $onda) comes only after the epilogue (text block $close_block)"
84
+ else
85
+ ok "the first wave is announced on screen (line $onda), before the epilogue"
86
+ fi
87
+
88
+ # The helper is called, never read (F-7): no shell that cats/seds/greps it, no Read of it.
89
+ reads="$(node -e '
90
+ const fs = require("fs");
91
+ let data; try { data = JSON.parse(fs.readFileSync(process.argv[1], "utf8")); }
92
+ catch { process.stdout.write("unparseable out.json"); process.exit(0); }
93
+ const events = Array.isArray(data) ? data : [data];
94
+ const shell = /(^|[\s;|&(])(cat|sed|grep|head|tail|less)\b[^\n;|&]*ll-tools\.js/; // the reading verb starts a command and the helper is its argument
95
+ for (const ev of events) {
96
+ if (!ev || ev.type !== "assistant") continue;
97
+ const content = ev.message && ev.message.content;
98
+ if (!Array.isArray(content)) continue;
99
+ for (const b of content) {
100
+ if (!b || b.type !== "tool_use") continue;
101
+ const input = b.input || {};
102
+ if (b.name === "Bash" && shell.test(String(input.command || ""))) {
103
+ process.stdout.write("a Bash call reads the helper: " + String(input.command).slice(0, 80));
104
+ process.exit(0);
105
+ }
106
+ if (b.name === "Read" && /ll-tools\.js$/.test(String(input.file_path || ""))) {
107
+ process.stdout.write("a Read call opens the helper: " + String(input.file_path));
108
+ process.exit(0);
109
+ }
110
+ }
111
+ }
112
+ process.stdout.write("");
113
+ ' "$OUT_JSON")"
114
+
115
+ if [ -z "$reads" ]; then
116
+ ok 'the helper was called, never read'
117
+ else
118
+ fail "$reads"
119
+ fi
120
+
121
+ finish
@@ -0,0 +1,9 @@
1
+ {
2
+ "max_turns": 30,
3
+ "history": true,
4
+ "min_pass": 2,
5
+ "agent": null,
6
+ "permission_mode": "bypassPermissions",
7
+ "reuse": "implement-review-gate",
8
+ "note": "Scores the same capture as implement-review-gate — no second claude call. Run both cases together, or this one reports the missing source."
9
+ }
@@ -0,0 +1 @@
1
+ /ll-implement 7 --no-talk
@@ -0,0 +1,17 @@
1
+ #!/usr/bin/env bash
2
+ # The typo is fixed and no ceremony file is born.
3
+ . "$(cd "$(dirname "${BASH_SOURCE[0]}")/../../lib" && pwd)/assert.sh"
4
+
5
+ WORK="$1"; OUT_JSON="$2"; OUT_TXT="$3"
6
+
7
+ contains "$WORK/src/a.ts" 'receive' 'src/a.ts now spells receive'
8
+ absent "$WORK/src/a.ts" 'recieve' 'no occurrence of the typo is left'
9
+
10
+ no_path "$WORK/PROGRESS.md" 'no PROGRESS.md was created'
11
+ no_path "$WORK/VERIFICATION.md" 'no VERIFICATION.md was created'
12
+ no_path "$WORK/phases" 'no phases/ was created'
13
+ no_path "$WORK/PLAN.md" 'no PLAN.md was created'
14
+
15
+ no_tool_use "$OUT_JSON" Skill 'no Skill tool call anywhere in the capture'
16
+
17
+ finish
@@ -0,0 +1,8 @@
1
+ {
2
+ "max_turns": 6,
3
+ "history": false,
4
+ "min_pass": 2,
5
+ "agent": null,
6
+ "permission_mode": "bypassPermissions",
7
+ "note": "SMALL forbids the ritual: no spec, no plan, no PROGRESS, no VERIFICATION, no subagent for what fits in three calls, and no skill started on the session's behalf."
8
+ }
@@ -0,0 +1,3 @@
1
+ # events
2
+
3
+ One module. One typo.
@@ -0,0 +1,3 @@
1
+ export function recieveEvent(id: string): string {
2
+ return `recieved:${id}`;
3
+ }
@@ -0,0 +1 @@
1
+ corrige o typo 'recieve' em src/a.ts
@@ -0,0 +1,33 @@
1
+ #!/usr/bin/env bash
2
+ # A skill runs only when the owner types it: a plain research request is answered here and
3
+ # now — no command handed back to paste, no ▶ Next, no Skill tool call — even though the
4
+ # fixture carries ll-skills state (PLAN.md, decisions/).
5
+ . "$(cd "$(dirname "${BASH_SOURCE[0]}")/../../lib" && pwd)/assert.sh"
6
+
7
+ WORK="$1"; OUT_JSON="$2"; OUT_TXT="$3"
8
+
9
+ # The queue engines a real answer names. Longest alternative first: grep -E is leftmost-longest,
10
+ # so "bullmq" never scores as "bull".
11
+ QUEUES='bullmq|bee-queue|pg-boss|rabbitmq|graphile|bull|redis|sqs|kafka|agenda'
12
+
13
+ absent "$OUT_TXT" '▶ Next' 'the answer ends in no ▶ Next: this turn is not a skill hand-off'
14
+
15
+ if [ -f "$OUT_TXT" ] && grep -Eqi -- 'cole:|cole isto|cole o comando|paste this' "$OUT_TXT"; then
16
+ fail "the answer hands a command back to paste: $(grep -Eio -- 'cole:|cole isto|cole o comando|paste this' "$OUT_TXT" | head -1)"
17
+ else
18
+ ok 'the answer hands no command back to paste'
19
+ fi
20
+
21
+ n="$(grep -Eoi -- "$QUEUES" "$OUT_TXT" 2>/dev/null | tr 'A-Z' 'a-z' | sort -u | wc -l)"
22
+ if [ "${n:-0}" -ge 2 ]; then
23
+ ok "the research was done here: $n queue options named"
24
+ else
25
+ fail "the answer names $n queue option(s): the research was not done"
26
+ fi
27
+
28
+ no_tool_use "$OUT_JSON" Skill 'no Skill tool call anywhere in the capture'
29
+
30
+ no_path "$WORK/PROGRESS.md" 'no PROGRESS.md was created'
31
+ no_path "$WORK/phases" 'no phases/ was created'
32
+
33
+ finish
@@ -0,0 +1,8 @@
1
+ {
2
+ "max_turns": 8,
3
+ "history": false,
4
+ "min_pass": 2,
5
+ "agent": null,
6
+ "permission_mode": "acceptEdits",
7
+ "note": "A skill runs only when the owner types it. A plain request is done as asked, never handed back as a command to paste — and the fixture carries ll-skills state (PLAN.md, decisions/) to prove the claim where a router would be loudest. No PROGRESS.md, so nothing competes with the answer. max_turns 8 leaves room for a couple of reads before the answer."
8
+ }
@@ -0,0 +1,21 @@
1
+ # PLAN — now
2
+
3
+ ## §0 Precedence
4
+
5
+ This file is the entry of the work; the project CLAUDE.md wins over it.
6
+
7
+ ## §1 Objective and truths
8
+
9
+ One sentence of delivery: the CLI prints the local time and exits 0.
10
+
11
+ - T1 `node now.js` prints one line and exits 0. — `node now.js`
12
+
13
+ ## §2 Invariants
14
+
15
+ I-01 NEVER delete, disable or weaken a test or acceptance criterion. [owner]
16
+
17
+ ## §3 Owner decisions
18
+
19
+ | id | question | decision | by | date |
20
+ |---|---|---|---|---|
21
+ | — | none recorded yet | — | — | — |
@@ -0,0 +1,7 @@
1
+ # now
2
+
3
+ A tiny CLI that prints the time.
4
+
5
+ ## Run
6
+
7
+ node now.js