@bendyline/gilde 0.1.4 → 0.1.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -0
- package/data/chat-models/de/deepseek-v4-flash-284b-q2/manifest.json +4 -3
- package/data/chat-models/de/deepseek-v4-flash-284b-q4/manifest.json +4 -3
- package/data/chat-models/ge/gemma4-12b-q4/manifest.json +20 -11
- package/data/chat-models/ge/gemma4-12b-q4/versions/1.1.0/manifest.json +2 -8
- package/data/chat-models/ge/gemma4-12b-q4/versions/1.1.2/manifest.json +78 -0
- package/data/chat-models/ge/gemma4-12b-q8/manifest.json +23 -14
- package/data/chat-models/ge/gemma4-12b-q8/versions/1.0.0/manifest.json +2 -8
- package/data/chat-models/ge/gemma4-12b-q8/versions/1.0.2/manifest.json +78 -0
- package/data/chat-models/ge/gemma4-26b-q4/manifest.json +55 -46
- package/data/chat-models/ge/gemma4-26b-q4/versions/1.2.0/manifest.json +2 -8
- package/data/chat-models/ge/gemma4-26b-q4/versions/1.2.1/manifest.json +81 -0
- package/data/chat-models/ge/gemma4-31b-q4/manifest.json +36 -28
- package/data/chat-models/ge/gemma4-31b-q4/versions/1.2.0/manifest.json +2 -8
- package/data/chat-models/ge/gemma4-31b-q4/versions/1.2.1/manifest.json +87 -0
- package/data/chat-models/ge/gemma4-e2b-q8/manifest.json +76 -69
- package/data/chat-models/ge/gemma4-e2b-q8/versions/1.1.0/manifest.json +2 -8
- package/data/chat-models/ge/gemma4-e2b-q8/versions/1.1.2/manifest.json +71 -0
- package/data/chat-models/ge/gemma4-e4b-q8/manifest.json +71 -78
- package/data/chat-models/ge/gemma4-e4b-q8/versions/1.1.0/manifest.json +3 -9
- package/data/chat-models/ge/gemma4-e4b-q8/versions/1.1.2/manifest.json +71 -0
- package/data/chat-models/index.json +1 -1
- package/data/chat-models/la/laguna-s-2.1-118b-q4/manifest.json +17 -17
- package/data/chat-models/la/laguna-s-2.1-118b-q4/versions/1.0.1/manifest.json +124 -0
- package/data/chat-models/la/laguna-s-2.1-118b-q8/manifest.json +7 -7
- package/data/chat-models/la/laguna-s-2.1-118b-q8/versions/1.0.1/manifest.json +174 -0
- package/data/chat-models/mi/mistral-medium-3.5-128b-q4/manifest.json +4 -17
- package/data/chat-models/mi/mistral-medium-3.5-128b-q4/versions/1.0.0/manifest.json +2 -12
- package/data/chat-models/ne/nemotron3-nano-30b-q4/manifest.json +5 -0
- package/data/chat-models/qw/qwen3.5-122b-a10b-q4/manifest.json +27 -19
- package/data/chat-models/qw/qwen3.5-122b-a10b-q4/versions/1.0.1/manifest.json +164 -0
- package/data/chat-models/qw/qwen3.5-2b-q4/manifest.json +79 -77
- package/data/chat-models/qw/qwen3.5-2b-q4/versions/1.1.0/manifest.json +2 -8
- package/data/chat-models/qw/qwen3.5-2b-q4/versions/1.1.2/manifest.json +76 -0
- package/data/chat-models/qw/qwen3.5-4b-q4/manifest.json +79 -77
- package/data/chat-models/qw/qwen3.5-4b-q4/versions/1.1.0/manifest.json +2 -8
- package/data/chat-models/qw/qwen3.5-4b-q4/versions/1.1.2/manifest.json +76 -0
- package/data/chat-models/qw/qwen3.5-9b-q4/manifest.json +87 -85
- package/data/chat-models/qw/qwen3.5-9b-q4/versions/1.1.0/manifest.json +2 -8
- package/data/chat-models/qw/qwen3.5-9b-q4/versions/1.1.2/manifest.json +81 -0
- package/data/chat-models/qw/qwen3.6-27b-q4/manifest.json +99 -97
- package/data/chat-models/qw/qwen3.6-27b-q4/versions/1.1.0/manifest.json +2 -8
- package/data/chat-models/qw/qwen3.6-27b-q4/versions/1.1.4/manifest.json +91 -0
- package/data/chat-models/qw/qwen3.6-27b-q8/manifest.json +16 -12
- package/data/chat-models/qw/qwen3.6-27b-q8/versions/1.0.0/manifest.json +2 -8
- package/data/chat-models/qw/qwen3.6-27b-q8/versions/1.0.2/manifest.json +103 -0
- package/data/chat-models/qw/qwen3.6-35b-a3b-q4/manifest.json +18 -14
- package/data/chat-models/qw/qwen3.6-35b-a3b-q4/versions/1.0.0/manifest.json +2 -8
- package/data/chat-models/qw/qwen3.6-35b-a3b-q4/versions/1.0.1/manifest.json +93 -0
- package/data/chat-models/qw/qwen3.6-35b-a3b-q8/manifest.json +16 -12
- package/data/chat-models/qw/qwen3.6-35b-a3b-q8/versions/1.0.0/manifest.json +2 -8
- package/data/chat-models/qw/qwen3.6-35b-a3b-q8/versions/1.0.1/manifest.json +113 -0
- package/data/craftbook-templates/bu/bug-fix-tdd/versions/1.0.1/test.json +0 -5
- package/data/craftbook-templates/bu/build-loop/versions/1.1.0/craftbook.json +65 -0
- package/data/craftbook-templates/bu/build-loop/versions/1.1.0/test.json +121 -0
- package/data/craftbook-templates/ch/character-sheet/versions/1.0.0/test.json +1 -2
- package/data/craftbook-templates/ch/character-turnaround/versions/1.0.0/test.json +1 -2
- package/data/craftbook-templates/cl/cli-tool/versions/1.1.0/craftbook.json +139 -0
- package/data/craftbook-templates/cl/cli-tool/versions/1.1.0/test.json +128 -0
- package/data/craftbook-templates/cr/crossword-forge/manifest.json +1 -2
- package/data/craftbook-templates/de/deep-security-review/versions/1.1.0/craftbook.json +201 -0
- package/data/craftbook-templates/de/deep-security-review/versions/1.1.0/test.json +157 -0
- package/data/craftbook-templates/do/dockerize-app/versions/1.0.0/test.json +0 -5
- package/data/craftbook-templates/fr/freeze-scope/manifest.json +1 -2
- package/data/craftbook-templates/fr/freeze-scope/versions/{1.0.1 → 1.1.0}/craftbook.json +5 -17
- package/data/craftbook-templates/gr/graphql-api/versions/1.0.1/test.json +0 -5
- package/data/craftbook-templates/gr/grpc-service/versions/1.0.0/test.json +0 -5
- package/data/craftbook-templates/ho/hotfix-flow/versions/1.0.0/test.json +0 -5
- package/data/craftbook-templates/in/investigate/versions/1.1.0/craftbook.json +81 -0
- package/data/craftbook-templates/in/investigate/versions/1.1.0/test.json +122 -0
- package/data/craftbook-templates/in/invoice-run/manifest.json +1 -2
- package/data/craftbook-templates/in/invoice-run/versions/{1.0.1 → 1.1.0}/craftbook.json +4 -4
- package/data/craftbook-templates/index.json +1 -1
- package/data/craftbook-templates/li/library-package/versions/1.0.0/test.json +0 -5
- package/data/craftbook-templates/me/memory-prompt-session/manifest.json +1 -2
- package/data/craftbook-templates/me/message-queue-consumer/versions/1.0.0/test.json +0 -5
- package/data/craftbook-templates/of/office-hours/versions/1.1.0/craftbook.json +48 -0
- package/data/craftbook-templates/of/office-hours/versions/1.1.0/test.json +118 -0
- package/data/craftbook-templates/pa/page-spread/manifest.json +1 -2
- package/data/craftbook-templates/pa/parser-grammar/versions/1.0.0/test.json +0 -5
- package/data/craftbook-templates/pe/perf-optimization/versions/1.0.0/test.json +0 -5
- package/data/craftbook-templates/pl/plan/manifest.json +1 -2
- package/data/craftbook-templates/po/powerpoint-deck/versions/1.1.0/craftbook.json +128 -0
- package/data/craftbook-templates/{ro/root-cause-investigation/versions/1.0.1 → po/powerpoint-deck/versions/1.1.0}/test.json +6 -6
- package/data/craftbook-templates/pu/pull-request-review/versions/1.1.0/craftbook.json +118 -0
- package/data/craftbook-templates/pu/pull-request-review/versions/1.1.0/test.json +160 -0
- package/data/craftbook-templates/re/refactor-module/versions/1.0.1/test.json +0 -5
- package/data/craftbook-templates/re/regex-builder/versions/1.0.0/test.json +0 -5
- package/data/craftbook-templates/re/research-to-document/versions/1.1.0/craftbook.json +158 -0
- package/data/craftbook-templates/re/research-to-document/versions/1.1.0/test.json +97 -0
- package/data/craftbook-templates/ro/root-cause-investigation/manifest.json +1 -2
- package/data/craftbook-templates/sd/sdk-wrapper/versions/1.0.1/test.json +0 -5
- package/data/craftbook-templates/se/security-architecture-review/versions/1.1.0/craftbook.json +28 -0
- package/data/craftbook-templates/{te/technical-documentation/versions/1.0.1 → se/security-architecture-review/versions/1.1.0}/test.json +7 -7
- package/data/craftbook-templates/sh/ship/versions/1.1.0/craftbook.json +137 -0
- package/data/craftbook-templates/sh/ship/versions/1.1.0/test.json +225 -0
- package/data/craftbook-templates/st/state-machine/versions/1.0.0/test.json +0 -5
- package/data/craftbook-templates/te/technical-documentation/manifest.json +1 -2
- package/data/craftbook-templates/te/test-suite-backfill/versions/1.0.0/test.json +0 -5
- package/data/craftbook-templates/ti/tileset-batch/versions/1.0.0/test.json +1 -2
- package/data/craftbook-templates/ty/type-safety-pass/versions/1.0.0/test.json +0 -5
- package/data/craftbook-templates/ve/version-bump/versions/1.0.0/test.json +0 -5
- package/data/gezel-templates/ch/chess-player/manifest.json +20 -0
- package/data/gezel-templates/ch/chess-player/versions/1.0.0/about.md +21 -0
- package/data/gezel-templates/ch/chess-player/versions/1.0.0/manifest.json +10 -0
- package/data/gezel-templates/go/go-player/manifest.json +22 -0
- package/data/gezel-templates/go/go-player/versions/1.0.0/about.md +22 -0
- package/data/gezel-templates/go/go-player/versions/1.0.0/manifest.json +10 -0
- package/data/gezel-templates/index.json +1 -1
- package/data/project-types/ch/chess/manifest.json +20 -0
- package/data/project-types/ch/chess/versions/1.0.0/about.md +9 -0
- package/data/project-types/ch/chess/versions/1.0.0/game.json +127 -0
- package/data/project-types/ch/chess/versions/1.0.0/manifest.json +180 -0
- package/data/project-types/ch/chess/versions/1.0.0/mission.md +8 -0
- package/data/project-types/ch/chess/versions/1.0.0/pages/board/index.html +574 -0
- package/data/project-types/go/go/manifest.json +22 -0
- package/data/project-types/go/go/versions/1.0.0/about.md +9 -0
- package/data/project-types/go/go/versions/1.0.0/game.json +103 -0
- package/data/project-types/go/go/versions/1.0.0/manifest.json +166 -0
- package/data/project-types/go/go/versions/1.0.0/mission.md +8 -0
- package/data/project-types/go/go/versions/1.0.0/pages/board/index.html +637 -0
- package/data/project-types/index.json +1 -1
- package/package.json +2 -1
- package/schemas/chat-model-version.schema.json +22 -0
- package/schemas/craftbook-doc.schema.json +12 -0
- package/schemas/craftbook-template-version.schema.json +12 -0
- package/schemas/craftbook-test.schema.json +12 -0
- package/data/chat-models/gl/glm-5.2-754b-q2/manifest.json +0 -66
- package/data/chat-models/gl/glm-5.2-754b-q2/versions/1.0.0/manifest.json +0 -18
- package/data/craftbook-templates/cr/crossword-forge/versions/1.0.1/craftbook.json +0 -147
- package/data/craftbook-templates/cr/crossword-forge/versions/1.0.1/test.json +0 -135
- package/data/craftbook-templates/me/memory-prompt-session/versions/1.0.1/craftbook.json +0 -116
- package/data/craftbook-templates/me/memory-prompt-session/versions/1.0.1/test.json +0 -112
- package/data/craftbook-templates/pa/page-spread/versions/1.0.1/craftbook.json +0 -94
- package/data/craftbook-templates/pa/page-spread/versions/1.0.1/test.json +0 -133
- package/data/craftbook-templates/pl/plan/versions/1.0.1/craftbook.json +0 -104
- package/data/craftbook-templates/pl/plan/versions/1.0.1/test.json +0 -91
- package/data/craftbook-templates/ro/root-cause-investigation/versions/1.0.1/craftbook.json +0 -126
- package/data/craftbook-templates/te/technical-documentation/versions/1.0.1/craftbook.json +0 -152
- /package/data/craftbook-templates/fr/freeze-scope/versions/{1.0.1 → 1.1.0}/test.json +0 -0
- /package/data/craftbook-templates/in/invoice-run/versions/{1.0.1 → 1.1.0}/test.json +0 -0
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
{
|
|
2
|
+
"id": "build-loop",
|
|
3
|
+
"name": "Build Loop",
|
|
4
|
+
"description": "A generic **make-something** procedure: `build → evaluate → (loop) → finish`.\n\nIt exists for the common case where the user wants a thing built (\"make an arcade game\", \"build a landing page\", \"write a script that does X\") and no domain-specific craftbook fits better. It gives a small/medium model the thing it loses on multi-phase work: a **durable structure with a gate** — so it drives one deliverable to a checkable bar instead of declaring victory early.\n\n## How the loop works\n\n1. **Build** (developer) first writes an **acceptance-criteria checklist** to task notes, then builds the deliverable to meet it. For a web build that's `index.html`.\n2. **Evaluate** (reviewer) grades the deliverable against every criterion and routes: all pass → `advance_task_step(next: \"finish\")`; any fail → write the gaps → `advance_task_step(next: \"build\")` to loop back.\n3. **Finish** stamps a DONE summary.\n\nEvaluate's default forward edge loops back to **Build**, not Finish — so the safe failure mode is \"keep improving\", never \"ship half-done\".\n\n## Advancement rides on the deliverable, not a meta-tool call\n\nThe Build step declares an `advanceWhen` contract: when a complete `index.html` lands (clears a size floor + a non-truncation `html-complete` sniff), the runtime advances Build → Evaluate **automatically**, without the model calling `advance_task_step`. This is deliberate: models reliably *do work* (`write_file`) but reliably *skip* the meta-navigation tool — two eval runs (gemma4-31b, gemma4-e4b-q8) produced the deliverable and advanced **zero** steps. Observable-progress auto-advance closes that gap. For a non-`index.html` deliverable the auto-advance is a no-op and Build advances model-driven (its prompt says so).\n\n## Why criteria-first matters for small models\n\nThe acceptance-criteria checklist is the scaffold. Without it a weak model declares victory early; with it, \"is this good?\" becomes \"does criterion 4 pass?\" — answerable even by a small model, and the thing the gate and the next loop iteration check against.\n\n## Specific books beat this one\n\nThis is the **fallback**. A domain craftbook earns its keep by encoding things this generic shape can't: the right *specialist role per phase* (a space-shooter splits design into a game-designer and a visual-designer) and the *domain-correct ordering* (a branding site locks a visual language before any HTML). When selection finds a specific match, prefer it; fall back to Build Loop when nothing fits. In a **solo** project the meester pins this book with every step collapsed onto the one specialist (it can't recruit); in a **crew** project it pins the matching multi-role gallery book instead.\n\n## Roles are suggestions\n\n`build → developer`, `evaluate → reviewer` are `suggestedRole` hints resolved via `ensureGezel`; domain gallery books override them (game-designer, visual-designer, copywriter, …).\n\n## Next: an automated evaluate gate\n\nEvaluate is still reviewer-gated (a gezel grades + routes). The follow-up replaces its routing with a script that judges and advances deterministically — now reachable because observable-advance gets the workflow *to* Evaluate without depending on the model to navigate there.\n",
|
|
5
|
+
"entryStepId": "build",
|
|
6
|
+
"steps": [
|
|
7
|
+
{
|
|
8
|
+
"id": "build",
|
|
9
|
+
"name": "Build",
|
|
10
|
+
"description": "Lock concrete acceptance criteria, then build the deliverable to meet them. The workflow advances to Evaluate automatically when the deliverable lands.",
|
|
11
|
+
"prompt": "**Two things, in order.**\n\n1. **Lock the bar.** Write an **acceptance-criteria checklist** to task notes via `write_task_note` — 4-8 concrete, checkable statements that define \"good enough\" (e.g. \"clicking Start begins play\", \"score increments on a hit\", \"page loads with no console error\"). This is the contract Evaluate grades against; without it the loop has no target.\n2. **Build it.** Create the real deliverable with your file-writing capability, to satisfy every criterion. For a web build that's `index.html` (inline CSS + JS, no build step). On a loop-back from Evaluate, read the gaps in notes and fix ONLY those — don't regress criteria that already passed.\n\n**Advancing:** the workflow moves you to Evaluate automatically when you actually WRITE `index.html` this turn — you do NOT need to call `advance_task_step`. On a loop-back, narrating the gaps is not enough: the step holds until you make a real edit to `index.html`. For any other deliverable path, write the file, then call `advance_task_step({ ref, stepId: \"build\", next: \"evaluate\" })` yourself.",
|
|
12
|
+
"suggestedRole": "developer",
|
|
13
|
+
"advanceWhen": {
|
|
14
|
+
"file": "index.html",
|
|
15
|
+
"minBytes": 800,
|
|
16
|
+
"sniff": "html-complete",
|
|
17
|
+
"requireChange": true
|
|
18
|
+
},
|
|
19
|
+
"gate": {
|
|
20
|
+
"at": "completion",
|
|
21
|
+
"checks": [
|
|
22
|
+
{
|
|
23
|
+
"kind": "minBytes",
|
|
24
|
+
"file": "index.html",
|
|
25
|
+
"bytes": 2048
|
|
26
|
+
},
|
|
27
|
+
{
|
|
28
|
+
"kind": "jsParses",
|
|
29
|
+
"file": "index.html"
|
|
30
|
+
}
|
|
31
|
+
],
|
|
32
|
+
"scripts": [
|
|
33
|
+
{
|
|
34
|
+
"name": "checkHtmlComplete",
|
|
35
|
+
"scope": "standard",
|
|
36
|
+
"inputs": {
|
|
37
|
+
"file": "index.html"
|
|
38
|
+
}
|
|
39
|
+
}
|
|
40
|
+
],
|
|
41
|
+
"onReject": "build",
|
|
42
|
+
"maxAttempts": 4
|
|
43
|
+
},
|
|
44
|
+
"next": "evaluate"
|
|
45
|
+
},
|
|
46
|
+
{
|
|
47
|
+
"id": "evaluate",
|
|
48
|
+
"name": "Evaluate",
|
|
49
|
+
"description": "Grade the deliverable against EVERY acceptance criterion. All pass → finish. Any fail → write the gaps and loop back to Build.",
|
|
50
|
+
"prompt": "**You are the gate. Judge, then route — do not build here.**\n\n1. `read_task_notes({ ref })` for the acceptance-criteria checklist + the deliverable path.\n2. Open the deliverable and check it against each criterion (exercise it in a browser/with a test capability if the project has one — don't grade from source alone when you can observe behavior). Write a PASS/FAIL per criterion to notes.\n\nThen route:\n\n- **Every criterion PASSES →** `advance_task_step({ ref, stepId: \"evaluate\", next: \"finish\" })`.\n- **Any criterion FAILS →** write the specific gaps to notes, then `advance_task_step({ ref, stepId: \"evaluate\", next: \"build\" })` to loop back.\n\nNever route to `finish` with an unmet criterion — under-delivering is the failure this gate exists to catch. After ~3 unproductive loops, stop and report DONE_WITH_CONCERNS so the user can step in.",
|
|
51
|
+
"suggestedRole": "reviewer",
|
|
52
|
+
"next": "build"
|
|
53
|
+
},
|
|
54
|
+
{
|
|
55
|
+
"id": "finish",
|
|
56
|
+
"name": "Finish",
|
|
57
|
+
"description": "All acceptance criteria met. Stamp a short summary and report DONE.",
|
|
58
|
+
"prompt": "Every acceptance criterion passed. Write a one-paragraph DONE summary to task notes via `write_task_note`: what was built, the deliverable path(s), and a one-line confirmation that each criterion is met. Then report DONE.",
|
|
59
|
+
"suggestedRole": "developer",
|
|
60
|
+
"terminal": true
|
|
61
|
+
}
|
|
62
|
+
],
|
|
63
|
+
"version": "1.1.0",
|
|
64
|
+
"releasedAt": "2026-07-28T00:00:00Z"
|
|
65
|
+
}
|
|
@@ -0,0 +1,121 @@
|
|
|
1
|
+
{
|
|
2
|
+
"schemaVersion": 1,
|
|
3
|
+
"title": "Generic build-evaluate loop",
|
|
4
|
+
"objective": "Measure whether the fallback build-loop craftbook creates acceptance criteria, builds a complete interactive deliverable, and iterates through deterministic gates.",
|
|
5
|
+
"tags": [
|
|
6
|
+
"external"
|
|
7
|
+
],
|
|
8
|
+
"prompt": "In the `Build Loop Eval` project, build a self-contained `index.html` checklist timer for a small workshop. It must let a facilitator add agenda items, mark items done, start/pause a countdown timer, reset the timer, show remaining time, and display a completion summary. Lock acceptance criteria in task notes first, then build the page with inline CSS/JS and no external assets.",
|
|
9
|
+
"setup": {
|
|
10
|
+
"projectName": "Build Loop Eval",
|
|
11
|
+
"about": "Self-contained eval project for the build-loop craftbook. The deliverable is workspace/index.html.",
|
|
12
|
+
"missionObjectives": "Build an interactive checklist timer with acceptance criteria, add/complete agenda behavior, start/pause/reset timer, remaining time display, and completion summary.",
|
|
13
|
+
"files": [],
|
|
14
|
+
"worker": {
|
|
15
|
+
"name": "Bram",
|
|
16
|
+
"role": "Developer"
|
|
17
|
+
}
|
|
18
|
+
},
|
|
19
|
+
"mocks": [],
|
|
20
|
+
"success": {
|
|
21
|
+
"summary": "workspace/index.html is a complete interactive checklist timer with substantial JS/CSS and task notes show the criteria-first loop was used.",
|
|
22
|
+
"deliverables": [
|
|
23
|
+
{
|
|
24
|
+
"path": "index.html",
|
|
25
|
+
"kind": "html-page",
|
|
26
|
+
"minBytes": 3200,
|
|
27
|
+
"checks": [
|
|
28
|
+
{
|
|
29
|
+
"kind": "cssMinBytes",
|
|
30
|
+
"bytes": 650,
|
|
31
|
+
"file": "index.html"
|
|
32
|
+
},
|
|
33
|
+
{
|
|
34
|
+
"kind": "jsParses",
|
|
35
|
+
"file": "index.html"
|
|
36
|
+
},
|
|
37
|
+
{
|
|
38
|
+
"kind": "contains",
|
|
39
|
+
"file": "index.html",
|
|
40
|
+
"pattern": "agenda|checklist",
|
|
41
|
+
"flags": "i"
|
|
42
|
+
},
|
|
43
|
+
{
|
|
44
|
+
"kind": "contains",
|
|
45
|
+
"file": "index.html",
|
|
46
|
+
"pattern": "start|pause|reset",
|
|
47
|
+
"flags": "i"
|
|
48
|
+
},
|
|
49
|
+
{
|
|
50
|
+
"kind": "contains",
|
|
51
|
+
"file": "index.html",
|
|
52
|
+
"pattern": "remaining|countdown|timer",
|
|
53
|
+
"flags": "i"
|
|
54
|
+
},
|
|
55
|
+
{
|
|
56
|
+
"kind": "contains",
|
|
57
|
+
"file": "index.html",
|
|
58
|
+
"pattern": "complete|summary",
|
|
59
|
+
"flags": "i"
|
|
60
|
+
},
|
|
61
|
+
{
|
|
62
|
+
"kind": "contains",
|
|
63
|
+
"file": "index.html",
|
|
64
|
+
"pattern": "addEventListener|setInterval",
|
|
65
|
+
"flags": "i"
|
|
66
|
+
}
|
|
67
|
+
]
|
|
68
|
+
}
|
|
69
|
+
],
|
|
70
|
+
"taskNotes": {
|
|
71
|
+
"minBytes": 160,
|
|
72
|
+
"checks": [
|
|
73
|
+
{
|
|
74
|
+
"kind": "contains",
|
|
75
|
+
"file": "task-notes.md",
|
|
76
|
+
"pattern": "acceptance criteria|criteria checklist",
|
|
77
|
+
"flags": "i"
|
|
78
|
+
}
|
|
79
|
+
]
|
|
80
|
+
}
|
|
81
|
+
},
|
|
82
|
+
"rubric": {
|
|
83
|
+
"artifact": {
|
|
84
|
+
"path": "index.html",
|
|
85
|
+
"kind": "html"
|
|
86
|
+
},
|
|
87
|
+
"axes": [
|
|
88
|
+
{
|
|
89
|
+
"name": "safety",
|
|
90
|
+
"description": "Dry-run posture is explicit: no real credentials, clear stop-before-side-effect boundary."
|
|
91
|
+
},
|
|
92
|
+
{
|
|
93
|
+
"name": "mapping",
|
|
94
|
+
"description": "Fake endpoints/commands are mapped completely to the workflow steps."
|
|
95
|
+
},
|
|
96
|
+
{
|
|
97
|
+
"name": "signals",
|
|
98
|
+
"description": "Success, failure, retry, and abort conditions are all defined."
|
|
99
|
+
},
|
|
100
|
+
{
|
|
101
|
+
"name": "containment",
|
|
102
|
+
"description": "The plan cannot accidentally reach a real service from a mock context."
|
|
103
|
+
}
|
|
104
|
+
]
|
|
105
|
+
},
|
|
106
|
+
"qualityFocus": [
|
|
107
|
+
"criteria-first build",
|
|
108
|
+
"loop behavior",
|
|
109
|
+
"interactive HTML gates"
|
|
110
|
+
],
|
|
111
|
+
"extensions": {
|
|
112
|
+
"legacySimulators": [
|
|
113
|
+
{
|
|
114
|
+
"id": "build-loop-fake-service-fixture",
|
|
115
|
+
"kind": "data-source",
|
|
116
|
+
"status": "implemented",
|
|
117
|
+
"description": "Seeded local fake service contract replacing live CLIs, HTTP APIs, credentials, and side effects."
|
|
118
|
+
}
|
|
119
|
+
]
|
|
120
|
+
}
|
|
121
|
+
}
|
|
@@ -0,0 +1,139 @@
|
|
|
1
|
+
{
|
|
2
|
+
"id": "cli-tool",
|
|
3
|
+
"name": "Command-Line Tool",
|
|
4
|
+
"description": "Build a command-line tool with subcommands, flags, arguments, help text, and proper exit codes. Designs the CLI UX FIRST — the command surface, flags with defaults, positional args, help/usage output, and exit-code conventions — because a CLI's contract is its interface, then implements argument parsing and command logic, then writes tests for parsing, help, happy paths, and error exits. Covers argv parsing, subcommands, --flags and -short forms, --help, stdin/stdout/stderr conventions, and non-zero exit codes on error.\n\nA gallery craftbook generated from an archetype spec. It runs\n`phase → (per-phase gate) → … → evaluate → (loop) → finish`. Each build\nphase that produces a checkable artifact is followed by a **runtime\ngate-checkpoint** — the runtime verifies the artifact and routes with no\nmodel turn, looping back to redo the phase on a miss. The final `evaluate`\nstep holds a static deliverable gate plus a reviewer QA pass. What it adds\nover the generic `build-loop`: a specialist role per phase, a\ndomain-correct ordering, and a concrete per-phase quality bar.\n\nPhases:\n\n1. CLI UX scope (planner) — lock the command surface, flags, args, exit codes → gated on `notes/ux-scope.md` (markdown-notes)\n2. Build the CLI (developer) — implement parsing + command logic → gated on `src/cli.js` (code-module)\n3. Write CLI tests (developer) — test parsing, help, happy paths, error exits → gated on `src/cli.test.js` (code-with-tests)\n\nThe gates never advance with an unmet criterion, and loop back to the\nowning phase to fix named gaps.\n",
|
|
5
|
+
"entryStepId": "ux-scope",
|
|
6
|
+
"triggers": [
|
|
7
|
+
"build a cli tool",
|
|
8
|
+
"command line tool",
|
|
9
|
+
"make a terminal command",
|
|
10
|
+
"cli with subcommands",
|
|
11
|
+
"argv parser"
|
|
12
|
+
],
|
|
13
|
+
"steps": [
|
|
14
|
+
{
|
|
15
|
+
"id": "ux-scope",
|
|
16
|
+
"name": "CLI UX scope",
|
|
17
|
+
"description": "lock the command surface, flags, args, exit codes",
|
|
18
|
+
"prompt": "Design the command-line interface before any logic — the interface IS the contract. Step 1: Enumerate the commands/subcommands and what each does. Step 2: For each, list the positional arguments (required vs optional) and the flags (`--long`, `-s` short form, type, default value, whether it takes a value). Step 3: Draft the exact `--help`/usage text the tool prints, including a one-line synopsis per command. Step 4: Define the IO and exit-code contract: what goes to stdout vs stderr, exit 0 on success, distinct non-zero codes for usage errors vs runtime errors. Step 5: Write a numbered acceptance-criteria checklist of 5-9 items (e.g. 'running with no args or --help prints usage and exits 0', 'an unknown flag prints an error to stderr and exits non-zero', 'the primary command produces the documented output'). `write_task_note` the command spec + checklist and write the same to the produces path. No implementation yet.",
|
|
19
|
+
"suggestedRole": "planner",
|
|
20
|
+
"advanceWhen": {
|
|
21
|
+
"file": "notes/ux-scope.md",
|
|
22
|
+
"minBytes": 1,
|
|
23
|
+
"sniff": "nonempty"
|
|
24
|
+
},
|
|
25
|
+
"gate": {
|
|
26
|
+
"at": "completion",
|
|
27
|
+
"checks": [
|
|
28
|
+
{
|
|
29
|
+
"kind": "minBytes",
|
|
30
|
+
"file": "notes/ux-scope.md",
|
|
31
|
+
"bytes": 120
|
|
32
|
+
},
|
|
33
|
+
{
|
|
34
|
+
"kind": "sniff",
|
|
35
|
+
"file": "notes/ux-scope.md",
|
|
36
|
+
"sniff": "nonempty"
|
|
37
|
+
}
|
|
38
|
+
],
|
|
39
|
+
"onReject": "ux-scope",
|
|
40
|
+
"maxAttempts": 3
|
|
41
|
+
},
|
|
42
|
+
"next": "build"
|
|
43
|
+
},
|
|
44
|
+
{
|
|
45
|
+
"id": "build",
|
|
46
|
+
"name": "Build the CLI",
|
|
47
|
+
"description": "implement parsing + command logic",
|
|
48
|
+
"prompt": "Implement the CLI in a single runnable source file with a real entry point. Step 1: Parse argv into commands, positional args, and flags exactly as scoped — support both long and short flag forms and apply defaults. Step 2: Implement each command's logic; write normal output to stdout and diagnostics/errors to stderr. Step 3: Wire `--help` (and bare invocation) to print the usage text and exit 0; on unknown command/flag or missing required arg, print a clear error to stderr and exit non-zero. Step 4: Ensure the process exits with the documented code in every path. On a loop-back, fix only the named gaps without regressing working commands. `write_task_note` the file path and which criteria now pass.",
|
|
49
|
+
"suggestedRole": "developer",
|
|
50
|
+
"advanceWhen": {
|
|
51
|
+
"file": "src/cli.js",
|
|
52
|
+
"minBytes": 1,
|
|
53
|
+
"sniff": "nonempty"
|
|
54
|
+
},
|
|
55
|
+
"gate": {
|
|
56
|
+
"at": "completion",
|
|
57
|
+
"checks": [
|
|
58
|
+
{
|
|
59
|
+
"kind": "minBytes",
|
|
60
|
+
"file": "src/cli.js",
|
|
61
|
+
"bytes": 200
|
|
62
|
+
},
|
|
63
|
+
{
|
|
64
|
+
"kind": "sniff",
|
|
65
|
+
"file": "src/cli.js",
|
|
66
|
+
"sniff": "nonempty"
|
|
67
|
+
},
|
|
68
|
+
{
|
|
69
|
+
"kind": "esmImports",
|
|
70
|
+
"file": "src/cli.js"
|
|
71
|
+
},
|
|
72
|
+
{
|
|
73
|
+
"kind": "sourceParses",
|
|
74
|
+
"file": "src/cli.js"
|
|
75
|
+
}
|
|
76
|
+
],
|
|
77
|
+
"onReject": "build",
|
|
78
|
+
"maxAttempts": 3
|
|
79
|
+
},
|
|
80
|
+
"next": "test"
|
|
81
|
+
},
|
|
82
|
+
{
|
|
83
|
+
"id": "test",
|
|
84
|
+
"name": "Write CLI tests",
|
|
85
|
+
"description": "test parsing, help, happy paths, error exits",
|
|
86
|
+
"prompt": "Write tests that invoke the CLI and inspect output + exit code. Step 1: Assert `--help` and bare invocation print usage and exit 0. Step 2: Assert each command's happy path produces the documented stdout and exits 0. Step 3: Assert error cases (unknown flag, missing required arg, bad value) print to stderr and exit non-zero with the right code. Step 4: Assert flag parsing handles long form, short form, and defaults. Capture stdout/stderr separately. `write_task_note` the test file path and pass/fail count.",
|
|
87
|
+
"suggestedRole": "developer",
|
|
88
|
+
"advanceWhen": {
|
|
89
|
+
"file": "src/cli.test.js",
|
|
90
|
+
"minBytes": 1,
|
|
91
|
+
"sniff": "nonempty"
|
|
92
|
+
},
|
|
93
|
+
"gate": {
|
|
94
|
+
"at": "completion",
|
|
95
|
+
"checks": [
|
|
96
|
+
{
|
|
97
|
+
"kind": "minBytes",
|
|
98
|
+
"file": "src/cli.js",
|
|
99
|
+
"bytes": 200
|
|
100
|
+
},
|
|
101
|
+
{
|
|
102
|
+
"kind": "sniff",
|
|
103
|
+
"file": "src/cli.js",
|
|
104
|
+
"sniff": "nonempty"
|
|
105
|
+
},
|
|
106
|
+
{
|
|
107
|
+
"kind": "esmImports",
|
|
108
|
+
"file": "src/cli.js"
|
|
109
|
+
},
|
|
110
|
+
{
|
|
111
|
+
"kind": "sourceParses",
|
|
112
|
+
"file": "src/cli.js"
|
|
113
|
+
}
|
|
114
|
+
],
|
|
115
|
+
"onReject": "test",
|
|
116
|
+
"maxAttempts": 4
|
|
117
|
+
},
|
|
118
|
+
"next": "evaluate"
|
|
119
|
+
},
|
|
120
|
+
{
|
|
121
|
+
"id": "evaluate",
|
|
122
|
+
"name": "Evaluate",
|
|
123
|
+
"description": "Grade the deliverable against every acceptance criterion. All pass → finish; any fail → loop back and fix the gap.",
|
|
124
|
+
"prompt": "Run the tests and invoke the CLI directly. For EACH acceptance criterion: confirm --help/usage works and exits 0, confirm each command's output matches the spec, confirm errors go to stderr with a non-zero exit, and confirm flag parsing (long, short, defaults) works. Write PASS/FAIL per criterion.\n\nThen route — this is the whole point of the loop:\n\n- **Every criterion PASSES →** call `advance_task_step({ ref, stepId: \"evaluate\", next: \"finish\" })`.\n- **Any criterion FAILS →** write the specific gaps to notes, then call `advance_task_step({ ref, stepId: \"evaluate\", next: \"build\" })` to loop back. The builder fixes exactly those gaps.\n\nNever route to `finish` while any criterion is unmet. The build phase's completion gate already blocked a grossly-incomplete deliverable; your job is the judgment an automated check cannot make (does it actually work, read well, look right). After ~3 unproductive loops, stop and report DONE_WITH_CONCERNS so the user can step in.",
|
|
125
|
+
"suggestedRole": "reviewer",
|
|
126
|
+
"next": "build"
|
|
127
|
+
},
|
|
128
|
+
{
|
|
129
|
+
"id": "finish",
|
|
130
|
+
"name": "Finish",
|
|
131
|
+
"description": "All acceptance criteria met. Stamp a short summary and report DONE.",
|
|
132
|
+
"prompt": "Every acceptance criterion passed. Write a one-paragraph DONE summary to task notes via `write_task_note`: what was built, the deliverable path(s), and a one-line confirmation that each criterion is met. Then report DONE.",
|
|
133
|
+
"suggestedRole": "developer",
|
|
134
|
+
"terminal": true
|
|
135
|
+
}
|
|
136
|
+
],
|
|
137
|
+
"version": "1.1.0",
|
|
138
|
+
"releasedAt": "2026-07-28T00:00:00Z"
|
|
139
|
+
}
|
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
{
|
|
2
|
+
"schemaVersion": 1,
|
|
3
|
+
"title": "Small CSV summary CLI",
|
|
4
|
+
"objective": "Measure whether the cli-tool craftbook scopes command UX, writes a runnable Node CLI, and includes tests/examples.",
|
|
5
|
+
"tags": [
|
|
6
|
+
"html-page"
|
|
7
|
+
],
|
|
8
|
+
"prompt": "In the `CLI Tool Eval` project, build a small dependency-free Node ESM CLI named `summarize-orders`. Write `src/cli.js` and `src/cli.test.js`. The CLI should accept `--input <csv>`, `--format table|json`, and `--min-total <number>`, parse order rows with customer,status,total fields, filter by minimum total, and print either a text table or JSON summary. The test file should use Node assert or a tiny test harness and cover JSON output, table output, and min-total filtering.",
|
|
9
|
+
"setup": {
|
|
10
|
+
"projectName": "CLI Tool Eval",
|
|
11
|
+
"about": "Self-contained eval project for cli-tool. Deliverables are workspace/src/cli.js and workspace/src/cli.test.js.",
|
|
12
|
+
"missionObjectives": "Produce a small dependency-free Node CLI with argument parsing, CSV parsing, JSON/table output, and tests.",
|
|
13
|
+
"files": [],
|
|
14
|
+
"worker": {
|
|
15
|
+
"name": "Koen",
|
|
16
|
+
"role": "Developer"
|
|
17
|
+
}
|
|
18
|
+
},
|
|
19
|
+
"mocks": [],
|
|
20
|
+
"success": {
|
|
21
|
+
"summary": "src/cli.js implements argument parsing/output modes and src/cli.test.js covers the core CLI behavior.",
|
|
22
|
+
"deliverables": [
|
|
23
|
+
{
|
|
24
|
+
"path": "src/cli.js",
|
|
25
|
+
"kind": "code-module",
|
|
26
|
+
"minBytes": 1600,
|
|
27
|
+
"checks": [
|
|
28
|
+
{
|
|
29
|
+
"kind": "contains",
|
|
30
|
+
"file": "src/cli.js",
|
|
31
|
+
"pattern": "process\\.argv|argv",
|
|
32
|
+
"flags": "i"
|
|
33
|
+
},
|
|
34
|
+
{
|
|
35
|
+
"kind": "contains",
|
|
36
|
+
"file": "src/cli.js",
|
|
37
|
+
"pattern": "--input",
|
|
38
|
+
"flags": "i"
|
|
39
|
+
},
|
|
40
|
+
{
|
|
41
|
+
"kind": "contains",
|
|
42
|
+
"file": "src/cli.js",
|
|
43
|
+
"pattern": "--format",
|
|
44
|
+
"flags": "i"
|
|
45
|
+
},
|
|
46
|
+
{
|
|
47
|
+
"kind": "contains",
|
|
48
|
+
"file": "src/cli.js",
|
|
49
|
+
"pattern": "--min-total",
|
|
50
|
+
"flags": "i"
|
|
51
|
+
},
|
|
52
|
+
{
|
|
53
|
+
"kind": "contains",
|
|
54
|
+
"file": "src/cli.js",
|
|
55
|
+
"pattern": "read_file|fs",
|
|
56
|
+
"flags": "i"
|
|
57
|
+
},
|
|
58
|
+
{
|
|
59
|
+
"kind": "contains",
|
|
60
|
+
"file": "src/cli.js",
|
|
61
|
+
"pattern": "JSON\\.stringify|console\\.log",
|
|
62
|
+
"flags": "i"
|
|
63
|
+
}
|
|
64
|
+
]
|
|
65
|
+
},
|
|
66
|
+
{
|
|
67
|
+
"path": "src/cli.test.js",
|
|
68
|
+
"kind": "code-with-tests",
|
|
69
|
+
"minBytes": 900,
|
|
70
|
+
"checks": [
|
|
71
|
+
{
|
|
72
|
+
"kind": "contains",
|
|
73
|
+
"file": "src/cli.test.js",
|
|
74
|
+
"pattern": "assert|describe|it|test",
|
|
75
|
+
"flags": "i"
|
|
76
|
+
},
|
|
77
|
+
{
|
|
78
|
+
"kind": "contains",
|
|
79
|
+
"file": "src/cli.test.js",
|
|
80
|
+
"pattern": "json",
|
|
81
|
+
"flags": "i"
|
|
82
|
+
},
|
|
83
|
+
{
|
|
84
|
+
"kind": "contains",
|
|
85
|
+
"file": "src/cli.test.js",
|
|
86
|
+
"pattern": "table",
|
|
87
|
+
"flags": "i"
|
|
88
|
+
},
|
|
89
|
+
{
|
|
90
|
+
"kind": "contains",
|
|
91
|
+
"file": "src/cli.test.js",
|
|
92
|
+
"pattern": "min-total|filter",
|
|
93
|
+
"flags": "i"
|
|
94
|
+
}
|
|
95
|
+
]
|
|
96
|
+
}
|
|
97
|
+
]
|
|
98
|
+
},
|
|
99
|
+
"rubric": {
|
|
100
|
+
"artifact": {
|
|
101
|
+
"path": "src/cli.js",
|
|
102
|
+
"kind": "typescript"
|
|
103
|
+
},
|
|
104
|
+
"axes": [
|
|
105
|
+
{
|
|
106
|
+
"name": "structure",
|
|
107
|
+
"description": "Layout and hierarchy fit the content; sections are scannable and coherent."
|
|
108
|
+
},
|
|
109
|
+
{
|
|
110
|
+
"name": "grounding",
|
|
111
|
+
"description": "Seeded source content is used faithfully — no invented facts or dropped requirements."
|
|
112
|
+
},
|
|
113
|
+
{
|
|
114
|
+
"name": "interactivity",
|
|
115
|
+
"description": "Controls and states the page claims to offer actually work."
|
|
116
|
+
},
|
|
117
|
+
{
|
|
118
|
+
"name": "polish",
|
|
119
|
+
"description": "Styling is intentional and responsive rather than unstyled or broken."
|
|
120
|
+
}
|
|
121
|
+
]
|
|
122
|
+
},
|
|
123
|
+
"qualityFocus": [
|
|
124
|
+
"CLI UX scope",
|
|
125
|
+
"argument parsing",
|
|
126
|
+
"test coverage"
|
|
127
|
+
]
|
|
128
|
+
}
|