@bendyline/gilde 0.1.54 → 0.1.56
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/authoring/chat-models/glm-5.3-flash-320b-q2.json +129 -0
- package/authoring/chat-models/qwen3.5-27b-q4.json +147 -0
- package/authoring/gstack/evals/cso.json +21 -18
- package/authoring/gstack/evals/investigate.json +25 -22
- package/authoring/gstack/evals/plan-ceo-review.json +19 -16
- package/authoring/gstack/overlays/retro.json +2 -2
- package/authoring/gstack/wave.json +2 -2
- package/data/chat-models/gl/glm-5.3-flash-320b-q2/manifest.json +137 -0
- package/data/chat-models/gl/glm-5.3-flash-320b-q2/versions/1.0.0/manifest.json +23 -0
- package/data/chat-models/index.json +1 -1
- package/data/chat-models/qw/qwen3.5-27b-q4/manifest.json +217 -0
- package/data/chat-models/qw/qwen3.5-27b-q4/versions/1.0.0/manifest.json +90 -0
- package/data/craftbook-templates/a1/a11y-audit/versions/1.1.3/test.json +16 -8
- package/data/craftbook-templates/ac/accessibility-retrofit/versions/2.0.2/craftbook.json +618 -0
- package/data/craftbook-templates/ac/accessibility-retrofit/versions/2.0.2/test.json +228 -0
- package/data/craftbook-templates/al/album-curate/versions/1.0.3/test.json +6 -3
- package/data/craftbook-templates/al/alert-rules/versions/1.0.3/test.json +8 -8
- package/data/craftbook-templates/an/anniversary-cut/versions/1.0.5/test.json +10 -5
- package/data/craftbook-templates/an/annual-document-review/versions/1.0.3/test.json +16 -8
- package/data/craftbook-templates/an/anomaly-scan/versions/1.0.3/test.json +12 -6
- package/data/craftbook-templates/ap/apply-review-findings/versions/1.0.2/craftbook.json +602 -0
- package/data/craftbook-templates/ap/apply-review-findings/versions/1.0.2/test.json +290 -0
- package/data/craftbook-templates/ar/artifact-integrity-review/versions/1.0.3/test.json +18 -9
- package/data/craftbook-templates/au/audiobook-master-pack/versions/1.0.5/test.json +10 -5
- package/data/craftbook-templates/au/auth-flow/versions/1.0.3/test.json +12 -6
- package/data/craftbook-templates/au/auth-flow/versions/1.0.4/craftbook.json +282 -0
- package/data/craftbook-templates/au/auth-flow/versions/1.0.4/test.json +143 -0
- package/data/craftbook-templates/au/automation-recipe/versions/1.0.3/test.json +12 -6
- package/data/craftbook-templates/ba/backup-routine/versions/1.0.4/craftbook.json +282 -0
- package/data/craftbook-templates/ba/backup-routine/versions/1.0.4/test.json +108 -0
- package/data/craftbook-templates/bl/blog-post/versions/1.0.3/test.json +8 -8
- package/data/craftbook-templates/br/branding-website/versions/1.1.3/craftbook.json +335 -0
- package/data/craftbook-templates/br/branding-website/versions/1.1.3/test.json +164 -0
- package/data/craftbook-templates/br/broadcast-announcement/versions/1.0.3/test.json +8 -8
- package/data/craftbook-templates/br/browser-qa-audit/versions/2.0.7/craftbook.json +620 -0
- package/data/craftbook-templates/br/browser-qa-audit/versions/2.0.7/test.json +376 -0
- package/data/craftbook-templates/br/browser-qa-audit/versions/2.0.8/craftbook.json +621 -0
- package/data/craftbook-templates/br/browser-qa-audit/versions/2.0.8/test.json +376 -0
- package/data/craftbook-templates/bu/bug-fix-tdd/versions/2.0.2/craftbook.json +730 -0
- package/data/craftbook-templates/bu/bug-fix-tdd/versions/2.0.2/test.json +221 -0
- package/data/craftbook-templates/bu/build-loop/versions/1.1.1/test.json +8 -8
- package/data/craftbook-templates/ca/case-study/versions/1.0.3/test.json +8 -8
- package/data/craftbook-templates/ch/chapter-narration-run/versions/1.0.4/test.json +10 -5
- package/data/craftbook-templates/ch/character-turnaround/versions/1.0.3/test.json +25 -11
- package/data/craftbook-templates/ci/ci-pipeline/versions/2.0.2/craftbook.json +587 -0
- package/data/craftbook-templates/ci/ci-pipeline/versions/2.0.2/test.json +225 -0
- package/data/craftbook-templates/ci/citation-audit/versions/1.0.3/test.json +28 -14
- package/data/craftbook-templates/cl/cli-tool/versions/1.1.3/test.json +8 -8
- package/data/craftbook-templates/co/codebase-refactoring-review/versions/1.0.3/test.json +16 -8
- package/data/craftbook-templates/co/codebase-ux-review/versions/1.0.3/test.json +18 -9
- package/data/craftbook-templates/co/codemod-sweep/versions/1.0.2/craftbook.json +605 -0
- package/data/craftbook-templates/co/codemod-sweep/versions/1.0.2/test.json +252 -0
- package/data/craftbook-templates/co/cohort-analysis/versions/1.0.3/test.json +18 -9
- package/data/craftbook-templates/co/competitive-analysis/versions/1.0.3/test.json +23 -15
- package/data/craftbook-templates/co/content-accuracy-review/versions/1.0.3/test.json +28 -14
- package/data/craftbook-templates/co/content-deck/versions/1.2.3/craftbook.json +4 -4
- package/data/craftbook-templates/co/content-deck/versions/1.2.3/test.json +3 -0
- package/data/craftbook-templates/co/content-deck/versions/1.2.4/craftbook.json +341 -0
- package/data/craftbook-templates/co/content-deck/versions/1.2.4/test.json +169 -0
- package/data/craftbook-templates/co/copy-review/versions/1.0.3/test.json +14 -7
- package/data/craftbook-templates/co/corpus-email-digest/versions/1.2.3/craftbook.json +4 -4
- package/data/craftbook-templates/co/corpus-email-digest/versions/1.2.3/test.json +3 -0
- package/data/craftbook-templates/co/corpus-email-digest/versions/1.2.4/craftbook.json +345 -0
- package/data/craftbook-templates/co/corpus-email-digest/versions/1.2.4/test.json +187 -0
- package/data/craftbook-templates/co/corpus-synthesis/versions/1.0.3/test.json +26 -13
- package/data/craftbook-templates/co/cover-letter/versions/1.0.3/test.json +8 -8
- package/data/craftbook-templates/cs/csv-transformer/versions/1.2.4/test.json +7 -7
- package/data/craftbook-templates/da/dashboard-spec/versions/1.0.3/test.json +7 -7
- package/data/craftbook-templates/da/data-export-migrate/versions/1.0.4/craftbook.json +285 -0
- package/data/craftbook-templates/da/data-export-migrate/versions/1.0.4/test.json +104 -0
- package/data/craftbook-templates/da/data-pipeline-etl/versions/1.0.5/craftbook.json +284 -0
- package/data/craftbook-templates/da/data-pipeline-etl/versions/1.0.5/test.json +162 -0
- package/data/craftbook-templates/da/data-quality-audit/versions/1.0.3/test.json +38 -19
- package/data/craftbook-templates/da/data-to-report/versions/1.0.3/test.json +12 -6
- package/data/craftbook-templates/db/db-index-tuning/versions/1.0.4/craftbook.json +286 -0
- package/data/craftbook-templates/db/db-index-tuning/versions/1.0.4/test.json +104 -0
- package/data/craftbook-templates/de/dependency-audit/versions/1.1.3/test.json +14 -7
- package/data/craftbook-templates/de/dependency-upgrade/versions/1.0.2/craftbook.json +621 -0
- package/data/craftbook-templates/de/dependency-upgrade/versions/1.0.2/test.json +235 -0
- package/data/craftbook-templates/de/design-review/versions/1.0.3/test.json +16 -8
- package/data/craftbook-templates/de/design-system-consultation/versions/2.0.7/craftbook.json +616 -0
- package/data/craftbook-templates/de/design-system-consultation/versions/2.0.7/test.json +201 -0
- package/data/craftbook-templates/de/design-system-consultation/versions/2.0.8/craftbook.json +616 -0
- package/data/craftbook-templates/de/design-system-consultation/versions/2.0.8/test.json +201 -0
- package/data/craftbook-templates/do/doc-intake-pipeline/versions/1.0.4/craftbook.json +285 -0
- package/data/craftbook-templates/do/doc-intake-pipeline/versions/1.0.4/test.json +104 -0
- package/data/craftbook-templates/do/doc-rewrite/versions/1.0.3/test.json +119 -55
- package/data/craftbook-templates/do/dockerize-app/versions/1.0.4/craftbook.json +285 -0
- package/data/craftbook-templates/do/dockerize-app/versions/1.0.4/test.json +92 -0
- package/data/craftbook-templates/do/documentation-drift-review/versions/1.0.3/test.json +18 -9
- package/data/craftbook-templates/dr/draft-social-post/versions/1.0.2/craftbook.json +324 -0
- package/data/craftbook-templates/dr/draft-social-post/versions/1.0.2/test.json +161 -0
- package/data/craftbook-templates/ed/edit-notes/versions/1.0.3/test.json +11 -4
- package/data/craftbook-templates/en/engineering-retrospective/versions/2.0.7/craftbook.json +566 -0
- package/data/craftbook-templates/en/engineering-retrospective/versions/2.0.7/test.json +191 -0
- package/data/craftbook-templates/en/engineering-retrospective/versions/2.0.8/craftbook.json +566 -0
- package/data/craftbook-templates/en/engineering-retrospective/versions/2.0.8/test.json +191 -0
- package/data/craftbook-templates/ep/episode-plan/versions/1.0.3/test.json +10 -5
- package/data/craftbook-templates/ex/executive-level-review/versions/2.0.7/craftbook.json +595 -0
- package/data/craftbook-templates/ex/executive-level-review/versions/2.0.7/test.json +136 -0
- package/data/craftbook-templates/ex/executive-level-review/versions/2.0.8/craftbook.json +595 -0
- package/data/craftbook-templates/ex/executive-level-review/versions/2.0.8/test.json +139 -0
- package/data/craftbook-templates/ex/expense-categorize/versions/1.0.4/craftbook.json +277 -0
- package/data/craftbook-templates/ex/expense-categorize/versions/1.0.4/test.json +104 -0
- package/data/craftbook-templates/fe/feature-flag-rollout/versions/1.1.2/craftbook.json +327 -0
- package/data/craftbook-templates/fe/feature-flag-rollout/versions/1.1.2/test.json +162 -0
- package/data/craftbook-templates/fl/flaky-test-fix/versions/1.0.2/craftbook.json +718 -0
- package/data/craftbook-templates/fl/flaky-test-fix/versions/1.0.2/test.json +239 -0
- package/data/craftbook-templates/fo/form-fill-batch/versions/1.0.4/craftbook.json +285 -0
- package/data/craftbook-templates/fo/form-fill-batch/versions/1.0.4/test.json +162 -0
- package/data/craftbook-templates/fo/form-wizard/versions/1.0.3/test.json +8 -8
- package/data/craftbook-templates/ho/holiday-card-run/versions/1.0.3/test.json +11 -4
- package/data/craftbook-templates/ho/hotfix-flow/versions/2.0.2/craftbook.json +610 -0
- package/data/craftbook-templates/ho/hotfix-flow/versions/2.0.2/test.json +220 -0
- package/data/craftbook-templates/ht/html-arcade-game/versions/1.2.3/craftbook.json +343 -0
- package/data/craftbook-templates/ht/html-arcade-game/versions/1.2.3/test.json +176 -0
- package/data/craftbook-templates/id/idea-office-hours/versions/2.0.7/craftbook.json +566 -0
- package/data/craftbook-templates/id/idea-office-hours/versions/2.0.7/test.json +141 -0
- package/data/craftbook-templates/id/idea-office-hours/versions/2.0.8/craftbook.json +566 -0
- package/data/craftbook-templates/id/idea-office-hours/versions/2.0.8/test.json +141 -0
- package/data/craftbook-templates/im/image-set-index/versions/1.2.3/craftbook.json +298 -0
- package/data/craftbook-templates/im/image-set-index/versions/1.2.3/test.json +180 -0
- package/data/craftbook-templates/in/insurance-inventory/versions/1.0.3/test.json +11 -4
- package/data/craftbook-templates/in/interactive-quiz/versions/1.0.3/test.json +48 -23
- package/data/craftbook-templates/in/invitation-design/versions/1.0.3/test.json +1 -1
- package/data/craftbook-templates/index.json +1 -1
- package/data/craftbook-templates/it/item-intake/versions/1.0.3/test.json +11 -4
- package/data/craftbook-templates/le/lead-enrichment/versions/1.0.4/craftbook.json +277 -0
- package/data/craftbook-templates/le/lead-enrichment/versions/1.0.4/test.json +104 -0
- package/data/craftbook-templates/me/meeting-prep-brief/versions/1.0.3/test.json +16 -8
- package/data/craftbook-templates/me/memory-prompt-session/versions/1.0.3/test.json +3 -0
- package/data/craftbook-templates/mo/morning-report/versions/1.0.3/test.json +21 -9
- package/data/craftbook-templates/of/office-hours/versions/1.1.1/test.json +8 -8
- package/data/craftbook-templates/pe/perf-optimization/versions/2.0.2/craftbook.json +697 -0
- package/data/craftbook-templates/pe/perf-optimization/versions/2.0.2/test.json +235 -0
- package/data/craftbook-templates/pe/pest-diagnosis/versions/1.0.3/test.json +19 -8
- package/data/craftbook-templates/pl/plan/versions/1.0.2/craftbook.json +222 -0
- package/data/craftbook-templates/pl/plan/versions/1.0.2/test.json +91 -0
- package/data/craftbook-templates/pl/playtest-report/versions/1.0.3/test.json +17 -7
- package/data/craftbook-templates/pr/practice-exam/versions/1.0.3/test.json +7 -2
- package/data/craftbook-templates/pr/practice-session/versions/1.0.3/test.json +11 -4
- package/data/craftbook-templates/pu/pull-request-review/versions/1.9.2/craftbook.json +435 -0
- package/data/craftbook-templates/pu/pull-request-review/versions/1.9.2/test.json +141 -0
- package/data/craftbook-templates/qu/quarterly-summary/versions/1.0.3/test.json +16 -8
- package/data/craftbook-templates/re/receipt-intake/versions/1.0.3/test.json +11 -4
- package/data/craftbook-templates/re/receipt-ocr-ledger/versions/1.0.4/craftbook.json +277 -0
- package/data/craftbook-templates/re/receipt-ocr-ledger/versions/1.0.4/test.json +104 -0
- package/data/craftbook-templates/re/recipe-capture/versions/1.0.3/test.json +13 -5
- package/data/craftbook-templates/re/record-a-relative/versions/1.0.3/test.json +3 -0
- package/data/craftbook-templates/re/recurring-invoice-run/versions/1.0.4/craftbook.json +285 -0
- package/data/craftbook-templates/re/recurring-invoice-run/versions/1.0.4/test.json +104 -0
- package/data/craftbook-templates/re/refactor-module/versions/2.0.2/craftbook.json +695 -0
- package/data/craftbook-templates/re/refactor-module/versions/2.0.2/test.json +241 -0
- package/data/craftbook-templates/re/release-artifact-sanity-check/versions/1.0.4/craftbook.json +361 -0
- package/data/craftbook-templates/re/release-artifact-sanity-check/versions/1.0.4/test.json +138 -0
- package/data/craftbook-templates/re/release-notes/versions/1.0.4/test.json +8 -8
- package/data/craftbook-templates/re/release-notes/versions/1.0.5/craftbook.json +339 -0
- package/data/craftbook-templates/re/release-notes/versions/1.0.5/test.json +111 -0
- package/data/craftbook-templates/re/reliability-review/versions/1.0.3/test.json +16 -8
- package/data/craftbook-templates/re/research-report/versions/1.0.3/test.json +16 -8
- package/data/craftbook-templates/ro/root-cause-investigation/versions/2.0.7/craftbook.json +595 -0
- package/data/craftbook-templates/ro/root-cause-investigation/versions/2.0.7/test.json +157 -0
- package/data/craftbook-templates/ro/root-cause-investigation/versions/2.0.8/craftbook.json +596 -0
- package/data/craftbook-templates/ro/root-cause-investigation/versions/2.0.8/test.json +157 -0
- package/data/craftbook-templates/ro/rough-cut-assembly/versions/1.0.4/test.json +13 -5
- package/data/craftbook-templates/sc/schema-migration/versions/2.0.2/craftbook.json +624 -0
- package/data/craftbook-templates/sc/schema-migration/versions/2.0.2/test.json +225 -0
- package/data/craftbook-templates/sc/script-automation/versions/1.0.4/craftbook.json +284 -0
- package/data/craftbook-templates/sc/script-automation/versions/1.0.4/test.json +162 -0
- package/data/craftbook-templates/se/security-architecture-review/versions/2.0.7/craftbook.json +597 -0
- package/data/craftbook-templates/se/security-architecture-review/versions/2.0.7/test.json +154 -0
- package/data/craftbook-templates/se/security-architecture-review/versions/2.0.8/craftbook.json +597 -0
- package/data/craftbook-templates/se/security-architecture-review/versions/2.0.8/test.json +157 -0
- package/data/craftbook-templates/sh/ship/versions/1.1.2/craftbook.json +346 -0
- package/data/craftbook-templates/sh/ship/versions/1.1.2/test.json +228 -0
- package/data/craftbook-templates/so/social-digest/versions/1.0.3/craftbook.json +239 -0
- package/data/craftbook-templates/so/social-digest/versions/1.0.3/test.json +168 -0
- package/data/craftbook-templates/so/source-quality-audit/versions/1.0.3/test.json +25 -11
- package/data/craftbook-templates/sp/spec-authoring/versions/2.0.7/craftbook.json +599 -0
- package/data/craftbook-templates/sp/spec-authoring/versions/2.0.7/test.json +162 -0
- package/data/craftbook-templates/sp/spec-authoring/versions/2.0.8/craftbook.json +599 -0
- package/data/craftbook-templates/sp/spec-authoring/versions/2.0.8/test.json +162 -0
- package/data/craftbook-templates/te/technical-documentation/versions/2.0.7/craftbook.json +577 -0
- package/data/craftbook-templates/te/technical-documentation/versions/2.0.7/test.json +174 -0
- package/data/craftbook-templates/te/technical-documentation/versions/2.0.8/craftbook.json +577 -0
- package/data/craftbook-templates/te/technical-documentation/versions/2.0.8/test.json +174 -0
- package/data/craftbook-templates/te/test-coverage-review/versions/1.2.1/test.json +16 -8
- package/data/craftbook-templates/te/test-suite-backfill/versions/2.0.2/craftbook.json +592 -0
- package/data/craftbook-templates/te/test-suite-backfill/versions/2.0.2/test.json +195 -0
- package/data/craftbook-templates/th/thank-you-batch/versions/1.0.3/test.json +6 -3
- package/data/craftbook-templates/th/thank-you-sweep/versions/1.0.3/test.json +11 -4
- package/data/craftbook-templates/ti/tileset-batch/versions/1.0.3/test.json +9 -3
- package/data/craftbook-templates/tr/transcribe-and-shownotes/versions/1.0.3/test.json +10 -5
- package/data/craftbook-templates/tr/transcribe-index/versions/1.0.3/test.json +6 -3
- package/data/craftbook-templates/tr/translate-content/versions/1.1.3/craftbook.json +140 -0
- package/data/craftbook-templates/tr/translate-content/versions/1.1.3/test.json +141 -0
- package/data/craftbook-templates/ty/type-safety-pass/versions/2.0.2/craftbook.json +594 -0
- package/data/craftbook-templates/ty/type-safety-pass/versions/2.0.2/test.json +221 -0
- package/data/craftbook-templates/ux/ux-update/versions/1.0.2/craftbook.json +606 -0
- package/data/craftbook-templates/ux/ux-update/versions/1.0.2/test.json +206 -0
- package/data/craftbook-templates/we/weak-spot-drill/versions/1.0.3/test.json +11 -4
- package/data/craftbook-templates/we/weekly-walkthrough/versions/1.0.3/test.json +21 -9
- package/data/craftbook-templates/wh/whitepaper/versions/1.0.3/test.json +16 -8
- package/data/craftbook-templates/ye/year-in-review/versions/1.0.3/test.json +8 -4
- package/data/project-types/index.json +1 -1
- package/data/project-types/ju/just-chat/manifest.json +1 -1
- package/data/project-types/ju/just-chat/versions/1.0.1/about.md +5 -0
- package/data/project-types/ju/just-chat/versions/1.0.1/manifest.json +19 -0
- package/package.json +1 -1
- package/schemas/chat-model-identity.schema.json +27 -0
- package/schemas/chat-model-version.schema.json +22 -0
|
@@ -0,0 +1,235 @@
|
|
|
1
|
+
{
|
|
2
|
+
"schemaVersion": 1,
|
|
3
|
+
"title": "Measured near-linear fix for a quadratic hot path",
|
|
4
|
+
"objective": "The craftbook drives a measured optimization of a seeded O(n^2) duplicate finder: baseline numbers recorded with their method, one bottleneck fixed at the real site, the same measurement re-run to prove the gain, and an enforced review — verified mechanically via a deterministic instrumented op-count budget, reference-equal results, and a green-suite run.",
|
|
5
|
+
"tags": [
|
|
6
|
+
"code",
|
|
7
|
+
"performance",
|
|
8
|
+
"tactical-fleet"
|
|
9
|
+
],
|
|
10
|
+
"prompt": "Use the Targeted Performance Fix craftbook for this: support says importing a 200-row sheet crawls, and profiling points at findDuplicates - a 200-item list burns about 40,000 comparisons. The brief is in source/perf-brief.md. Get findDuplicates near-linear without changing its results.",
|
|
11
|
+
"setup": {
|
|
12
|
+
"projectName": "Sheet Importer",
|
|
13
|
+
"about": "A tiny sheet-import module with an instrumented duplicate finder. Tests run with `npm run test` (node --test).",
|
|
14
|
+
"files": [
|
|
15
|
+
{
|
|
16
|
+
"path": "package.json",
|
|
17
|
+
"content": "{\n \"name\": \"sheet-importer\",\n \"private\": true,\n \"version\": \"1.0.0\",\n \"type\": \"module\",\n \"scripts\": {\n \"test\": \"node --test\"\n }\n}\n"
|
|
18
|
+
},
|
|
19
|
+
{
|
|
20
|
+
"path": "src/instrument.js",
|
|
21
|
+
"content": "let comparisons = 0;\n\nexport function countComparison() {\n comparisons += 1;\n}\n\nexport function comparisonCount() {\n return comparisons;\n}\n\nexport function resetComparisons() {\n comparisons = 0;\n}\n"
|
|
22
|
+
},
|
|
23
|
+
{
|
|
24
|
+
"path": "src/find-duplicates.js",
|
|
25
|
+
"content": "import { countComparison } from './instrument.js';\n\nexport function findDuplicates(items) {\n const duplicates = [];\n for (let i = 0; i < items.length; i += 1) {\n let count = 0;\n for (let j = 0; j < items.length; j += 1) {\n countComparison();\n if (items[j] === items[i]) count += 1;\n }\n if (count > 1 && !duplicates.includes(items[i])) duplicates.push(items[i]);\n }\n return duplicates;\n}\n"
|
|
26
|
+
},
|
|
27
|
+
{
|
|
28
|
+
"path": "tests/duplicates.test.mjs",
|
|
29
|
+
"content": "import assert from 'node:assert/strict';\nimport { test } from 'node:test';\nimport { findDuplicates } from '../src/find-duplicates.js';\n\ntest('reports each duplicate once, in first-appearance order', () => {\n assert.deepEqual(findDuplicates(['a', 'b', 'a', 'c', 'b']), ['a', 'b']);\n});\n\ntest('returns empty for unique lists', () => {\n assert.deepEqual(findDuplicates(['x', 'y', 'z']), []);\n});\n\ntest('order pins to first appearance, not second occurrence', () => {\n assert.deepEqual(findDuplicates(['a', 'b', 'b', 'a']), ['a', 'b']);\n});\n"
|
|
30
|
+
},
|
|
31
|
+
{
|
|
32
|
+
"path": "source/perf-brief.md",
|
|
33
|
+
"content": "# Performance brief\n\nSupport reports that importing a 200-row sheet crawls. Profiling points at `findDuplicates` in src/find-duplicates.js: for a 200-item list it performs about 40,000 instrumented comparisons (200 x 200 full rescans). The counter lives in src/instrument.js - `comparisonCount()` reads it, `resetComparisons()` clears it.\n\nTarget: get `findDuplicates` near-linear - well under 2,000 instrumented comparisons for a 200-item list - WITHOUT changing its results. Each duplicated value is reported once, in order of FIRST appearance.\n\nRules:\n\n- src/instrument.js is measurement infrastructure; leave it exactly as it is.\n- Keep the measurement honest: keep calling `countComparison()` for every equality comparison the new implementation still performs. Deleting instrumentation instead of deleting work is measurement theater.\n- Measure with a repeatable command: a small script that builds a 200-item list, calls `findDuplicates`, and prints `comparisonCount()`.\n"
|
|
34
|
+
},
|
|
35
|
+
{
|
|
36
|
+
"path": "tests/verify-perf.mjs",
|
|
37
|
+
"content": "import assert from 'node:assert/strict';\nimport { execFileSync } from 'node:child_process';\nimport { findDuplicates } from '../src/find-duplicates.js';\nimport { comparisonCount, resetComparisons } from '../src/instrument.js';\n\n// Independent reference: each duplicated value once, first-appearance order.\nfunction referenceDuplicates(items) {\n const counts = new Map();\n for (const item of items) counts.set(item, (counts.get(item) ?? 0) + 1);\n const out = [];\n const added = new Set();\n for (const item of items) {\n if ((counts.get(item) ?? 0) > 1 && !added.has(item)) {\n added.add(item);\n out.push(item);\n }\n }\n return out;\n}\n\n// 1) Results identical to reference on small pinned cases.\nassert.deepEqual(findDuplicates(['a', 'b', 'a', 'c', 'b']), ['a', 'b']);\nassert.deepEqual(findDuplicates([]), []);\nassert.deepEqual(findDuplicates(['a', 'b', 'b', 'a']), ['a', 'b']);\n\n// 2) Deterministic 200-item workload: results equal reference AND the\n// instrumented op-count is under the stated budget (seeded code: 40000).\nconst items = [];\nfor (let i = 0; i < 200; i += 1) items.push('v' + ((i * 7) % 149));\nresetComparisons();\nconst result = findDuplicates(items);\nassert.deepEqual(result, referenceDuplicates(items), 'results changed - speed that changes answers is a bug');\nconsole.log('PERF_OPTIMIZATION_ORACLE results ok');\nconst ops = comparisonCount();\nassert.ok(\n ops <= 2000,\n 'findDuplicates on 200 items performed ' + ops + ' instrumented comparisons - the budget is 2000 and the seeded code did 40000',\n);\nconsole.log('PERF_OPTIMIZATION_ORACLE ops ' + ops + ' within budget');\n\n// 3) The whole suite is green on the optimized tree (real run).\nfunction runSuite(cwd) {\n try {\n execFileSync(process.execPath, ['--test'], { cwd, stdio: 'pipe', timeout: 30000 });\n return 0;\n } catch (err) {\n return typeof err.status === 'number' ? err.status : 1;\n }\n}\nassert.equal(runSuite(process.cwd()), 0, 'the suite must be green on the optimized tree');\nconsole.log('PERF_OPTIMIZATION_ORACLE suite green');\nconsole.log('PERF_OPTIMIZATION_ORACLE done');\n",
|
|
38
|
+
"surface": "harness"
|
|
39
|
+
}
|
|
40
|
+
],
|
|
41
|
+
"craftbookParams": {
|
|
42
|
+
"scope": "findDuplicates in src/find-duplicates.js does about 40,000 instrumented comparisons for a 200-item list; get it near-linear (well under 2,000 comparisons) without changing its results. See source/perf-brief.md."
|
|
43
|
+
}
|
|
44
|
+
},
|
|
45
|
+
"mocks": [],
|
|
46
|
+
"success": {
|
|
47
|
+
"summary": "The quadratic rescan is gone, the instrumented op-count for 200 items is under budget, results are identical to the reference on a deterministic workload, the suite is green on a real run, and every fleet artifact exists with honest numbers and citations.",
|
|
48
|
+
"deliverables": [
|
|
49
|
+
{
|
|
50
|
+
"path": "{{task.dir}}/baseline.md",
|
|
51
|
+
"kind": "markdown-notes",
|
|
52
|
+
"artifact": true,
|
|
53
|
+
"minBytes": 500,
|
|
54
|
+
"checks": [
|
|
55
|
+
{
|
|
56
|
+
"kind": "contains",
|
|
57
|
+
"file": "{{task.dir}}/baseline.md",
|
|
58
|
+
"pattern": "^##\\s+Current behavior[\\s\\S]*^##\\s+Measurements[\\s\\S]*^##\\s+Guardrail",
|
|
59
|
+
"flags": "im",
|
|
60
|
+
"label": "baseline sections in order"
|
|
61
|
+
},
|
|
62
|
+
{
|
|
63
|
+
"kind": "contains",
|
|
64
|
+
"file": "{{task.dir}}/baseline.md",
|
|
65
|
+
"pattern": "^##\\s+Measurements[\\s\\S]*\\d",
|
|
66
|
+
"flags": "im",
|
|
67
|
+
"label": "baseline numbers recorded"
|
|
68
|
+
},
|
|
69
|
+
{
|
|
70
|
+
"kind": "citationsResolve",
|
|
71
|
+
"file": "{{task.dir}}/baseline.md",
|
|
72
|
+
"minCitations": 2,
|
|
73
|
+
"artifact": true
|
|
74
|
+
}
|
|
75
|
+
]
|
|
76
|
+
},
|
|
77
|
+
{
|
|
78
|
+
"path": "{{task.dir}}/plan.md",
|
|
79
|
+
"kind": "markdown-notes",
|
|
80
|
+
"artifact": true,
|
|
81
|
+
"minBytes": 500,
|
|
82
|
+
"checks": [
|
|
83
|
+
{
|
|
84
|
+
"kind": "contains",
|
|
85
|
+
"file": "{{task.dir}}/plan.md",
|
|
86
|
+
"pattern": "^##\\s+Target[\\s\\S]*^##\\s+Stages[\\s\\S]*^##\\s+Acceptance criteria",
|
|
87
|
+
"flags": "im",
|
|
88
|
+
"label": "plan sections in order"
|
|
89
|
+
},
|
|
90
|
+
{
|
|
91
|
+
"kind": "citationsResolve",
|
|
92
|
+
"file": "{{task.dir}}/plan.md",
|
|
93
|
+
"minCitations": 2,
|
|
94
|
+
"artifact": true
|
|
95
|
+
}
|
|
96
|
+
]
|
|
97
|
+
},
|
|
98
|
+
{
|
|
99
|
+
"path": "{{task.dir}}/change-notes.md",
|
|
100
|
+
"kind": "markdown-notes",
|
|
101
|
+
"artifact": true,
|
|
102
|
+
"minBytes": 600,
|
|
103
|
+
"checks": [
|
|
104
|
+
{
|
|
105
|
+
"kind": "contains",
|
|
106
|
+
"file": "{{task.dir}}/change-notes.md",
|
|
107
|
+
"pattern": "^##\\s+Stages executed[\\s\\S]*^##\\s+Files touched[\\s\\S]*^##\\s+Deviations from plan",
|
|
108
|
+
"flags": "im",
|
|
109
|
+
"label": "change-notes sections in order"
|
|
110
|
+
},
|
|
111
|
+
{
|
|
112
|
+
"kind": "citationsResolve",
|
|
113
|
+
"file": "{{task.dir}}/change-notes.md",
|
|
114
|
+
"minCitations": 2,
|
|
115
|
+
"artifact": true
|
|
116
|
+
}
|
|
117
|
+
]
|
|
118
|
+
},
|
|
119
|
+
{
|
|
120
|
+
"path": "{{task.dir}}/verification.md",
|
|
121
|
+
"kind": "markdown-notes",
|
|
122
|
+
"artifact": true,
|
|
123
|
+
"minBytes": 400,
|
|
124
|
+
"checks": [
|
|
125
|
+
{
|
|
126
|
+
"kind": "contains",
|
|
127
|
+
"file": "{{task.dir}}/verification.md",
|
|
128
|
+
"pattern": "^##\\s+Before\\s*/\\s*after[\\s\\S]*^##\\s+Suite[\\s\\S]*^##\\s+Result",
|
|
129
|
+
"flags": "im",
|
|
130
|
+
"label": "verification sections in order"
|
|
131
|
+
},
|
|
132
|
+
{
|
|
133
|
+
"kind": "contains",
|
|
134
|
+
"file": "{{task.dir}}/verification.md",
|
|
135
|
+
"pattern": "^##\\s+Before\\s*/\\s*after[\\s\\S]*\\d",
|
|
136
|
+
"flags": "im",
|
|
137
|
+
"label": "both numbers quoted"
|
|
138
|
+
}
|
|
139
|
+
]
|
|
140
|
+
},
|
|
141
|
+
{
|
|
142
|
+
"path": "{{task.dir}}/review.md",
|
|
143
|
+
"kind": "markdown-report",
|
|
144
|
+
"artifact": true,
|
|
145
|
+
"minBytes": 400,
|
|
146
|
+
"checks": [
|
|
147
|
+
{
|
|
148
|
+
"kind": "contains",
|
|
149
|
+
"file": "{{task.dir}}/review.md",
|
|
150
|
+
"pattern": "Verdict:\\s*(?:PASS|REVISE)",
|
|
151
|
+
"flags": "i",
|
|
152
|
+
"label": "explicit reviewer verdict"
|
|
153
|
+
}
|
|
154
|
+
]
|
|
155
|
+
}
|
|
156
|
+
],
|
|
157
|
+
"checks": [
|
|
158
|
+
{
|
|
159
|
+
"kind": "notContains",
|
|
160
|
+
"file": "src/find-duplicates.js",
|
|
161
|
+
"pattern": "for\\s*\\(let j = 0; j < items\\.length",
|
|
162
|
+
"label": "the seeded inner full-list rescan is gone"
|
|
163
|
+
},
|
|
164
|
+
{
|
|
165
|
+
"kind": "sourceParses",
|
|
166
|
+
"file": "src/find-duplicates.js"
|
|
167
|
+
},
|
|
168
|
+
{
|
|
169
|
+
"kind": "nodeScriptPasses",
|
|
170
|
+
"script": "tests/verify-perf.mjs",
|
|
171
|
+
"timeoutMs": 60000,
|
|
172
|
+
"requiredOutput": [
|
|
173
|
+
{
|
|
174
|
+
"pattern": "PERF_OPTIMIZATION_ORACLE results ok",
|
|
175
|
+
"label": "results identical to the reference"
|
|
176
|
+
},
|
|
177
|
+
{
|
|
178
|
+
"pattern": "PERF_OPTIMIZATION_ORACLE ops \\d+ within budget",
|
|
179
|
+
"label": "instrumented op-count under the stated budget"
|
|
180
|
+
},
|
|
181
|
+
{
|
|
182
|
+
"pattern": "PERF_OPTIMIZATION_ORACLE suite green",
|
|
183
|
+
"label": "suite green on the optimized tree"
|
|
184
|
+
}
|
|
185
|
+
]
|
|
186
|
+
}
|
|
187
|
+
],
|
|
188
|
+
"taskNotes": {
|
|
189
|
+
"minBytes": 120,
|
|
190
|
+
"checks": [
|
|
191
|
+
{
|
|
192
|
+
"kind": "contains",
|
|
193
|
+
"file": "task-notes.md",
|
|
194
|
+
"pattern": "\\bDONE\\b[\\s\\S]*(npm\\s+run\\s+test|node\\s+--test)",
|
|
195
|
+
"flags": "i",
|
|
196
|
+
"label": "DONE note names the real test command"
|
|
197
|
+
}
|
|
198
|
+
]
|
|
199
|
+
},
|
|
200
|
+
"taskGraph": {
|
|
201
|
+
"requireCraftbookTask": true,
|
|
202
|
+
"requireTerminalStep": true
|
|
203
|
+
},
|
|
204
|
+
"unchangedFixtures": [
|
|
205
|
+
"source/perf-brief.md",
|
|
206
|
+
"package.json",
|
|
207
|
+
"src/instrument.js"
|
|
208
|
+
]
|
|
209
|
+
},
|
|
210
|
+
"rubric": {
|
|
211
|
+
"artifact": {
|
|
212
|
+
"path": "{{task.dir}}/verification.md",
|
|
213
|
+
"kind": "markdown"
|
|
214
|
+
},
|
|
215
|
+
"axes": [
|
|
216
|
+
{
|
|
217
|
+
"name": "Measurement honesty",
|
|
218
|
+
"description": "Baseline numbers were gathered by a stated, repeatable method before the change, and verification re-ran the identical method and quoted both figures side by side - no adjectives standing in for numbers, no swapped workload."
|
|
219
|
+
},
|
|
220
|
+
{
|
|
221
|
+
"name": "Bottleneck focus",
|
|
222
|
+
"description": "One bottleneck was named with its mechanism and fixed at the real site by that mechanism - not a synthetic proxy, a weakened measurement, or scattered micro-tweaks."
|
|
223
|
+
},
|
|
224
|
+
{
|
|
225
|
+
"name": "Correctness under speed",
|
|
226
|
+
"description": "Results stayed identical on the same inputs, the suite stayed green between stages with real receipts, and residual risk or unverified claims are labeled honestly."
|
|
227
|
+
}
|
|
228
|
+
]
|
|
229
|
+
},
|
|
230
|
+
"qualityFocus": [
|
|
231
|
+
"numbers with their method",
|
|
232
|
+
"same-method re-measurement",
|
|
233
|
+
"enforced review loop"
|
|
234
|
+
]
|
|
235
|
+
}
|
|
@@ -27,6 +27,9 @@
|
|
|
27
27
|
"worker": {
|
|
28
28
|
"name": "Wim",
|
|
29
29
|
"role": "Gardener"
|
|
30
|
+
},
|
|
31
|
+
"craftbookParams": {
|
|
32
|
+
"workPath": "tasks/eval"
|
|
30
33
|
}
|
|
31
34
|
},
|
|
32
35
|
"mocks": [],
|
|
@@ -42,51 +45,59 @@
|
|
|
42
45
|
"kind": "contains",
|
|
43
46
|
"file": "tasks/eval/report.md",
|
|
44
47
|
"pattern": "(?:^|\\n)#{1,3}\\s+\\S",
|
|
45
|
-
"flags": "i"
|
|
48
|
+
"flags": "i",
|
|
49
|
+
"artifact": true
|
|
46
50
|
},
|
|
47
51
|
{
|
|
48
52
|
"kind": "contains",
|
|
49
53
|
"file": "tasks/eval/report.md",
|
|
50
54
|
"pattern": "likely|probable|possible|suspect",
|
|
51
55
|
"flags": "i",
|
|
52
|
-
"label": "candidates stated as likelihoods, not verdicts"
|
|
56
|
+
"label": "candidates stated as likelihoods, not verdicts",
|
|
57
|
+
"artifact": true
|
|
53
58
|
},
|
|
54
59
|
{
|
|
55
60
|
"kind": "contains",
|
|
56
61
|
"file": "tasks/eval/report.md",
|
|
57
62
|
"pattern": "caterpillar",
|
|
58
63
|
"flags": "i",
|
|
59
|
-
"label": "engages the caterpillar evidence from the kale photo note"
|
|
64
|
+
"label": "engages the caterpillar evidence from the kale photo note",
|
|
65
|
+
"artifact": true
|
|
60
66
|
},
|
|
61
67
|
{
|
|
62
68
|
"kind": "contains",
|
|
63
69
|
"file": "tasks/eval/report.md",
|
|
64
70
|
"pattern": "treat",
|
|
65
71
|
"flags": "i",
|
|
66
|
-
"label": "treatment ladder present"
|
|
72
|
+
"label": "treatment ladder present",
|
|
73
|
+
"artifact": true
|
|
67
74
|
},
|
|
68
75
|
{
|
|
69
76
|
"kind": "contains",
|
|
70
77
|
"file": "tasks/eval/report.md",
|
|
71
78
|
"pattern": "monitor|watch|check",
|
|
72
79
|
"flags": "i",
|
|
73
|
-
"label": "monitoring plan present"
|
|
80
|
+
"label": "monitoring plan present",
|
|
81
|
+
"artifact": true
|
|
74
82
|
},
|
|
75
83
|
{
|
|
76
84
|
"kind": "notContains",
|
|
77
85
|
"file": "tasks/eval/report.md",
|
|
78
86
|
"pattern": "(definitely|guaranteed|certainly) (is|will cure)",
|
|
79
87
|
"flags": "i",
|
|
80
|
-
"label": "diagnosis stays hedged, never absolute"
|
|
88
|
+
"label": "diagnosis stays hedged, never absolute",
|
|
89
|
+
"artifact": true
|
|
81
90
|
},
|
|
82
91
|
{
|
|
83
92
|
"kind": "contains",
|
|
84
93
|
"file": "tasks/eval/report.md",
|
|
85
94
|
"pattern": "extension|professional|nursery|expert",
|
|
86
95
|
"flags": "i",
|
|
87
|
-
"label": "escalation path named"
|
|
96
|
+
"label": "escalation path named",
|
|
97
|
+
"artifact": true
|
|
88
98
|
}
|
|
89
|
-
]
|
|
99
|
+
],
|
|
100
|
+
"artifact": true
|
|
90
101
|
}
|
|
91
102
|
]
|
|
92
103
|
},
|
|
@@ -0,0 +1,222 @@
|
|
|
1
|
+
{
|
|
2
|
+
"id": "plan",
|
|
3
|
+
"name": "Plan",
|
|
4
|
+
"description": "Author a PLAN for a piece of work as a reviewable **draft task** — the gezel equivalent of `/plan` in other tools.\n\nUnlike a normal craftbook, this one's deliverable is not a workspace file: it's a high-quality *draft task* (a plan) that the user reviews and then activates to run. Invoke it via the `start_plan` tool, which creates the draft and an authoring task that walks this craftbook. The authoring gates inspect the draft being built (via the authoring task's `craftbookParams.draftRef`), so a weak model can't skip the parts that make a plan good.\n\nSteps:\n\n1. **Frame the goal** (planner) — write a strong \"about\" on the draft and set 3–8 concrete **outcomes** (what should be created or updated at success). Gated by `checkPlanFramed`.\n2. **Design the build steps** (planner) — add ordered build steps to the draft, each given a concrete static deliverable + enforced gate via `set_step_deliverable`. Gated by `checkStepsHaveDeliverables`.\n3. **Add the verification step** (planner) — append a terminal step (via `add_verification_step`) that confirms every outcome with evidence before the task can close. Gated by `checkVerificationStep`.\n4. **Review the plan** (planner) — sanity-check the whole graph. Gated by `checkPlanReady`, which returns one list of any remaining gaps.\n5. **Hand off to the user** (planner, terminal) — summarize the plan and tell the user to review and **activate** it.\n\nWhen activated (draft → active), the produced task runs its build steps — each looping back on a missed deliverable — and its terminal step verifies the outcomes were kept.\n",
|
|
5
|
+
"entryStepId": "frame",
|
|
6
|
+
"triggers": [
|
|
7
|
+
"plan this",
|
|
8
|
+
"make a plan",
|
|
9
|
+
"plan for",
|
|
10
|
+
"/plan"
|
|
11
|
+
],
|
|
12
|
+
"steps": [
|
|
13
|
+
{
|
|
14
|
+
"id": "frame",
|
|
15
|
+
"name": "Frame the goal",
|
|
16
|
+
"description": "Write a strong 'about' and set the outcomes on the draft plan.",
|
|
17
|
+
"prompt": "You are drafting a PLAN. The plan is a separate **draft task** — its ref is in your task's `craftbookParams.draftRef` (shown in your task description / entry note). Everything you author here targets that draft using the exact draft ref.\n\n**Do two things:**\n\n1. **Write a strong \"about\".** Call `update_task({ ref: \"<draftRef>\", description })` with a clear job-to-be-done from the user's perspective — what success looks like, in 120+ characters.\n2. **Set the outcomes.** Call `set_outcomes({ task: \"<draftRef>\", outcomes: [ ... ] })` with 3–8 concrete, individually-verifiable statements of what should be **created or updated** at completion (e.g. \"An index.html with a playable snake game and a game-over/restart screen\", \"A README documenting how to run it\").\n\nWhen done, call `advance_task_step({ ref, stepId: \"frame\", next: \"outline\" })`. The gate rejects until the draft has a real about and at least 3 outcomes — fix what it names and advance again.",
|
|
18
|
+
"suggestedRole": "planner",
|
|
19
|
+
"gate": {
|
|
20
|
+
"at": "completion",
|
|
21
|
+
"scripts": [
|
|
22
|
+
{
|
|
23
|
+
"name": "checkPlanFramed",
|
|
24
|
+
"scope": "standard",
|
|
25
|
+
"inputs": {}
|
|
26
|
+
}
|
|
27
|
+
],
|
|
28
|
+
"onReject": "frame",
|
|
29
|
+
"maxAttempts": 4
|
|
30
|
+
},
|
|
31
|
+
"next": "outline",
|
|
32
|
+
"toolPolicy": {
|
|
33
|
+
"disallowBuiltinToolsets": [
|
|
34
|
+
"ai-apps",
|
|
35
|
+
"archives",
|
|
36
|
+
"artifacts",
|
|
37
|
+
"audio",
|
|
38
|
+
"browser-automation",
|
|
39
|
+
"code-execution",
|
|
40
|
+
"craftbooks",
|
|
41
|
+
"data-tables",
|
|
42
|
+
"entity-intel",
|
|
43
|
+
"git",
|
|
44
|
+
"image-intel",
|
|
45
|
+
"images",
|
|
46
|
+
"role-delegation",
|
|
47
|
+
"role-delegation-escalation",
|
|
48
|
+
"security-intel",
|
|
49
|
+
"team-management",
|
|
50
|
+
"videos",
|
|
51
|
+
"web",
|
|
52
|
+
"workspace-fs-write"
|
|
53
|
+
],
|
|
54
|
+
"outputMedium": "none"
|
|
55
|
+
}
|
|
56
|
+
},
|
|
57
|
+
{
|
|
58
|
+
"id": "outline",
|
|
59
|
+
"name": "Design the build steps",
|
|
60
|
+
"description": "Add ordered build steps to the draft, each gated on a concrete static deliverable.",
|
|
61
|
+
"prompt": "Design the implementation as an ordered series of **build steps** on the draft (`craftbookParams.draftRef`). Each build step must produce a **concrete static file** with an enforced gate.\n\nThe draft starts with one placeholder step named **\"Implement\"**. First call `craftbook_read({ task: \"<draftRef>\" })` and note the actual id of that placeholder step. Then either:\n- repurpose that exact step with `craftbook_update_step({ task: \"<draftRef>\", stepId: \"<actualId>\", name, prompt, suggestedRole })`, then immediately call `set_step_deliverable({ task: \"<draftRef>\", stepId: \"<actualId>\", path, kind })`; or\n- remove that exact step with `craftbook_remove_step({ task: \"<draftRef>\", stepId: \"<actualId>\" })` after you have added the real build steps.\n\nFor each added step, in order:\n1. `craftbook_add_step({ task: \"<draftRef>\", name, prompt, suggestedRole })` — a concrete, instructive step (e.g. name \"Build the game\", role \"developer\", a prompt that says exactly what to make).\n2. Call `craftbook_read({ task: \"<draftRef>\" })` and use the actual id of the step you just added.\n3. `set_step_deliverable({ task: \"<draftRef>\", stepId: \"<actualId>\", path, kind })` — declare the named file it produces and auto-attach the gate. Pick `kind` to match the artifact: `html-page` / `html-game` / `markdown-report` / `code-with-tests` / `json` / etc.\n\nDo not invent step ids like `build` unless `craftbook_read` shows that exact id. Do not advance while any build step lacks `advanceWhen`/`gate`. Order the steps so each builds on the last. When every build step has a deliverable, call `advance_task_step({ ref, stepId: \"outline\", next: \"verify-step\" })`. The gate lists any step still missing a deliverable.",
|
|
62
|
+
"suggestedRole": "planner",
|
|
63
|
+
"gate": {
|
|
64
|
+
"at": "completion",
|
|
65
|
+
"scripts": [
|
|
66
|
+
{
|
|
67
|
+
"name": "checkStepsHaveDeliverables",
|
|
68
|
+
"scope": "standard",
|
|
69
|
+
"inputs": {}
|
|
70
|
+
}
|
|
71
|
+
],
|
|
72
|
+
"onReject": "outline",
|
|
73
|
+
"maxAttempts": 4
|
|
74
|
+
},
|
|
75
|
+
"next": "verify-step",
|
|
76
|
+
"toolPolicy": {
|
|
77
|
+
"disallowBuiltinToolsets": [
|
|
78
|
+
"ai-apps",
|
|
79
|
+
"archives",
|
|
80
|
+
"artifacts",
|
|
81
|
+
"audio",
|
|
82
|
+
"browser-automation",
|
|
83
|
+
"code-execution",
|
|
84
|
+
"data-tables",
|
|
85
|
+
"entity-intel",
|
|
86
|
+
"git",
|
|
87
|
+
"image-intel",
|
|
88
|
+
"images",
|
|
89
|
+
"role-delegation",
|
|
90
|
+
"role-delegation-escalation",
|
|
91
|
+
"security-intel",
|
|
92
|
+
"team-management",
|
|
93
|
+
"videos",
|
|
94
|
+
"web",
|
|
95
|
+
"workspace-fs-write"
|
|
96
|
+
],
|
|
97
|
+
"outputMedium": "none"
|
|
98
|
+
}
|
|
99
|
+
},
|
|
100
|
+
{
|
|
101
|
+
"id": "verify-step",
|
|
102
|
+
"name": "Add the verification step",
|
|
103
|
+
"description": "Append a terminal verification step that checks the outcomes were kept.",
|
|
104
|
+
"prompt": "Add the final **verification step** to the draft (`craftbookParams.draftRef`): call `add_verification_step({ task: \"<draftRef>\" })`. This appends a terminal step — gated by `checkOutcomesMet` — that makes the executor confirm each outcome (with evidence) before the task can close. Pass a custom `prompt` only if you need domain-specific verification instructions.\n\nThen call `advance_task_step({ ref, stepId: \"verify-step\", next: \"review\" })`. The gate rejects until the draft has a terminal step that verifies the outcomes.",
|
|
105
|
+
"suggestedRole": "planner",
|
|
106
|
+
"gate": {
|
|
107
|
+
"at": "completion",
|
|
108
|
+
"scripts": [
|
|
109
|
+
{
|
|
110
|
+
"name": "checkVerificationStep",
|
|
111
|
+
"scope": "standard",
|
|
112
|
+
"inputs": {}
|
|
113
|
+
}
|
|
114
|
+
],
|
|
115
|
+
"onReject": "verify-step",
|
|
116
|
+
"maxAttempts": 4
|
|
117
|
+
},
|
|
118
|
+
"next": "review",
|
|
119
|
+
"toolPolicy": {
|
|
120
|
+
"disallowBuiltinToolsets": [
|
|
121
|
+
"ai-apps",
|
|
122
|
+
"archives",
|
|
123
|
+
"artifacts",
|
|
124
|
+
"audio",
|
|
125
|
+
"browser-automation",
|
|
126
|
+
"code-execution",
|
|
127
|
+
"craftbooks",
|
|
128
|
+
"data-tables",
|
|
129
|
+
"entity-intel",
|
|
130
|
+
"git",
|
|
131
|
+
"image-intel",
|
|
132
|
+
"images",
|
|
133
|
+
"role-delegation",
|
|
134
|
+
"role-delegation-escalation",
|
|
135
|
+
"security-intel",
|
|
136
|
+
"team-management",
|
|
137
|
+
"videos",
|
|
138
|
+
"web",
|
|
139
|
+
"workspace-fs-write"
|
|
140
|
+
],
|
|
141
|
+
"outputMedium": "none"
|
|
142
|
+
}
|
|
143
|
+
},
|
|
144
|
+
{
|
|
145
|
+
"id": "review",
|
|
146
|
+
"name": "Review the plan",
|
|
147
|
+
"description": "Sanity-check the whole plan before handing it to the user.",
|
|
148
|
+
"prompt": "Final check before handing the plan to the user. `craftbook_read({ task: \"<draftRef>\" })` and sanity-check: the steps are ordered sensibly, each build step has a deliverable, and the last step verifies the outcomes. Fix anything off with the `craftbook_*` / `set_step_deliverable` tools on the draft.\n\nWhen the plan is solid, call `advance_task_step({ ref, stepId: \"review\", next: \"done\" })`. The gate returns a single list of anything still missing (thin about, fewer than 3 outcomes, ungated steps, or no verification step).",
|
|
149
|
+
"suggestedRole": "planner",
|
|
150
|
+
"gate": {
|
|
151
|
+
"at": "completion",
|
|
152
|
+
"scripts": [
|
|
153
|
+
{
|
|
154
|
+
"name": "checkPlanReady",
|
|
155
|
+
"scope": "standard",
|
|
156
|
+
"inputs": {}
|
|
157
|
+
}
|
|
158
|
+
],
|
|
159
|
+
"onReject": "review",
|
|
160
|
+
"maxAttempts": 4
|
|
161
|
+
},
|
|
162
|
+
"next": "done",
|
|
163
|
+
"toolPolicy": {
|
|
164
|
+
"disallowBuiltinToolsets": [
|
|
165
|
+
"ai-apps",
|
|
166
|
+
"archives",
|
|
167
|
+
"artifacts",
|
|
168
|
+
"audio",
|
|
169
|
+
"browser-automation",
|
|
170
|
+
"code-execution",
|
|
171
|
+
"data-tables",
|
|
172
|
+
"entity-intel",
|
|
173
|
+
"git",
|
|
174
|
+
"image-intel",
|
|
175
|
+
"images",
|
|
176
|
+
"role-delegation",
|
|
177
|
+
"role-delegation-escalation",
|
|
178
|
+
"security-intel",
|
|
179
|
+
"team-management",
|
|
180
|
+
"videos",
|
|
181
|
+
"web",
|
|
182
|
+
"workspace-fs-write"
|
|
183
|
+
],
|
|
184
|
+
"outputMedium": "none"
|
|
185
|
+
}
|
|
186
|
+
},
|
|
187
|
+
{
|
|
188
|
+
"id": "done",
|
|
189
|
+
"name": "Hand off to the user",
|
|
190
|
+
"description": "Summarize the plan and tell the user to review and activate it.",
|
|
191
|
+
"prompt": "The plan is ready. Summarize it for the user in your reply: mention the draft task **<draftRef>** by its ref (so it renders as a reviewable plan card), and recap the about, the outcomes, and the build steps. Tell the user they can **review and activate** the plan to run it — or ask you to tweak it first. Then report DONE.",
|
|
192
|
+
"suggestedRole": "planner",
|
|
193
|
+
"terminal": true,
|
|
194
|
+
"toolPolicy": {
|
|
195
|
+
"disallowBuiltinToolsets": [
|
|
196
|
+
"ai-apps",
|
|
197
|
+
"archives",
|
|
198
|
+
"artifacts",
|
|
199
|
+
"audio",
|
|
200
|
+
"browser-automation",
|
|
201
|
+
"code-execution",
|
|
202
|
+
"craftbooks",
|
|
203
|
+
"data-tables",
|
|
204
|
+
"entity-intel",
|
|
205
|
+
"git",
|
|
206
|
+
"image-intel",
|
|
207
|
+
"images",
|
|
208
|
+
"role-delegation",
|
|
209
|
+
"role-delegation-escalation",
|
|
210
|
+
"security-intel",
|
|
211
|
+
"team-management",
|
|
212
|
+
"videos",
|
|
213
|
+
"web",
|
|
214
|
+
"workspace-fs-write"
|
|
215
|
+
],
|
|
216
|
+
"outputMedium": "none"
|
|
217
|
+
}
|
|
218
|
+
}
|
|
219
|
+
],
|
|
220
|
+
"version": "1.0.2",
|
|
221
|
+
"releasedAt": "2026-09-04T13:46:25Z"
|
|
222
|
+
}
|
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
{
|
|
2
|
+
"schemaVersion": 1,
|
|
3
|
+
"title": "Reviewable draft task plan",
|
|
4
|
+
"objective": "Measure whether the plan craftbook creates a real draft task with a substantial about, concrete outcomes, gated build steps, and a terminal verification step.",
|
|
5
|
+
"tags": [
|
|
6
|
+
"planning"
|
|
7
|
+
],
|
|
8
|
+
"prompt": "In the `Plan Eval` project, author a reviewable draft plan for building a self-contained `index.html` bug triage board for a small support team. The eventual board should let users add bugs, assign severity, filter open/closed items, and show a summary. Use the plan flow/start_plan rather than building the board now: the output should be a draft task with strong outcomes, ordered gated build steps, and a final verification step for the user to review and activate. Important: every non-terminal build step on the draft, including the initial placeholder step if you keep or rename it, must get `set_step_deliverable({ task: \"<draftRef>\", stepId, path: \"index.html\", kind: \"html-page\" })` before you finish.",
|
|
9
|
+
"setup": {
|
|
10
|
+
"projectName": "Plan Eval",
|
|
11
|
+
"about": "Self-contained eval project for the plan craftbook. The deliverable is a reviewable draft task, not workspace files.",
|
|
12
|
+
"missionObjectives": "Author a high-quality draft task plan for a bug triage board, including outcomes, gated build steps, and verification before activation. A build step is not gated until set_step_deliverable is called on that exact step.",
|
|
13
|
+
"files": [],
|
|
14
|
+
"worker": {
|
|
15
|
+
"name": "Pieter",
|
|
16
|
+
"role": "Planner"
|
|
17
|
+
}
|
|
18
|
+
},
|
|
19
|
+
"mocks": [],
|
|
20
|
+
"success": {
|
|
21
|
+
"summary": "A task sourced from the plan craftbook points to a draft task with outcomes, gated build steps, and terminal verification.",
|
|
22
|
+
"taskGraph": {
|
|
23
|
+
"checks": [
|
|
24
|
+
{
|
|
25
|
+
"kind": "contains",
|
|
26
|
+
"file": "task-graph.md",
|
|
27
|
+
"pattern": "index.html",
|
|
28
|
+
"flags": "i"
|
|
29
|
+
},
|
|
30
|
+
{
|
|
31
|
+
"kind": "contains",
|
|
32
|
+
"file": "task-graph.md",
|
|
33
|
+
"pattern": "bug|triage",
|
|
34
|
+
"flags": "i"
|
|
35
|
+
},
|
|
36
|
+
{
|
|
37
|
+
"kind": "contains",
|
|
38
|
+
"file": "task-graph.md",
|
|
39
|
+
"pattern": "severity",
|
|
40
|
+
"flags": "i"
|
|
41
|
+
},
|
|
42
|
+
{
|
|
43
|
+
"kind": "contains",
|
|
44
|
+
"file": "task-graph.md",
|
|
45
|
+
"pattern": "verify|outcome|evidence",
|
|
46
|
+
"flags": "i"
|
|
47
|
+
}
|
|
48
|
+
],
|
|
49
|
+
"requireCraftbookTask": true,
|
|
50
|
+
"requireDraftRef": true,
|
|
51
|
+
"draft": {
|
|
52
|
+
"status": "draft",
|
|
53
|
+
"minDescriptionBytes": 120,
|
|
54
|
+
"minOutcomes": 3,
|
|
55
|
+
"minSteps": 3,
|
|
56
|
+
"requireTerminalVerification": true,
|
|
57
|
+
"requireGatedBuildSteps": true
|
|
58
|
+
}
|
|
59
|
+
}
|
|
60
|
+
},
|
|
61
|
+
"rubric": {
|
|
62
|
+
"artifact": {
|
|
63
|
+
"path": "task-graph.md",
|
|
64
|
+
"kind": "markdown"
|
|
65
|
+
},
|
|
66
|
+
"axes": [
|
|
67
|
+
{
|
|
68
|
+
"name": "decomposition",
|
|
69
|
+
"description": "The draft breaks the goal into ordered steps a crew could actually execute."
|
|
70
|
+
},
|
|
71
|
+
{
|
|
72
|
+
"name": "outcomes",
|
|
73
|
+
"description": "Outcomes are concrete and checkable rather than restatements of the goal."
|
|
74
|
+
},
|
|
75
|
+
{
|
|
76
|
+
"name": "gating",
|
|
77
|
+
"description": "Build steps carry a deliverable so progress is verifiable, and verification comes last."
|
|
78
|
+
},
|
|
79
|
+
{
|
|
80
|
+
"name": "grounding",
|
|
81
|
+
"description": "The plan reflects the requested scope — no invented requirements, none dropped."
|
|
82
|
+
}
|
|
83
|
+
]
|
|
84
|
+
},
|
|
85
|
+
"qualityFocus": [
|
|
86
|
+
"task-native deliverable",
|
|
87
|
+
"outcomes",
|
|
88
|
+
"gated build steps",
|
|
89
|
+
"verification step"
|
|
90
|
+
]
|
|
91
|
+
}
|