@bendyline/gilde 0.1.55 → 0.1.57
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/authoring/chat-models/glm-5.3-flash-320b-q2.json +129 -0
- package/authoring/chat-models/qwen3.8-flash-next-iq3.json +198 -0
- package/authoring/chat-models/qwen3.8-flash-next-iq4.json +198 -0
- package/authoring/gstack/wave.json +2 -2
- package/data/chat-models/gl/glm-5.3-flash-320b-q2/manifest.json +137 -0
- package/data/chat-models/gl/glm-5.3-flash-320b-q2/versions/1.0.0/manifest.json +23 -0
- package/data/chat-models/index.json +1 -1
- package/data/chat-models/qw/qwen3.8-flash-next-iq3/manifest.json +221 -0
- package/data/chat-models/qw/qwen3.8-flash-next-iq3/versions/1.0.0/manifest.json +36 -0
- package/data/chat-models/qw/qwen3.8-flash-next-iq4/manifest.json +221 -0
- package/data/chat-models/qw/qwen3.8-flash-next-iq4/versions/1.0.0/manifest.json +36 -0
- package/data/craftbook-templates/a1/a11y-audit/versions/1.1.3/test.json +16 -8
- package/data/craftbook-templates/ac/accessibility-retrofit/versions/2.0.2/craftbook.json +618 -0
- package/data/craftbook-templates/ac/accessibility-retrofit/versions/2.0.2/test.json +228 -0
- package/data/craftbook-templates/al/album-curate/versions/1.0.3/test.json +6 -3
- package/data/craftbook-templates/al/alert-rules/versions/1.0.3/test.json +8 -8
- package/data/craftbook-templates/an/anniversary-cut/versions/1.0.5/test.json +10 -5
- package/data/craftbook-templates/an/annual-document-review/versions/1.0.3/test.json +16 -8
- package/data/craftbook-templates/an/anomaly-scan/versions/1.0.3/test.json +12 -6
- package/data/craftbook-templates/ap/apply-review-findings/versions/1.0.2/craftbook.json +602 -0
- package/data/craftbook-templates/ap/apply-review-findings/versions/1.0.2/test.json +290 -0
- package/data/craftbook-templates/ar/artifact-integrity-review/versions/1.0.3/test.json +18 -9
- package/data/craftbook-templates/au/audiobook-master-pack/versions/1.0.5/test.json +10 -5
- package/data/craftbook-templates/au/auth-flow/versions/1.0.3/test.json +12 -6
- package/data/craftbook-templates/au/auth-flow/versions/1.0.4/craftbook.json +282 -0
- package/data/craftbook-templates/au/auth-flow/versions/1.0.4/test.json +143 -0
- package/data/craftbook-templates/au/automation-recipe/versions/1.0.3/test.json +12 -6
- package/data/craftbook-templates/ba/backup-routine/versions/1.0.4/craftbook.json +282 -0
- package/data/craftbook-templates/ba/backup-routine/versions/1.0.4/test.json +108 -0
- package/data/craftbook-templates/bl/blog-post/versions/1.0.3/test.json +8 -8
- package/data/craftbook-templates/br/branding-website/versions/1.1.3/craftbook.json +335 -0
- package/data/craftbook-templates/br/branding-website/versions/1.1.3/test.json +164 -0
- package/data/craftbook-templates/br/broadcast-announcement/versions/1.0.3/test.json +8 -8
- package/data/craftbook-templates/br/browser-qa-audit/versions/2.0.8/craftbook.json +621 -0
- package/data/craftbook-templates/br/browser-qa-audit/versions/2.0.8/test.json +376 -0
- package/data/craftbook-templates/bu/bug-fix-tdd/versions/2.0.2/craftbook.json +730 -0
- package/data/craftbook-templates/bu/bug-fix-tdd/versions/2.0.2/test.json +221 -0
- package/data/craftbook-templates/bu/build-loop/versions/1.1.1/test.json +8 -8
- package/data/craftbook-templates/ca/case-study/versions/1.0.3/test.json +8 -8
- package/data/craftbook-templates/ch/chapter-narration-run/versions/1.0.4/test.json +10 -5
- package/data/craftbook-templates/ch/character-turnaround/versions/1.0.3/test.json +25 -11
- package/data/craftbook-templates/ci/ci-pipeline/versions/2.0.2/craftbook.json +587 -0
- package/data/craftbook-templates/ci/ci-pipeline/versions/2.0.2/test.json +225 -0
- package/data/craftbook-templates/ci/citation-audit/versions/1.0.3/test.json +28 -14
- package/data/craftbook-templates/cl/cli-tool/versions/1.1.3/test.json +8 -8
- package/data/craftbook-templates/co/codebase-refactoring-review/versions/1.0.3/test.json +16 -8
- package/data/craftbook-templates/co/codebase-ux-review/versions/1.0.3/test.json +18 -9
- package/data/craftbook-templates/co/codemod-sweep/versions/1.0.2/craftbook.json +605 -0
- package/data/craftbook-templates/co/codemod-sweep/versions/1.0.2/test.json +252 -0
- package/data/craftbook-templates/co/cohort-analysis/versions/1.0.3/test.json +18 -9
- package/data/craftbook-templates/co/competitive-analysis/versions/1.0.3/test.json +23 -15
- package/data/craftbook-templates/co/content-accuracy-review/versions/1.0.3/test.json +28 -14
- package/data/craftbook-templates/co/content-deck/versions/1.2.3/craftbook.json +4 -4
- package/data/craftbook-templates/co/content-deck/versions/1.2.3/test.json +3 -0
- package/data/craftbook-templates/co/content-deck/versions/1.2.4/craftbook.json +341 -0
- package/data/craftbook-templates/co/content-deck/versions/1.2.4/test.json +169 -0
- package/data/craftbook-templates/co/copy-review/versions/1.0.3/test.json +14 -7
- package/data/craftbook-templates/co/corpus-email-digest/versions/1.2.3/craftbook.json +4 -4
- package/data/craftbook-templates/co/corpus-email-digest/versions/1.2.3/test.json +3 -0
- package/data/craftbook-templates/co/corpus-email-digest/versions/1.2.4/craftbook.json +345 -0
- package/data/craftbook-templates/co/corpus-email-digest/versions/1.2.4/test.json +187 -0
- package/data/craftbook-templates/co/corpus-synthesis/versions/1.0.3/test.json +26 -13
- package/data/craftbook-templates/co/cover-letter/versions/1.0.3/test.json +8 -8
- package/data/craftbook-templates/cs/csv-transformer/versions/1.2.4/test.json +7 -7
- package/data/craftbook-templates/da/dashboard-spec/versions/1.0.3/test.json +7 -7
- package/data/craftbook-templates/da/data-export-migrate/versions/1.0.4/craftbook.json +285 -0
- package/data/craftbook-templates/da/data-export-migrate/versions/1.0.4/test.json +104 -0
- package/data/craftbook-templates/da/data-pipeline-etl/versions/1.0.5/craftbook.json +284 -0
- package/data/craftbook-templates/da/data-pipeline-etl/versions/1.0.5/test.json +162 -0
- package/data/craftbook-templates/da/data-quality-audit/versions/1.0.3/test.json +38 -19
- package/data/craftbook-templates/da/data-to-report/versions/1.0.3/test.json +12 -6
- package/data/craftbook-templates/db/db-index-tuning/versions/1.0.4/craftbook.json +286 -0
- package/data/craftbook-templates/db/db-index-tuning/versions/1.0.4/test.json +104 -0
- package/data/craftbook-templates/de/dependency-audit/versions/1.1.3/test.json +14 -7
- package/data/craftbook-templates/de/dependency-upgrade/versions/1.0.2/craftbook.json +621 -0
- package/data/craftbook-templates/de/dependency-upgrade/versions/1.0.2/test.json +235 -0
- package/data/craftbook-templates/de/design-review/versions/1.0.3/test.json +16 -8
- package/data/craftbook-templates/de/design-system-consultation/versions/2.0.8/craftbook.json +616 -0
- package/data/craftbook-templates/de/design-system-consultation/versions/2.0.8/test.json +201 -0
- package/data/craftbook-templates/do/doc-intake-pipeline/versions/1.0.4/craftbook.json +285 -0
- package/data/craftbook-templates/do/doc-intake-pipeline/versions/1.0.4/test.json +104 -0
- package/data/craftbook-templates/do/doc-rewrite/versions/1.0.3/test.json +119 -55
- package/data/craftbook-templates/do/dockerize-app/versions/1.0.4/craftbook.json +285 -0
- package/data/craftbook-templates/do/dockerize-app/versions/1.0.4/test.json +92 -0
- package/data/craftbook-templates/do/documentation-drift-review/versions/1.0.3/test.json +18 -9
- package/data/craftbook-templates/dr/draft-social-post/versions/1.0.2/craftbook.json +324 -0
- package/data/craftbook-templates/dr/draft-social-post/versions/1.0.2/test.json +161 -0
- package/data/craftbook-templates/ed/edit-notes/versions/1.0.3/test.json +11 -4
- package/data/craftbook-templates/en/engineering-retrospective/versions/2.0.8/craftbook.json +566 -0
- package/data/craftbook-templates/en/engineering-retrospective/versions/2.0.8/test.json +191 -0
- package/data/craftbook-templates/ep/episode-plan/versions/1.0.3/test.json +10 -5
- package/data/craftbook-templates/ex/executive-level-review/versions/2.0.7/test.json +7 -10
- package/data/craftbook-templates/ex/executive-level-review/versions/2.0.8/craftbook.json +595 -0
- package/data/craftbook-templates/ex/executive-level-review/versions/2.0.8/test.json +139 -0
- package/data/craftbook-templates/ex/expense-categorize/versions/1.0.4/craftbook.json +277 -0
- package/data/craftbook-templates/ex/expense-categorize/versions/1.0.4/test.json +104 -0
- package/data/craftbook-templates/fe/feature-flag-rollout/versions/1.1.2/craftbook.json +327 -0
- package/data/craftbook-templates/fe/feature-flag-rollout/versions/1.1.2/test.json +162 -0
- package/data/craftbook-templates/fl/flaky-test-fix/versions/1.0.2/craftbook.json +718 -0
- package/data/craftbook-templates/fl/flaky-test-fix/versions/1.0.2/test.json +239 -0
- package/data/craftbook-templates/fo/form-fill-batch/versions/1.0.4/craftbook.json +285 -0
- package/data/craftbook-templates/fo/form-fill-batch/versions/1.0.4/test.json +162 -0
- package/data/craftbook-templates/fo/form-wizard/versions/1.0.3/test.json +8 -8
- package/data/craftbook-templates/ho/holiday-card-run/versions/1.0.3/test.json +11 -4
- package/data/craftbook-templates/ho/hotfix-flow/versions/2.0.2/craftbook.json +610 -0
- package/data/craftbook-templates/ho/hotfix-flow/versions/2.0.2/test.json +220 -0
- package/data/craftbook-templates/ht/html-arcade-game/versions/1.2.3/craftbook.json +343 -0
- package/data/craftbook-templates/ht/html-arcade-game/versions/1.2.3/test.json +176 -0
- package/data/craftbook-templates/id/idea-office-hours/versions/2.0.8/craftbook.json +566 -0
- package/data/craftbook-templates/id/idea-office-hours/versions/2.0.8/test.json +141 -0
- package/data/craftbook-templates/im/image-set-index/versions/1.2.3/craftbook.json +298 -0
- package/data/craftbook-templates/im/image-set-index/versions/1.2.3/test.json +180 -0
- package/data/craftbook-templates/in/insurance-inventory/versions/1.0.3/test.json +11 -4
- package/data/craftbook-templates/in/interactive-quiz/versions/1.0.3/test.json +48 -23
- package/data/craftbook-templates/in/invitation-design/versions/1.0.3/test.json +1 -1
- package/data/craftbook-templates/index.json +1 -1
- package/data/craftbook-templates/it/item-intake/versions/1.0.3/test.json +11 -4
- package/data/craftbook-templates/le/lead-enrichment/versions/1.0.4/craftbook.json +277 -0
- package/data/craftbook-templates/le/lead-enrichment/versions/1.0.4/test.json +104 -0
- package/data/craftbook-templates/me/meeting-prep-brief/versions/1.0.3/test.json +16 -8
- package/data/craftbook-templates/me/memory-prompt-session/versions/1.0.3/test.json +3 -0
- package/data/craftbook-templates/mo/morning-report/versions/1.0.3/test.json +21 -9
- package/data/craftbook-templates/of/office-hours/versions/1.1.1/test.json +8 -8
- package/data/craftbook-templates/pe/perf-optimization/versions/2.0.2/craftbook.json +697 -0
- package/data/craftbook-templates/pe/perf-optimization/versions/2.0.2/test.json +235 -0
- package/data/craftbook-templates/pe/pest-diagnosis/versions/1.0.3/test.json +19 -8
- package/data/craftbook-templates/pl/plan/versions/1.0.2/test.json +8 -8
- package/data/craftbook-templates/pl/playtest-report/versions/1.0.3/test.json +17 -7
- package/data/craftbook-templates/pr/practice-exam/versions/1.0.3/test.json +7 -2
- package/data/craftbook-templates/pr/practice-session/versions/1.0.3/test.json +11 -4
- package/data/craftbook-templates/pu/pull-request-review/versions/1.9.2/craftbook.json +435 -0
- package/data/craftbook-templates/pu/pull-request-review/versions/1.9.2/test.json +141 -0
- package/data/craftbook-templates/qu/quarterly-summary/versions/1.0.3/test.json +16 -8
- package/data/craftbook-templates/re/receipt-intake/versions/1.0.3/test.json +11 -4
- package/data/craftbook-templates/re/receipt-ocr-ledger/versions/1.0.4/craftbook.json +277 -0
- package/data/craftbook-templates/re/receipt-ocr-ledger/versions/1.0.4/test.json +104 -0
- package/data/craftbook-templates/re/recipe-capture/versions/1.0.3/test.json +13 -5
- package/data/craftbook-templates/re/record-a-relative/versions/1.0.3/test.json +3 -0
- package/data/craftbook-templates/re/recurring-invoice-run/versions/1.0.4/craftbook.json +285 -0
- package/data/craftbook-templates/re/recurring-invoice-run/versions/1.0.4/test.json +104 -0
- package/data/craftbook-templates/re/refactor-module/versions/2.0.2/craftbook.json +695 -0
- package/data/craftbook-templates/re/refactor-module/versions/2.0.2/test.json +241 -0
- package/data/craftbook-templates/re/release-artifact-sanity-check/versions/1.0.4/craftbook.json +361 -0
- package/data/craftbook-templates/re/release-artifact-sanity-check/versions/1.0.4/test.json +138 -0
- package/data/craftbook-templates/re/release-notes/versions/1.0.4/test.json +8 -8
- package/data/craftbook-templates/re/release-notes/versions/1.0.5/craftbook.json +339 -0
- package/data/craftbook-templates/re/release-notes/versions/1.0.5/test.json +111 -0
- package/data/craftbook-templates/re/reliability-review/versions/1.0.3/test.json +16 -8
- package/data/craftbook-templates/re/research-report/versions/1.0.3/test.json +16 -8
- package/data/craftbook-templates/ro/root-cause-investigation/versions/2.0.8/craftbook.json +596 -0
- package/data/craftbook-templates/ro/root-cause-investigation/versions/2.0.8/test.json +157 -0
- package/data/craftbook-templates/ro/rough-cut-assembly/versions/1.0.4/test.json +13 -5
- package/data/craftbook-templates/sc/schema-migration/versions/2.0.2/craftbook.json +624 -0
- package/data/craftbook-templates/sc/schema-migration/versions/2.0.2/test.json +225 -0
- package/data/craftbook-templates/sc/script-automation/versions/1.0.4/craftbook.json +284 -0
- package/data/craftbook-templates/sc/script-automation/versions/1.0.4/test.json +162 -0
- package/data/craftbook-templates/se/security-architecture-review/versions/2.0.7/test.json +10 -13
- package/data/craftbook-templates/se/security-architecture-review/versions/2.0.8/craftbook.json +597 -0
- package/data/craftbook-templates/se/security-architecture-review/versions/2.0.8/test.json +157 -0
- package/data/craftbook-templates/so/social-digest/versions/1.0.3/craftbook.json +239 -0
- package/data/craftbook-templates/so/social-digest/versions/1.0.3/test.json +168 -0
- package/data/craftbook-templates/so/source-quality-audit/versions/1.0.3/test.json +25 -11
- package/data/craftbook-templates/sp/spec-authoring/versions/2.0.8/craftbook.json +599 -0
- package/data/craftbook-templates/sp/spec-authoring/versions/2.0.8/test.json +162 -0
- package/data/craftbook-templates/te/technical-documentation/versions/2.0.8/craftbook.json +577 -0
- package/data/craftbook-templates/te/technical-documentation/versions/2.0.8/test.json +174 -0
- package/data/craftbook-templates/te/test-coverage-review/versions/1.2.1/test.json +16 -8
- package/data/craftbook-templates/te/test-suite-backfill/versions/2.0.2/craftbook.json +592 -0
- package/data/craftbook-templates/te/test-suite-backfill/versions/2.0.2/test.json +195 -0
- package/data/craftbook-templates/th/thank-you-batch/versions/1.0.3/test.json +6 -3
- package/data/craftbook-templates/th/thank-you-sweep/versions/1.0.3/test.json +11 -4
- package/data/craftbook-templates/ti/tileset-batch/versions/1.0.3/test.json +9 -3
- package/data/craftbook-templates/tr/transcribe-and-shownotes/versions/1.0.3/test.json +10 -5
- package/data/craftbook-templates/tr/transcribe-index/versions/1.0.3/test.json +6 -3
- package/data/craftbook-templates/tr/translate-content/versions/1.1.3/craftbook.json +140 -0
- package/data/craftbook-templates/tr/translate-content/versions/1.1.3/test.json +141 -0
- package/data/craftbook-templates/ty/type-safety-pass/versions/2.0.2/craftbook.json +594 -0
- package/data/craftbook-templates/ty/type-safety-pass/versions/2.0.2/test.json +221 -0
- package/data/craftbook-templates/ux/ux-update/versions/1.0.2/craftbook.json +606 -0
- package/data/craftbook-templates/ux/ux-update/versions/1.0.2/test.json +206 -0
- package/data/craftbook-templates/we/weak-spot-drill/versions/1.0.3/test.json +11 -4
- package/data/craftbook-templates/we/weekly-walkthrough/versions/1.0.3/test.json +21 -9
- package/data/craftbook-templates/wh/whitepaper/versions/1.0.3/test.json +16 -8
- package/data/craftbook-templates/ye/year-in-review/versions/1.0.3/test.json +8 -4
- package/data/project-types/index.json +1 -1
- package/data/project-types/ju/just-chat/manifest.json +1 -1
- package/data/project-types/ju/just-chat/versions/1.0.1/about.md +5 -0
- package/data/project-types/ju/just-chat/versions/1.0.1/manifest.json +19 -0
- package/package.json +1 -1
- package/schemas/chat-model-identity.schema.json +27 -0
- package/schemas/chat-model-version.schema.json +22 -0
|
@@ -0,0 +1,239 @@
|
|
|
1
|
+
{
|
|
2
|
+
"schemaVersion": 1,
|
|
3
|
+
"title": "Deflake an order-dependent test with receipts",
|
|
4
|
+
"objective": "The craftbook drives a real deflake of a seeded order-dependence flake: statistical reproduction with recorded failing runs, a mechanism-level diagnosis, cleanup at the level every test shares (never skipping or weakening the test), and a stability proof - verified mechanically, including a five-run consecutive-green oracle, a late-probe order-robustness check, and a mutant restoring the original leaking order.",
|
|
5
|
+
"tags": [
|
|
6
|
+
"code",
|
|
7
|
+
"flaky-test",
|
|
8
|
+
"tests",
|
|
9
|
+
"tactical-fleet"
|
|
10
|
+
],
|
|
11
|
+
"prompt": "Use the Fix a Flaky Test craftbook: our suite is red in CI but the failing test passes when people run it by itself, so it keeps getting waved through. The report is in source/flake-report.md. Find the nondeterminism, remove it for real, and prove the suite stays green.",
|
|
12
|
+
"setup": {
|
|
13
|
+
"projectName": "Flagship Flags",
|
|
14
|
+
"about": "A tiny feature-flags module and its test suite. Tests run with `npm run test` (node --test).",
|
|
15
|
+
"files": [
|
|
16
|
+
{
|
|
17
|
+
"path": "package.json",
|
|
18
|
+
"content": "{\n \"name\": \"flagship\",\n \"private\": true,\n \"version\": \"1.0.0\",\n \"type\": \"module\",\n \"scripts\": {\n \"test\": \"node --test\"\n }\n}\n",
|
|
19
|
+
"modelInput": false
|
|
20
|
+
},
|
|
21
|
+
{
|
|
22
|
+
"path": "src/flags.js",
|
|
23
|
+
"content": "const flags = new Map();\n\nexport function enable(name) {\n flags.set(name, true);\n}\n\nexport function disable(name) {\n flags.delete(name);\n}\n\nexport function isEnabled(name) {\n return flags.get(name) === true;\n}\n\nexport function enabledCount() {\n return flags.size;\n}\n\nexport function resetFlags() {\n flags.clear();\n}\n"
|
|
24
|
+
},
|
|
25
|
+
{
|
|
26
|
+
"path": "tests/flags.test.mjs",
|
|
27
|
+
"content": "import assert from 'node:assert/strict';\nimport { test } from 'node:test';\nimport { enable, disable, isEnabled, enabledCount } from '../src/flags.js';\n\ntest('enable turns a flag on', () => {\n enable('beta-banner');\n assert.equal(isEnabled('beta-banner'), true);\n});\n\ntest('disable turns a flag off again', () => {\n enable('holiday-theme');\n disable('holiday-theme');\n assert.equal(isEnabled('holiday-theme'), false);\n});\n\ntest('unknown flags read as disabled', () => {\n assert.equal(isEnabled('dark-mode'), false);\n});\n\ntest('a fresh session starts with no flags enabled', () => {\n assert.equal(enabledCount(), 0);\n});\n"
|
|
28
|
+
},
|
|
29
|
+
{
|
|
30
|
+
"path": "source/flake-report.md",
|
|
31
|
+
"content": "# Flake report\n\nThe suite has been red in CI all week, and every developer who ran the failing test by itself saw it pass - so it kept getting waved through.\n\n npm run test\n not ok 4 - a fresh session starts with no flags enabled\n AssertionError: 1 !== 0\n\nRunning the full suite fails on that test consistently; running just that one test with `--test-name-pattern \"fresh session\"` is green every time. Nobody has touched `src/flags.js` in weeks.\n\nConstraints:\n\n- The pristine-session test guards a real invariant (feature sessions must start clean). Do not delete or weaken it.\n- The app depends on `src/flags.js` exactly as it is - fix the tests' hygiene, not the module's API.\n",
|
|
32
|
+
"modelInput": true
|
|
33
|
+
},
|
|
34
|
+
{
|
|
35
|
+
"path": "tests/verify-deflake.mjs",
|
|
36
|
+
"content": "import assert from 'node:assert/strict';\nimport { execFileSync } from 'node:child_process';\nimport { cpSync, mkdirSync, readFileSync, readdirSync, rmSync, writeFileSync } from 'node:fs';\nimport { join } from 'node:path';\nimport { enable, isEnabled, resetFlags, enabledCount } from '../src/flags.js';\n\nconst root = process.cwd();\nconst ORACLE_FILE = 'verify-deflake.mjs';\n\n// 1) The flags module still means what it meant.\nresetFlags();\nenable('oracle-probe');\nassert.equal(isEnabled('oracle-probe'), true, 'enable must still work');\nassert.equal(enabledCount(), 1, 'enabledCount must still count');\nresetFlags();\nassert.equal(enabledCount(), 0, 'resetFlags must still clear');\nconsole.log('DEFLAKE_ORACLE flags semantics ok');\n\nfunction runSuite(cwd) {\n try {\n execFileSync(process.execPath, ['--test'], { cwd, stdio: 'pipe', timeout: 30000 });\n return 0;\n } catch (err) {\n return typeof err.status === 'number' ? err.status : 1;\n }\n}\n\nfunction copyTree(dest) {\n rmSync(dest, { recursive: true, force: true });\n mkdirSync(dest, { recursive: true });\n cpSync(join(root, 'src'), join(dest, 'src'), { recursive: true });\n cpSync(join(root, 'tests'), join(dest, 'tests'), { recursive: true });\n rmSync(join(dest, 'tests', ORACLE_FILE), { force: true });\n cpSync(join(root, 'package.json'), join(dest, 'package.json'));\n return dest;\n}\n\n// 2) Five consecutive green runs of the real suite.\nfor (let run = 1; run <= 5; run += 1) {\n assert.equal(runSuite(root), 0, `full-suite run ${run} of 5 must be green`);\n}\nconsole.log('DEFLAKE_ORACLE stable 5x');\n\n// 3) The pristine-session assertion survives, unweakened and unskipped.\nconst testFiles = readdirSync(join(root, 'tests')).filter(\n (name) => name !== ORACLE_FILE && /\\.(mjs|cjs|js)$/.test(name),\n);\nconst combined = testFiles\n .map((name) => readFileSync(join(root, 'tests', name), 'utf8'))\n .join('\\n');\nassert.match(\n combined,\n /enabledCount\\s*\\(\\s*\\)\\s*(?:,\\s*0|===\\s*0)/,\n 'the pristine-session assertion (enabledCount() must be 0) has to survive',\n);\nassert.ok(\n !/\\.skip\\s*\\(|\\.todo\\s*\\(|\\.only\\s*\\(|\\{\\s*skip\\s*:|\\{\\s*todo\\s*:/.test(combined),\n 'no test may be skipped, stubbed as todo, or narrowed with only',\n);\nconsole.log('DEFLAKE_ORACLE assertion intact');\n\n// 4) Order robustness: tests appended at the END of the suite must inherit\n// the cleanup. The first probe enables a flag and never cleans up; the\n// second expects a pristine session anyway. Only hygiene that covers every\n// test (beforeEach/afterEach-style) passes both - a one-off reset inside\n// the seeded failing test, or a reorder, fails here. Aliased imports keep\n// the probe valid whatever the model did to the file's own imports.\nconst hostFile = testFiles.find((name) =>\n /enabledCount/.test(readFileSync(join(root, 'tests', name), 'utf8')),\n);\nassert.ok(hostFile, 'no test file references enabledCount');\nconst probeRoot = copyTree(join(root, '.oracle-probe'));\nconst probeTests =\n readFileSync(join(probeRoot, 'tests', hostFile), 'utf8') +\n \"\\n\\nimport { test as zzTest } from 'node:test';\\nimport zzAssert from 'node:assert/strict';\\nimport { enable as zzEnable, isEnabled as zzIsEnabled, enabledCount as zzEnabledCount } from '../src/flags.js';\\n\\nzzTest('zz probe enables a flag of its own', () => {\\n zzEnable('zz-probe-flag');\\n zzAssert.equal(zzIsEnabled('zz-probe-flag'), true);\\n});\\n\\nzzTest('zz probe still gets a pristine session', () => {\\n zzAssert.equal(zzEnabledCount(), 0);\\n});\\n\";\nwriteFileSync(join(probeRoot, 'tests', hostFile), probeTests);\nassert.equal(\n runSuite(probeRoot),\n 0,\n 'a test appended after the suite must still see a pristine session - cleanup has to cover every test, not only the seeded order',\n);\nconsole.log('DEFLAKE_ORACLE order robust');\n\n// 5) Mutant: restore the ORIGINAL leaking test file - the suite must fail\n// again, proving the seeded defect is still detectable and the green suite\n// comes from removed nondeterminism, not from a neutered module or dodge.\nconst ORIGINAL_TEST_FILE = `import assert from 'node:assert/strict';\nimport { test } from 'node:test';\nimport { enable, disable, isEnabled, enabledCount } from '../src/flags.js';\n\ntest('enable turns a flag on', () => {\n enable('beta-banner');\n assert.equal(isEnabled('beta-banner'), true);\n});\n\ntest('disable turns a flag off again', () => {\n enable('holiday-theme');\n disable('holiday-theme');\n assert.equal(isEnabled('holiday-theme'), false);\n});\n\ntest('unknown flags read as disabled', () => {\n assert.equal(isEnabled('dark-mode'), false);\n});\n\ntest('a fresh session starts with no flags enabled', () => {\n assert.equal(enabledCount(), 0);\n});\n`;\nconst mutantRoot = copyTree(join(root, '.oracle-mutant'));\nwriteFileSync(join(mutantRoot, 'tests', 'flags.test.mjs'), ORIGINAL_TEST_FILE);\nassert.notEqual(\n runSuite(mutantRoot),\n 0,\n 'restoring the original leaking order must make the suite fail again - the seeded defect stopped being detectable',\n);\nconsole.log('DEFLAKE_ORACLE mutant caught');\nconsole.log('DEFLAKE_ORACLE done');\n",
|
|
37
|
+
"surface": "harness"
|
|
38
|
+
}
|
|
39
|
+
],
|
|
40
|
+
"craftbookParams": {
|
|
41
|
+
"scope": "The test 'a fresh session starts with no flags enabled' in tests/flags.test.mjs fails whenever the whole suite runs and passes by itself; see source/flake-report.md. Make the suite deterministic without weakening that test."
|
|
42
|
+
}
|
|
43
|
+
},
|
|
44
|
+
"mocks": [],
|
|
45
|
+
"success": {
|
|
46
|
+
"summary": "The order-dependence mechanism is removed with hygiene that covers every test, the pristine-session assertion survives unweakened, five consecutive suite runs are green, a late-appended probe still sees a pristine session, and restoring the original test file makes the suite fail again.",
|
|
47
|
+
"deliverables": [
|
|
48
|
+
{
|
|
49
|
+
"path": "{{task.dir}}/flake-evidence.md",
|
|
50
|
+
"kind": "markdown-notes",
|
|
51
|
+
"artifact": true,
|
|
52
|
+
"minBytes": 500,
|
|
53
|
+
"checks": [
|
|
54
|
+
{
|
|
55
|
+
"kind": "contains",
|
|
56
|
+
"file": "{{task.dir}}/flake-evidence.md",
|
|
57
|
+
"pattern": "^##\\s+Symptom[\\s\\S]*^##\\s+Runs observed[\\s\\S]*^##\\s+Failure signature[\\s\\S]*^##\\s+Suspected class",
|
|
58
|
+
"flags": "im",
|
|
59
|
+
"label": "flake-evidence sections in order"
|
|
60
|
+
},
|
|
61
|
+
{
|
|
62
|
+
"kind": "citationsResolve",
|
|
63
|
+
"file": "{{task.dir}}/flake-evidence.md",
|
|
64
|
+
"minCitations": 2,
|
|
65
|
+
"artifact": true
|
|
66
|
+
}
|
|
67
|
+
]
|
|
68
|
+
},
|
|
69
|
+
{
|
|
70
|
+
"path": "{{task.dir}}/diagnosis.md",
|
|
71
|
+
"kind": "markdown-notes",
|
|
72
|
+
"artifact": true,
|
|
73
|
+
"minBytes": 600,
|
|
74
|
+
"checks": [
|
|
75
|
+
{
|
|
76
|
+
"kind": "contains",
|
|
77
|
+
"file": "{{task.dir}}/diagnosis.md",
|
|
78
|
+
"pattern": "^##\\s+Mechanism[\\s\\S]*^##\\s+Experiment[\\s\\S]*^##\\s+Defect site[\\s\\S]*^##\\s+Siblings checked",
|
|
79
|
+
"flags": "im",
|
|
80
|
+
"label": "diagnosis sections in order"
|
|
81
|
+
},
|
|
82
|
+
{
|
|
83
|
+
"kind": "contains",
|
|
84
|
+
"file": "{{task.dir}}/diagnosis.md",
|
|
85
|
+
"pattern": "tests/flags\\.test\\.mjs|src/flags\\.js",
|
|
86
|
+
"label": "names the real site of the shared-state hazard"
|
|
87
|
+
},
|
|
88
|
+
{
|
|
89
|
+
"kind": "citationsResolve",
|
|
90
|
+
"file": "{{task.dir}}/diagnosis.md",
|
|
91
|
+
"minCitations": 2,
|
|
92
|
+
"artifact": true
|
|
93
|
+
}
|
|
94
|
+
]
|
|
95
|
+
},
|
|
96
|
+
{
|
|
97
|
+
"path": "{{task.dir}}/fix-notes.md",
|
|
98
|
+
"kind": "markdown-notes",
|
|
99
|
+
"artifact": true,
|
|
100
|
+
"minBytes": 700,
|
|
101
|
+
"checks": [
|
|
102
|
+
{
|
|
103
|
+
"kind": "contains",
|
|
104
|
+
"file": "{{task.dir}}/fix-notes.md",
|
|
105
|
+
"pattern": "^##\\s+Problem[\\s\\S]*^##\\s+Change[\\s\\S]*^##\\s+Files touched[\\s\\S]*^##\\s+Regression coverage[\\s\\S]*^##\\s+Risk[\\s\\S]*^##\\s+How to verify",
|
|
106
|
+
"flags": "im",
|
|
107
|
+
"label": "fix-notes sections in order"
|
|
108
|
+
},
|
|
109
|
+
{
|
|
110
|
+
"kind": "citationsResolve",
|
|
111
|
+
"file": "{{task.dir}}/fix-notes.md",
|
|
112
|
+
"minCitations": 2,
|
|
113
|
+
"artifact": true
|
|
114
|
+
}
|
|
115
|
+
]
|
|
116
|
+
},
|
|
117
|
+
{
|
|
118
|
+
"path": "{{task.dir}}/validation.md",
|
|
119
|
+
"kind": "markdown-notes",
|
|
120
|
+
"artifact": true,
|
|
121
|
+
"minBytes": 300,
|
|
122
|
+
"checks": [
|
|
123
|
+
{
|
|
124
|
+
"kind": "contains",
|
|
125
|
+
"file": "{{task.dir}}/validation.md",
|
|
126
|
+
"pattern": "^##\\s+Runs[\\s\\S]*^##\\s+Result",
|
|
127
|
+
"flags": "im",
|
|
128
|
+
"label": "validation sections in order"
|
|
129
|
+
}
|
|
130
|
+
]
|
|
131
|
+
},
|
|
132
|
+
{
|
|
133
|
+
"path": "{{task.dir}}/review.md",
|
|
134
|
+
"kind": "markdown-report",
|
|
135
|
+
"artifact": true,
|
|
136
|
+
"minBytes": 400,
|
|
137
|
+
"checks": [
|
|
138
|
+
{
|
|
139
|
+
"kind": "contains",
|
|
140
|
+
"file": "{{task.dir}}/review.md",
|
|
141
|
+
"pattern": "Verdict:\\s*(?:PASS|REVISE)",
|
|
142
|
+
"flags": "i",
|
|
143
|
+
"label": "explicit reviewer verdict"
|
|
144
|
+
}
|
|
145
|
+
]
|
|
146
|
+
}
|
|
147
|
+
],
|
|
148
|
+
"checks": [
|
|
149
|
+
{
|
|
150
|
+
"kind": "notContains",
|
|
151
|
+
"file": "tests/flags.test.mjs",
|
|
152
|
+
"pattern": "\\.skip\\s*\\(|\\.todo\\s*\\(|\\.only\\s*\\(|\\{\\s*skip\\s*:|\\{\\s*todo\\s*:",
|
|
153
|
+
"label": "the flaky test is not skipped, stubbed, or narrowed"
|
|
154
|
+
},
|
|
155
|
+
{
|
|
156
|
+
"kind": "sourceParses",
|
|
157
|
+
"file": "tests/flags.test.mjs"
|
|
158
|
+
},
|
|
159
|
+
{
|
|
160
|
+
"kind": "nodeRuns",
|
|
161
|
+
"file": "tests/flags.test.mjs",
|
|
162
|
+
"timeoutMs": 30000
|
|
163
|
+
},
|
|
164
|
+
{
|
|
165
|
+
"kind": "nodeScriptPasses",
|
|
166
|
+
"script": "tests/verify-deflake.mjs",
|
|
167
|
+
"timeoutMs": 60000,
|
|
168
|
+
"requiredOutput": [
|
|
169
|
+
{
|
|
170
|
+
"pattern": "DEFLAKE_ORACLE flags semantics ok",
|
|
171
|
+
"label": "the flags module still behaves"
|
|
172
|
+
},
|
|
173
|
+
{
|
|
174
|
+
"pattern": "DEFLAKE_ORACLE stable 5x",
|
|
175
|
+
"label": "five consecutive green suite runs"
|
|
176
|
+
},
|
|
177
|
+
{
|
|
178
|
+
"pattern": "DEFLAKE_ORACLE assertion intact",
|
|
179
|
+
"label": "the pristine-session assertion survives unweakened"
|
|
180
|
+
},
|
|
181
|
+
{
|
|
182
|
+
"pattern": "DEFLAKE_ORACLE order robust",
|
|
183
|
+
"label": "a late-appended test still sees a pristine session"
|
|
184
|
+
},
|
|
185
|
+
{
|
|
186
|
+
"pattern": "DEFLAKE_ORACLE mutant caught",
|
|
187
|
+
"label": "restoring the original order makes the suite fail again"
|
|
188
|
+
}
|
|
189
|
+
]
|
|
190
|
+
}
|
|
191
|
+
],
|
|
192
|
+
"taskNotes": {
|
|
193
|
+
"minBytes": 120,
|
|
194
|
+
"checks": [
|
|
195
|
+
{
|
|
196
|
+
"kind": "contains",
|
|
197
|
+
"file": "task-notes.md",
|
|
198
|
+
"pattern": "\\bDONE\\b[\\s\\S]*(npm\\s+run\\s+test|node\\s+--test|consecutive)",
|
|
199
|
+
"flags": "i",
|
|
200
|
+
"label": "DONE note cites the stability proof"
|
|
201
|
+
}
|
|
202
|
+
]
|
|
203
|
+
},
|
|
204
|
+
"taskGraph": {
|
|
205
|
+
"requireCraftbookTask": true,
|
|
206
|
+
"requireTerminalStep": true
|
|
207
|
+
},
|
|
208
|
+
"unchangedFixtures": [
|
|
209
|
+
"package.json",
|
|
210
|
+
"src/flags.js",
|
|
211
|
+
"source/flake-report.md"
|
|
212
|
+
]
|
|
213
|
+
},
|
|
214
|
+
"rubric": {
|
|
215
|
+
"artifact": {
|
|
216
|
+
"path": "{{task.dir}}/fix-notes.md",
|
|
217
|
+
"kind": "markdown"
|
|
218
|
+
},
|
|
219
|
+
"axes": [
|
|
220
|
+
{
|
|
221
|
+
"name": "Mechanism-level diagnosis",
|
|
222
|
+
"description": "The nondeterminism is named as a concrete mechanism (shared module state leaking across tests) and pinned with a real experiment, not guessed from the symptom."
|
|
223
|
+
},
|
|
224
|
+
{
|
|
225
|
+
"name": "No masking",
|
|
226
|
+
"description": "The flake is removed with cleanup or isolation that covers every test; the flaky test keeps its exact assertion and gains no skip, retry, or widened timeout."
|
|
227
|
+
},
|
|
228
|
+
{
|
|
229
|
+
"name": "Receipt-backed stability",
|
|
230
|
+
"description": "Stability is proven with consecutive green runs recorded as real receipts, and anything unverified is stated plainly instead of claimed."
|
|
231
|
+
}
|
|
232
|
+
]
|
|
233
|
+
},
|
|
234
|
+
"qualityFocus": [
|
|
235
|
+
"statistical reproduction",
|
|
236
|
+
"mechanism removal over masking",
|
|
237
|
+
"consecutive-green receipts"
|
|
238
|
+
]
|
|
239
|
+
}
|
|
@@ -0,0 +1,285 @@
|
|
|
1
|
+
{
|
|
2
|
+
"id": "form-fill-batch",
|
|
3
|
+
"name": "Batch-Fill Forms from a Dataset",
|
|
4
|
+
"description": "Automate filling a web or PDF form once per row of an input dataset — registrations, applications, data-entry portals, mail-merge into a form. Scopes the field-to-column mapping and a per-row success signal FIRST, then builds a loop that loads each record, maps it onto the form fields, submits, and records the outcome, then verifies every row was attempted, each submission was confirmed, and failures are logged with the reason. Mapping-before-build is the wisdom: lock the column→field map and the 'submitted OK' signal so the run is idempotent, resumable, and auditable rather than a blind click-through.\n\nA gallery craftbook generated from an archetype spec. It runs\n`phase → (per-phase gate) → … → evaluate → (loop) → finish`. Each build\nphase that produces a checkable artifact is followed by a **runtime\ngate-checkpoint** — the runtime verifies the artifact and routes with no\nmodel turn, looping back to redo the phase on a miss. The final `evaluate`\nstep holds a static deliverable gate plus a reviewer QA pass. What it adds\nover the generic `build-loop`: a specialist role per phase, a\ndomain-correct ordering, and a concrete per-phase quality bar.\n\nDeliverables marked \"artifact\" land in the project's artifacts drawer (`write_artifact` / `read_artifact`), not the shipped workspace — review output is not product source.\n\nPhases:\n\n1. Scope the fill job (planner) — map dataset columns to form fields + success signal → gated on artifact `{{workPath}}/scope.md` (markdown-notes)\n2. Build the fill automation (developer) — loop records, fill, submit, log each outcome → gated on `form_fill.py` (code-module)\n3. Verify the run (reviewer) — coverage, confirmations, and failure logging → gated on artifact `{{workPath}}/verify.md` (markdown-notes)\n\nThe gates never advance with an unmet criterion, and loop back to the\nowning phase to fix named gaps.\n",
|
|
5
|
+
"entryStepId": "scope",
|
|
6
|
+
"triggers": [
|
|
7
|
+
"fill out forms automatically",
|
|
8
|
+
"batch form submission",
|
|
9
|
+
"mail merge into a form",
|
|
10
|
+
"auto-fill a portal",
|
|
11
|
+
"submit a form for each row"
|
|
12
|
+
],
|
|
13
|
+
"paramSchema": {
|
|
14
|
+
"type": "object",
|
|
15
|
+
"properties": {
|
|
16
|
+
"workPath": {
|
|
17
|
+
"type": "string",
|
|
18
|
+
"title": "Working folder",
|
|
19
|
+
"description": "Per-task working folder in the artifacts drawer. Defaults to this task's own folder so runs never collide; override with a stable name when you deliberately want runs to share files.",
|
|
20
|
+
"default": "{{task.dir}}"
|
|
21
|
+
}
|
|
22
|
+
}
|
|
23
|
+
},
|
|
24
|
+
"steps": [
|
|
25
|
+
{
|
|
26
|
+
"id": "scope",
|
|
27
|
+
"name": "Scope the fill job",
|
|
28
|
+
"description": "map dataset columns to form fields + success signal",
|
|
29
|
+
"prompt": "Plan the batch fill before automating anything. Step 1: identify the input dataset (path + columns) and the target form (URL or PDF) and list every fillable field with its type (text, select, checkbox, date, file). Step 2: build an explicit mapping from each dataset column to a form field, and note required form fields that have no column (these need a default or must abort the row). Step 3: define the per-row success signal — what proves a submission landed (a confirmation URL, a success banner text, a generated ID, or a saved PDF). Step 4: decide the results log shape (one row per input record with status submitted/failed/skipped and a reason/confirmation). Step 5: write an acceptance-criteria checklist ('every input row is attempted', 'each success is confirmed by the signal', 'failures logged with a reason', 'no row submitted twice', 'run is resumable from the log'). Call write_task_note with the mapping + signal + checklist and write it to {{workPath}}/scope.md. No automation yet.\n\nThe deliverable `{{workPath}}/scope.md` lands in the project's artifacts drawer — write it with `write_artifact` and read it back with `read_artifact`; the shipped workspace stays untouched.",
|
|
30
|
+
"suggestedRole": "planner",
|
|
31
|
+
"toolPolicy": {
|
|
32
|
+
"disallowBuiltinToolsets": [
|
|
33
|
+
"ai-apps",
|
|
34
|
+
"archives",
|
|
35
|
+
"audio",
|
|
36
|
+
"browser-automation",
|
|
37
|
+
"code-execution",
|
|
38
|
+
"craftbooks",
|
|
39
|
+
"data-tables",
|
|
40
|
+
"entity-intel",
|
|
41
|
+
"git",
|
|
42
|
+
"image-intel",
|
|
43
|
+
"images",
|
|
44
|
+
"role-delegation",
|
|
45
|
+
"role-delegation-escalation",
|
|
46
|
+
"security-intel",
|
|
47
|
+
"team-management",
|
|
48
|
+
"videos",
|
|
49
|
+
"web",
|
|
50
|
+
"workspace-fs-write"
|
|
51
|
+
],
|
|
52
|
+
"outputMedium": "artifact",
|
|
53
|
+
"additionalOutputMedia": [
|
|
54
|
+
"task-note"
|
|
55
|
+
]
|
|
56
|
+
},
|
|
57
|
+
"advanceWhen": {
|
|
58
|
+
"file": "{{workPath}}/scope.md",
|
|
59
|
+
"minBytes": 1,
|
|
60
|
+
"sniff": "nonempty",
|
|
61
|
+
"artifact": true
|
|
62
|
+
},
|
|
63
|
+
"gate": {
|
|
64
|
+
"at": "completion",
|
|
65
|
+
"checks": [
|
|
66
|
+
{
|
|
67
|
+
"kind": "minBytes",
|
|
68
|
+
"file": "{{workPath}}/scope.md",
|
|
69
|
+
"bytes": 120,
|
|
70
|
+
"artifact": true
|
|
71
|
+
},
|
|
72
|
+
{
|
|
73
|
+
"kind": "sniff",
|
|
74
|
+
"file": "{{workPath}}/scope.md",
|
|
75
|
+
"sniff": "nonempty",
|
|
76
|
+
"artifact": true
|
|
77
|
+
}
|
|
78
|
+
],
|
|
79
|
+
"onReject": "scope",
|
|
80
|
+
"maxAttempts": 3
|
|
81
|
+
},
|
|
82
|
+
"next": "build"
|
|
83
|
+
},
|
|
84
|
+
{
|
|
85
|
+
"id": "build",
|
|
86
|
+
"name": "Build the fill automation",
|
|
87
|
+
"description": "loop records, fill, submit, log each outcome",
|
|
88
|
+
"prompt": "Implement the batch-fill driver to the locked mapping. Step 1: load the dataset and the results log; skip any row already marked submitted so the run is resumable. Step 2: for each remaining row, populate every mapped form field by type (set text, choose selects, toggle checkboxes, attach files), applying defaults for required-but-unmapped fields. Step 3: submit, then assert the success signal from scope before recording success — never assume submission worked. Step 4: append a results-log row with status and the confirmation token or the failure reason; on error, catch it, log failed, and continue to the next row. Step 5: write the automation to a runnable code module and produce/update the results log. Call write_task_note with the module path, the log path, and counts (submitted/failed/skipped).",
|
|
89
|
+
"suggestedRole": "developer",
|
|
90
|
+
"toolPolicy": {
|
|
91
|
+
"disallowBuiltinToolsets": [
|
|
92
|
+
"ai-apps",
|
|
93
|
+
"archives",
|
|
94
|
+
"artifacts",
|
|
95
|
+
"audio",
|
|
96
|
+
"browser-automation",
|
|
97
|
+
"craftbooks",
|
|
98
|
+
"data-tables",
|
|
99
|
+
"entity-intel",
|
|
100
|
+
"git",
|
|
101
|
+
"image-intel",
|
|
102
|
+
"images",
|
|
103
|
+
"role-delegation",
|
|
104
|
+
"role-delegation-escalation",
|
|
105
|
+
"security-intel",
|
|
106
|
+
"team-management",
|
|
107
|
+
"videos",
|
|
108
|
+
"web"
|
|
109
|
+
],
|
|
110
|
+
"outputMedium": "workspace",
|
|
111
|
+
"additionalOutputMedia": [
|
|
112
|
+
"task-note"
|
|
113
|
+
]
|
|
114
|
+
},
|
|
115
|
+
"advanceWhen": {
|
|
116
|
+
"file": "form_fill.py",
|
|
117
|
+
"minBytes": 1,
|
|
118
|
+
"sniff": "nonempty"
|
|
119
|
+
},
|
|
120
|
+
"gate": {
|
|
121
|
+
"at": "completion",
|
|
122
|
+
"checks": [
|
|
123
|
+
{
|
|
124
|
+
"kind": "minBytes",
|
|
125
|
+
"file": "form_fill.py",
|
|
126
|
+
"bytes": 200
|
|
127
|
+
},
|
|
128
|
+
{
|
|
129
|
+
"kind": "sniff",
|
|
130
|
+
"file": "form_fill.py",
|
|
131
|
+
"sniff": "nonempty"
|
|
132
|
+
},
|
|
133
|
+
{
|
|
134
|
+
"kind": "esmImports",
|
|
135
|
+
"file": "form_fill.py"
|
|
136
|
+
},
|
|
137
|
+
{
|
|
138
|
+
"kind": "sourceParses",
|
|
139
|
+
"file": "form_fill.py"
|
|
140
|
+
}
|
|
141
|
+
],
|
|
142
|
+
"onReject": "build",
|
|
143
|
+
"maxAttempts": 3
|
|
144
|
+
},
|
|
145
|
+
"next": "verify"
|
|
146
|
+
},
|
|
147
|
+
{
|
|
148
|
+
"id": "verify",
|
|
149
|
+
"name": "Verify the run",
|
|
150
|
+
"description": "coverage, confirmations, and failure logging",
|
|
151
|
+
"prompt": "Verify the batch run did what scope promised. Step 1: confirm the automation runs without throwing and produces a results log. Step 2: check coverage — the log has exactly one entry per input row, none missing. Step 3: confirm every 'submitted' entry carries the success signal (confirmation token / banner / saved file) and is not just assumed. Step 4: confirm every 'failed' entry has a concrete reason and that failures did not abort the rest of the batch. Step 5: confirm idempotency — a re-run skips already-submitted rows and does not double-submit. Write PASS/FAIL per criterion to {{workPath}}/verify.md and task notes; on failure, name the rows and loop back to build.\n\nThe deliverable `{{workPath}}/verify.md` lands in the project's artifacts drawer — write it with `write_artifact` and read it back with `read_artifact`; the shipped workspace stays untouched.",
|
|
152
|
+
"suggestedRole": "reviewer",
|
|
153
|
+
"toolPolicy": {
|
|
154
|
+
"disallowBuiltinToolsets": [
|
|
155
|
+
"ai-apps",
|
|
156
|
+
"archives",
|
|
157
|
+
"audio",
|
|
158
|
+
"browser-automation",
|
|
159
|
+
"code-execution",
|
|
160
|
+
"craftbooks",
|
|
161
|
+
"data-tables",
|
|
162
|
+
"entity-intel",
|
|
163
|
+
"git",
|
|
164
|
+
"image-intel",
|
|
165
|
+
"images",
|
|
166
|
+
"role-delegation",
|
|
167
|
+
"role-delegation-escalation",
|
|
168
|
+
"security-intel",
|
|
169
|
+
"team-management",
|
|
170
|
+
"videos",
|
|
171
|
+
"web",
|
|
172
|
+
"workspace-fs-write"
|
|
173
|
+
],
|
|
174
|
+
"outputMedium": "artifact",
|
|
175
|
+
"additionalOutputMedia": [
|
|
176
|
+
"task-note"
|
|
177
|
+
]
|
|
178
|
+
},
|
|
179
|
+
"advanceWhen": {
|
|
180
|
+
"file": "{{workPath}}/verify.md",
|
|
181
|
+
"minBytes": 1,
|
|
182
|
+
"sniff": "nonempty",
|
|
183
|
+
"artifact": true
|
|
184
|
+
},
|
|
185
|
+
"gate": {
|
|
186
|
+
"at": "completion",
|
|
187
|
+
"checks": [
|
|
188
|
+
{
|
|
189
|
+
"kind": "minBytes",
|
|
190
|
+
"file": "form_fill.py",
|
|
191
|
+
"bytes": 200
|
|
192
|
+
},
|
|
193
|
+
{
|
|
194
|
+
"kind": "sniff",
|
|
195
|
+
"file": "form_fill.py",
|
|
196
|
+
"sniff": "nonempty"
|
|
197
|
+
},
|
|
198
|
+
{
|
|
199
|
+
"kind": "esmImports",
|
|
200
|
+
"file": "form_fill.py"
|
|
201
|
+
},
|
|
202
|
+
{
|
|
203
|
+
"kind": "sourceParses",
|
|
204
|
+
"file": "form_fill.py"
|
|
205
|
+
}
|
|
206
|
+
],
|
|
207
|
+
"onReject": "verify",
|
|
208
|
+
"maxAttempts": 4
|
|
209
|
+
},
|
|
210
|
+
"next": "evaluate"
|
|
211
|
+
},
|
|
212
|
+
{
|
|
213
|
+
"id": "evaluate",
|
|
214
|
+
"name": "Evaluate",
|
|
215
|
+
"description": "Grade the deliverable against every acceptance criterion. All pass → finish; any fail → loop back and fix the gap.",
|
|
216
|
+
"prompt": "Exercise form_fill.py against the dataset (or trace it if execution is unavailable) and verify every criterion from {{workPath}}/scope.md. Confirm: one log entry per input row, each success carries the defined success signal, each failure has a reason and did not abort the batch, no row is submitted twice, and a re-run resumes from the log. Write PASS/FAIL per criterion; on any failure, name the rows and loop back to build.\n\nThen route — this is the whole point of the loop:\n\n- **Every criterion PASSES →** call `advance_task_step({ ref, stepId: \"evaluate\", next: \"finish\" })`.\n- **Any criterion FAILS →** write the specific gaps to notes, then call `advance_task_step({ ref, stepId: \"evaluate\", next: \"build\" })` to loop back. The builder fixes exactly those gaps.\n\nNever route to `finish` while any criterion is unmet. The build phase's completion gate already blocked a grossly-incomplete deliverable; your job is the judgment an automated check cannot make (does it actually work, read well, look right). After ~3 unproductive loops, stop and report DONE_WITH_CONCERNS so the user can step in.",
|
|
217
|
+
"suggestedRole": "reviewer",
|
|
218
|
+
"toolPolicy": {
|
|
219
|
+
"disallowBuiltinToolsets": [
|
|
220
|
+
"ai-apps",
|
|
221
|
+
"archives",
|
|
222
|
+
"artifacts",
|
|
223
|
+
"audio",
|
|
224
|
+
"browser-automation",
|
|
225
|
+
"code-execution",
|
|
226
|
+
"craftbooks",
|
|
227
|
+
"data-tables",
|
|
228
|
+
"entity-intel",
|
|
229
|
+
"git",
|
|
230
|
+
"image-intel",
|
|
231
|
+
"images",
|
|
232
|
+
"role-delegation",
|
|
233
|
+
"role-delegation-escalation",
|
|
234
|
+
"security-intel",
|
|
235
|
+
"team-management",
|
|
236
|
+
"videos",
|
|
237
|
+
"web",
|
|
238
|
+
"workspace-fs-write"
|
|
239
|
+
],
|
|
240
|
+
"outputMedium": "task-note"
|
|
241
|
+
},
|
|
242
|
+
"consumes": [
|
|
243
|
+
{
|
|
244
|
+
"file": "form_fill.py"
|
|
245
|
+
}
|
|
246
|
+
],
|
|
247
|
+
"next": "build"
|
|
248
|
+
},
|
|
249
|
+
{
|
|
250
|
+
"id": "finish",
|
|
251
|
+
"name": "Finish",
|
|
252
|
+
"description": "All acceptance criteria met. Stamp a short summary and report DONE.",
|
|
253
|
+
"prompt": "Every acceptance criterion passed. Write a one-paragraph DONE summary to task notes via `write_task_note`: what was built, the deliverable path(s), and a one-line confirmation that each criterion is met. Then report DONE.",
|
|
254
|
+
"suggestedRole": "developer",
|
|
255
|
+
"toolPolicy": {
|
|
256
|
+
"disallowBuiltinToolsets": [
|
|
257
|
+
"ai-apps",
|
|
258
|
+
"archives",
|
|
259
|
+
"artifacts",
|
|
260
|
+
"audio",
|
|
261
|
+
"browser-automation",
|
|
262
|
+
"code-execution",
|
|
263
|
+
"craftbooks",
|
|
264
|
+
"data-tables",
|
|
265
|
+
"entity-intel",
|
|
266
|
+
"git",
|
|
267
|
+
"image-intel",
|
|
268
|
+
"images",
|
|
269
|
+
"role-delegation",
|
|
270
|
+
"role-delegation-escalation",
|
|
271
|
+
"security-intel",
|
|
272
|
+
"team-management",
|
|
273
|
+
"videos",
|
|
274
|
+
"web",
|
|
275
|
+
"workspace-fs-write"
|
|
276
|
+
],
|
|
277
|
+
"outputMedium": "task-note"
|
|
278
|
+
},
|
|
279
|
+
"terminal": true
|
|
280
|
+
}
|
|
281
|
+
],
|
|
282
|
+
"version": "1.0.4",
|
|
283
|
+
"releasedAt": "2026-09-05T03:30:00Z",
|
|
284
|
+
"minGezelVersion": "1.26233"
|
|
285
|
+
}
|