@bendyline/gilde 0.1.55 → 0.1.56
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/authoring/chat-models/glm-5.3-flash-320b-q2.json +129 -0
- package/authoring/gstack/wave.json +2 -2
- package/data/chat-models/gl/glm-5.3-flash-320b-q2/manifest.json +137 -0
- package/data/chat-models/gl/glm-5.3-flash-320b-q2/versions/1.0.0/manifest.json +23 -0
- package/data/chat-models/index.json +1 -1
- package/data/craftbook-templates/a1/a11y-audit/versions/1.1.3/test.json +16 -8
- package/data/craftbook-templates/ac/accessibility-retrofit/versions/2.0.2/craftbook.json +618 -0
- package/data/craftbook-templates/ac/accessibility-retrofit/versions/2.0.2/test.json +228 -0
- package/data/craftbook-templates/al/album-curate/versions/1.0.3/test.json +6 -3
- package/data/craftbook-templates/al/alert-rules/versions/1.0.3/test.json +8 -8
- package/data/craftbook-templates/an/anniversary-cut/versions/1.0.5/test.json +10 -5
- package/data/craftbook-templates/an/annual-document-review/versions/1.0.3/test.json +16 -8
- package/data/craftbook-templates/an/anomaly-scan/versions/1.0.3/test.json +12 -6
- package/data/craftbook-templates/ap/apply-review-findings/versions/1.0.2/craftbook.json +602 -0
- package/data/craftbook-templates/ap/apply-review-findings/versions/1.0.2/test.json +290 -0
- package/data/craftbook-templates/ar/artifact-integrity-review/versions/1.0.3/test.json +18 -9
- package/data/craftbook-templates/au/audiobook-master-pack/versions/1.0.5/test.json +10 -5
- package/data/craftbook-templates/au/auth-flow/versions/1.0.3/test.json +12 -6
- package/data/craftbook-templates/au/auth-flow/versions/1.0.4/craftbook.json +282 -0
- package/data/craftbook-templates/au/auth-flow/versions/1.0.4/test.json +143 -0
- package/data/craftbook-templates/au/automation-recipe/versions/1.0.3/test.json +12 -6
- package/data/craftbook-templates/ba/backup-routine/versions/1.0.4/craftbook.json +282 -0
- package/data/craftbook-templates/ba/backup-routine/versions/1.0.4/test.json +108 -0
- package/data/craftbook-templates/bl/blog-post/versions/1.0.3/test.json +8 -8
- package/data/craftbook-templates/br/branding-website/versions/1.1.3/craftbook.json +335 -0
- package/data/craftbook-templates/br/branding-website/versions/1.1.3/test.json +164 -0
- package/data/craftbook-templates/br/broadcast-announcement/versions/1.0.3/test.json +8 -8
- package/data/craftbook-templates/br/browser-qa-audit/versions/2.0.8/craftbook.json +621 -0
- package/data/craftbook-templates/br/browser-qa-audit/versions/2.0.8/test.json +376 -0
- package/data/craftbook-templates/bu/bug-fix-tdd/versions/2.0.2/craftbook.json +730 -0
- package/data/craftbook-templates/bu/bug-fix-tdd/versions/2.0.2/test.json +221 -0
- package/data/craftbook-templates/bu/build-loop/versions/1.1.1/test.json +8 -8
- package/data/craftbook-templates/ca/case-study/versions/1.0.3/test.json +8 -8
- package/data/craftbook-templates/ch/chapter-narration-run/versions/1.0.4/test.json +10 -5
- package/data/craftbook-templates/ch/character-turnaround/versions/1.0.3/test.json +25 -11
- package/data/craftbook-templates/ci/ci-pipeline/versions/2.0.2/craftbook.json +587 -0
- package/data/craftbook-templates/ci/ci-pipeline/versions/2.0.2/test.json +225 -0
- package/data/craftbook-templates/ci/citation-audit/versions/1.0.3/test.json +28 -14
- package/data/craftbook-templates/cl/cli-tool/versions/1.1.3/test.json +8 -8
- package/data/craftbook-templates/co/codebase-refactoring-review/versions/1.0.3/test.json +16 -8
- package/data/craftbook-templates/co/codebase-ux-review/versions/1.0.3/test.json +18 -9
- package/data/craftbook-templates/co/codemod-sweep/versions/1.0.2/craftbook.json +605 -0
- package/data/craftbook-templates/co/codemod-sweep/versions/1.0.2/test.json +252 -0
- package/data/craftbook-templates/co/cohort-analysis/versions/1.0.3/test.json +18 -9
- package/data/craftbook-templates/co/competitive-analysis/versions/1.0.3/test.json +23 -15
- package/data/craftbook-templates/co/content-accuracy-review/versions/1.0.3/test.json +28 -14
- package/data/craftbook-templates/co/content-deck/versions/1.2.3/craftbook.json +4 -4
- package/data/craftbook-templates/co/content-deck/versions/1.2.3/test.json +3 -0
- package/data/craftbook-templates/co/content-deck/versions/1.2.4/craftbook.json +341 -0
- package/data/craftbook-templates/co/content-deck/versions/1.2.4/test.json +169 -0
- package/data/craftbook-templates/co/copy-review/versions/1.0.3/test.json +14 -7
- package/data/craftbook-templates/co/corpus-email-digest/versions/1.2.3/craftbook.json +4 -4
- package/data/craftbook-templates/co/corpus-email-digest/versions/1.2.3/test.json +3 -0
- package/data/craftbook-templates/co/corpus-email-digest/versions/1.2.4/craftbook.json +345 -0
- package/data/craftbook-templates/co/corpus-email-digest/versions/1.2.4/test.json +187 -0
- package/data/craftbook-templates/co/corpus-synthesis/versions/1.0.3/test.json +26 -13
- package/data/craftbook-templates/co/cover-letter/versions/1.0.3/test.json +8 -8
- package/data/craftbook-templates/cs/csv-transformer/versions/1.2.4/test.json +7 -7
- package/data/craftbook-templates/da/dashboard-spec/versions/1.0.3/test.json +7 -7
- package/data/craftbook-templates/da/data-export-migrate/versions/1.0.4/craftbook.json +285 -0
- package/data/craftbook-templates/da/data-export-migrate/versions/1.0.4/test.json +104 -0
- package/data/craftbook-templates/da/data-pipeline-etl/versions/1.0.5/craftbook.json +284 -0
- package/data/craftbook-templates/da/data-pipeline-etl/versions/1.0.5/test.json +162 -0
- package/data/craftbook-templates/da/data-quality-audit/versions/1.0.3/test.json +38 -19
- package/data/craftbook-templates/da/data-to-report/versions/1.0.3/test.json +12 -6
- package/data/craftbook-templates/db/db-index-tuning/versions/1.0.4/craftbook.json +286 -0
- package/data/craftbook-templates/db/db-index-tuning/versions/1.0.4/test.json +104 -0
- package/data/craftbook-templates/de/dependency-audit/versions/1.1.3/test.json +14 -7
- package/data/craftbook-templates/de/dependency-upgrade/versions/1.0.2/craftbook.json +621 -0
- package/data/craftbook-templates/de/dependency-upgrade/versions/1.0.2/test.json +235 -0
- package/data/craftbook-templates/de/design-review/versions/1.0.3/test.json +16 -8
- package/data/craftbook-templates/de/design-system-consultation/versions/2.0.8/craftbook.json +616 -0
- package/data/craftbook-templates/de/design-system-consultation/versions/2.0.8/test.json +201 -0
- package/data/craftbook-templates/do/doc-intake-pipeline/versions/1.0.4/craftbook.json +285 -0
- package/data/craftbook-templates/do/doc-intake-pipeline/versions/1.0.4/test.json +104 -0
- package/data/craftbook-templates/do/doc-rewrite/versions/1.0.3/test.json +119 -55
- package/data/craftbook-templates/do/dockerize-app/versions/1.0.4/craftbook.json +285 -0
- package/data/craftbook-templates/do/dockerize-app/versions/1.0.4/test.json +92 -0
- package/data/craftbook-templates/do/documentation-drift-review/versions/1.0.3/test.json +18 -9
- package/data/craftbook-templates/dr/draft-social-post/versions/1.0.2/craftbook.json +324 -0
- package/data/craftbook-templates/dr/draft-social-post/versions/1.0.2/test.json +161 -0
- package/data/craftbook-templates/ed/edit-notes/versions/1.0.3/test.json +11 -4
- package/data/craftbook-templates/en/engineering-retrospective/versions/2.0.8/craftbook.json +566 -0
- package/data/craftbook-templates/en/engineering-retrospective/versions/2.0.8/test.json +191 -0
- package/data/craftbook-templates/ep/episode-plan/versions/1.0.3/test.json +10 -5
- package/data/craftbook-templates/ex/executive-level-review/versions/2.0.7/test.json +7 -10
- package/data/craftbook-templates/ex/executive-level-review/versions/2.0.8/craftbook.json +595 -0
- package/data/craftbook-templates/ex/executive-level-review/versions/2.0.8/test.json +139 -0
- package/data/craftbook-templates/ex/expense-categorize/versions/1.0.4/craftbook.json +277 -0
- package/data/craftbook-templates/ex/expense-categorize/versions/1.0.4/test.json +104 -0
- package/data/craftbook-templates/fe/feature-flag-rollout/versions/1.1.2/craftbook.json +327 -0
- package/data/craftbook-templates/fe/feature-flag-rollout/versions/1.1.2/test.json +162 -0
- package/data/craftbook-templates/fl/flaky-test-fix/versions/1.0.2/craftbook.json +718 -0
- package/data/craftbook-templates/fl/flaky-test-fix/versions/1.0.2/test.json +239 -0
- package/data/craftbook-templates/fo/form-fill-batch/versions/1.0.4/craftbook.json +285 -0
- package/data/craftbook-templates/fo/form-fill-batch/versions/1.0.4/test.json +162 -0
- package/data/craftbook-templates/fo/form-wizard/versions/1.0.3/test.json +8 -8
- package/data/craftbook-templates/ho/holiday-card-run/versions/1.0.3/test.json +11 -4
- package/data/craftbook-templates/ho/hotfix-flow/versions/2.0.2/craftbook.json +610 -0
- package/data/craftbook-templates/ho/hotfix-flow/versions/2.0.2/test.json +220 -0
- package/data/craftbook-templates/ht/html-arcade-game/versions/1.2.3/craftbook.json +343 -0
- package/data/craftbook-templates/ht/html-arcade-game/versions/1.2.3/test.json +176 -0
- package/data/craftbook-templates/id/idea-office-hours/versions/2.0.8/craftbook.json +566 -0
- package/data/craftbook-templates/id/idea-office-hours/versions/2.0.8/test.json +141 -0
- package/data/craftbook-templates/im/image-set-index/versions/1.2.3/craftbook.json +298 -0
- package/data/craftbook-templates/im/image-set-index/versions/1.2.3/test.json +180 -0
- package/data/craftbook-templates/in/insurance-inventory/versions/1.0.3/test.json +11 -4
- package/data/craftbook-templates/in/interactive-quiz/versions/1.0.3/test.json +48 -23
- package/data/craftbook-templates/in/invitation-design/versions/1.0.3/test.json +1 -1
- package/data/craftbook-templates/index.json +1 -1
- package/data/craftbook-templates/it/item-intake/versions/1.0.3/test.json +11 -4
- package/data/craftbook-templates/le/lead-enrichment/versions/1.0.4/craftbook.json +277 -0
- package/data/craftbook-templates/le/lead-enrichment/versions/1.0.4/test.json +104 -0
- package/data/craftbook-templates/me/meeting-prep-brief/versions/1.0.3/test.json +16 -8
- package/data/craftbook-templates/me/memory-prompt-session/versions/1.0.3/test.json +3 -0
- package/data/craftbook-templates/mo/morning-report/versions/1.0.3/test.json +21 -9
- package/data/craftbook-templates/of/office-hours/versions/1.1.1/test.json +8 -8
- package/data/craftbook-templates/pe/perf-optimization/versions/2.0.2/craftbook.json +697 -0
- package/data/craftbook-templates/pe/perf-optimization/versions/2.0.2/test.json +235 -0
- package/data/craftbook-templates/pe/pest-diagnosis/versions/1.0.3/test.json +19 -8
- package/data/craftbook-templates/pl/plan/versions/1.0.2/test.json +8 -8
- package/data/craftbook-templates/pl/playtest-report/versions/1.0.3/test.json +17 -7
- package/data/craftbook-templates/pr/practice-exam/versions/1.0.3/test.json +7 -2
- package/data/craftbook-templates/pr/practice-session/versions/1.0.3/test.json +11 -4
- package/data/craftbook-templates/pu/pull-request-review/versions/1.9.2/craftbook.json +435 -0
- package/data/craftbook-templates/pu/pull-request-review/versions/1.9.2/test.json +141 -0
- package/data/craftbook-templates/qu/quarterly-summary/versions/1.0.3/test.json +16 -8
- package/data/craftbook-templates/re/receipt-intake/versions/1.0.3/test.json +11 -4
- package/data/craftbook-templates/re/receipt-ocr-ledger/versions/1.0.4/craftbook.json +277 -0
- package/data/craftbook-templates/re/receipt-ocr-ledger/versions/1.0.4/test.json +104 -0
- package/data/craftbook-templates/re/recipe-capture/versions/1.0.3/test.json +13 -5
- package/data/craftbook-templates/re/record-a-relative/versions/1.0.3/test.json +3 -0
- package/data/craftbook-templates/re/recurring-invoice-run/versions/1.0.4/craftbook.json +285 -0
- package/data/craftbook-templates/re/recurring-invoice-run/versions/1.0.4/test.json +104 -0
- package/data/craftbook-templates/re/refactor-module/versions/2.0.2/craftbook.json +695 -0
- package/data/craftbook-templates/re/refactor-module/versions/2.0.2/test.json +241 -0
- package/data/craftbook-templates/re/release-artifact-sanity-check/versions/1.0.4/craftbook.json +361 -0
- package/data/craftbook-templates/re/release-artifact-sanity-check/versions/1.0.4/test.json +138 -0
- package/data/craftbook-templates/re/release-notes/versions/1.0.4/test.json +8 -8
- package/data/craftbook-templates/re/release-notes/versions/1.0.5/craftbook.json +339 -0
- package/data/craftbook-templates/re/release-notes/versions/1.0.5/test.json +111 -0
- package/data/craftbook-templates/re/reliability-review/versions/1.0.3/test.json +16 -8
- package/data/craftbook-templates/re/research-report/versions/1.0.3/test.json +16 -8
- package/data/craftbook-templates/ro/root-cause-investigation/versions/2.0.8/craftbook.json +596 -0
- package/data/craftbook-templates/ro/root-cause-investigation/versions/2.0.8/test.json +157 -0
- package/data/craftbook-templates/ro/rough-cut-assembly/versions/1.0.4/test.json +13 -5
- package/data/craftbook-templates/sc/schema-migration/versions/2.0.2/craftbook.json +624 -0
- package/data/craftbook-templates/sc/schema-migration/versions/2.0.2/test.json +225 -0
- package/data/craftbook-templates/sc/script-automation/versions/1.0.4/craftbook.json +284 -0
- package/data/craftbook-templates/sc/script-automation/versions/1.0.4/test.json +162 -0
- package/data/craftbook-templates/se/security-architecture-review/versions/2.0.7/test.json +10 -13
- package/data/craftbook-templates/se/security-architecture-review/versions/2.0.8/craftbook.json +597 -0
- package/data/craftbook-templates/se/security-architecture-review/versions/2.0.8/test.json +157 -0
- package/data/craftbook-templates/so/social-digest/versions/1.0.3/craftbook.json +239 -0
- package/data/craftbook-templates/so/social-digest/versions/1.0.3/test.json +168 -0
- package/data/craftbook-templates/so/source-quality-audit/versions/1.0.3/test.json +25 -11
- package/data/craftbook-templates/sp/spec-authoring/versions/2.0.8/craftbook.json +599 -0
- package/data/craftbook-templates/sp/spec-authoring/versions/2.0.8/test.json +162 -0
- package/data/craftbook-templates/te/technical-documentation/versions/2.0.8/craftbook.json +577 -0
- package/data/craftbook-templates/te/technical-documentation/versions/2.0.8/test.json +174 -0
- package/data/craftbook-templates/te/test-coverage-review/versions/1.2.1/test.json +16 -8
- package/data/craftbook-templates/te/test-suite-backfill/versions/2.0.2/craftbook.json +592 -0
- package/data/craftbook-templates/te/test-suite-backfill/versions/2.0.2/test.json +195 -0
- package/data/craftbook-templates/th/thank-you-batch/versions/1.0.3/test.json +6 -3
- package/data/craftbook-templates/th/thank-you-sweep/versions/1.0.3/test.json +11 -4
- package/data/craftbook-templates/ti/tileset-batch/versions/1.0.3/test.json +9 -3
- package/data/craftbook-templates/tr/transcribe-and-shownotes/versions/1.0.3/test.json +10 -5
- package/data/craftbook-templates/tr/transcribe-index/versions/1.0.3/test.json +6 -3
- package/data/craftbook-templates/tr/translate-content/versions/1.1.3/craftbook.json +140 -0
- package/data/craftbook-templates/tr/translate-content/versions/1.1.3/test.json +141 -0
- package/data/craftbook-templates/ty/type-safety-pass/versions/2.0.2/craftbook.json +594 -0
- package/data/craftbook-templates/ty/type-safety-pass/versions/2.0.2/test.json +221 -0
- package/data/craftbook-templates/ux/ux-update/versions/1.0.2/craftbook.json +606 -0
- package/data/craftbook-templates/ux/ux-update/versions/1.0.2/test.json +206 -0
- package/data/craftbook-templates/we/weak-spot-drill/versions/1.0.3/test.json +11 -4
- package/data/craftbook-templates/we/weekly-walkthrough/versions/1.0.3/test.json +21 -9
- package/data/craftbook-templates/wh/whitepaper/versions/1.0.3/test.json +16 -8
- package/data/craftbook-templates/ye/year-in-review/versions/1.0.3/test.json +8 -4
- package/data/project-types/index.json +1 -1
- package/data/project-types/ju/just-chat/manifest.json +1 -1
- package/data/project-types/ju/just-chat/versions/1.0.1/about.md +5 -0
- package/data/project-types/ju/just-chat/versions/1.0.1/manifest.json +19 -0
- package/package.json +1 -1
- package/schemas/chat-model-identity.schema.json +27 -0
- package/schemas/chat-model-version.schema.json +22 -0
|
@@ -31,6 +31,9 @@
|
|
|
31
31
|
"worker": {
|
|
32
32
|
"name": "Casper",
|
|
33
33
|
"role": "Curator"
|
|
34
|
+
},
|
|
35
|
+
"craftbookParams": {
|
|
36
|
+
"workPath": "tasks/eval"
|
|
34
37
|
}
|
|
35
38
|
},
|
|
36
39
|
"mocks": [],
|
|
@@ -46,23 +49,27 @@
|
|
|
46
49
|
"kind": "contains",
|
|
47
50
|
"file": "tasks/eval/report.md",
|
|
48
51
|
"pattern": "(?:^|\\n)#{1,3}\\s+\\S",
|
|
49
|
-
"flags": "i"
|
|
52
|
+
"flags": "i",
|
|
53
|
+
"artifact": true
|
|
50
54
|
},
|
|
51
55
|
{
|
|
52
56
|
"kind": "contains",
|
|
53
57
|
"file": "tasks/eval/report.md",
|
|
54
58
|
"pattern": "confiden|uncertain|not sure|unclear|unverified|unread",
|
|
55
59
|
"flags": "i",
|
|
56
|
-
"label": "identification confidence stated honestly"
|
|
60
|
+
"label": "identification confidence stated honestly",
|
|
61
|
+
"artifact": true
|
|
57
62
|
},
|
|
58
63
|
{
|
|
59
64
|
"kind": "contains",
|
|
60
65
|
"file": "tasks/eval/report.md",
|
|
61
66
|
"pattern": "photo|expert",
|
|
62
67
|
"flags": "i",
|
|
63
|
-
"label": "follow-up buckets present"
|
|
68
|
+
"label": "follow-up buckets present",
|
|
69
|
+
"artifact": true
|
|
64
70
|
}
|
|
65
|
-
]
|
|
71
|
+
],
|
|
72
|
+
"artifact": true
|
|
66
73
|
}
|
|
67
74
|
],
|
|
68
75
|
"checks": [
|
|
@@ -0,0 +1,277 @@
|
|
|
1
|
+
{
|
|
2
|
+
"id": "lead-enrichment",
|
|
3
|
+
"name": "Enrich a Leads List",
|
|
4
|
+
"description": "Take a raw leads or contacts list and enrich each row with derived and looked-up attributes — company domain, industry, size band, role seniority, normalized location, and a quality/score flag — into a clean CRM-ready file. Scopes the enrichment fields, their sources, and the dedup/normalization rules FIRST, then enriches each lead row-by-row, then verifies coverage, that nothing was fabricated when a source had no data, and that duplicates are merged. Defining 'what each field means and where it comes from' before enriching is what stops a small model from inventing plausible-but-false company data.\n\nA gallery craftbook generated from an archetype spec. It runs\n`phase → (per-phase gate) → … → evaluate → (loop) → finish`. Each build\nphase that produces a checkable artifact is followed by a **runtime\ngate-checkpoint** — the runtime verifies the artifact and routes with no\nmodel turn, looping back to redo the phase on a miss. The final `evaluate`\nstep holds a static deliverable gate plus a reviewer QA pass. What it adds\nover the generic `build-loop`: a specialist role per phase, a\ndomain-correct ordering, and a concrete per-phase quality bar.\n\nDeliverables marked \"artifact\" land in the project's artifacts drawer (`write_artifact` / `read_artifact`), not the shipped workspace — review output is not product source.\n\nPhases:\n\n1. Scope the enrichment (planner) — fields, sources, dedup + normalization rules → gated on artifact `{{workPath}}/scope.md` (markdown-notes)\n2. Enrich each lead (developer) — derive, look up, normalize, and flag each row → gated on `leads_enriched.json` (json)\n3. Verify the enrichment (reviewer) — coverage, allowed values, no fabrication, dedup → gated on artifact `{{workPath}}/verify.md` (markdown-notes)\n\nThe gates never advance with an unmet criterion, and loop back to the\nowning phase to fix named gaps.\n",
|
|
5
|
+
"entryStepId": "scope",
|
|
6
|
+
"triggers": [
|
|
7
|
+
"enrich my leads",
|
|
8
|
+
"clean up a contacts list",
|
|
9
|
+
"add company info to leads",
|
|
10
|
+
"enrich a CRM export",
|
|
11
|
+
"fill in missing lead fields"
|
|
12
|
+
],
|
|
13
|
+
"paramSchema": {
|
|
14
|
+
"type": "object",
|
|
15
|
+
"properties": {
|
|
16
|
+
"workPath": {
|
|
17
|
+
"type": "string",
|
|
18
|
+
"title": "Working folder",
|
|
19
|
+
"description": "Per-task working folder in the artifacts drawer. Defaults to this task's own folder so runs never collide; override with a stable name when you deliberately want runs to share files.",
|
|
20
|
+
"default": "{{task.dir}}"
|
|
21
|
+
}
|
|
22
|
+
}
|
|
23
|
+
},
|
|
24
|
+
"steps": [
|
|
25
|
+
{
|
|
26
|
+
"id": "scope",
|
|
27
|
+
"name": "Scope the enrichment",
|
|
28
|
+
"description": "fields, sources, dedup + normalization rules",
|
|
29
|
+
"prompt": "Define enrichment before touching the list. Step 1: load the input leads file and list its existing columns. Step 2: define each enrichment field to add (e.g. company_domain, industry, employee_band, seniority, normalized_country, lead_score) with its type, its source (derive from email domain, look up via an available tool, or normalize an existing value), and an allowed-value set where applicable (industry list, seniority levels, size bands). Step 3: define the dedup key (e.g. lowercased email, or name+company) and the merge rule for duplicates. Step 4: state the no-fabrication rule explicitly — if a field cannot be sourced, leave it blank and mark a confidence/flag, never guess. Step 5: write an acceptance-criteria checklist ('every input lead present in output', 'enriched fields use only allowed values', 'unsourced fields are blank+flagged, not invented', 'duplicates merged by the dedup key', 'output is CRM-ready in the chosen format'). Call write_task_note with fields + sources + rules + checklist and write {{workPath}}/scope.md. No enrichment yet.\n\nThe deliverable `{{workPath}}/scope.md` lands in the project's artifacts drawer — write it with `write_artifact` and read it back with `read_artifact`; the shipped workspace stays untouched.",
|
|
30
|
+
"suggestedRole": "planner",
|
|
31
|
+
"toolPolicy": {
|
|
32
|
+
"disallowBuiltinToolsets": [
|
|
33
|
+
"ai-apps",
|
|
34
|
+
"archives",
|
|
35
|
+
"audio",
|
|
36
|
+
"browser-automation",
|
|
37
|
+
"code-execution",
|
|
38
|
+
"craftbooks",
|
|
39
|
+
"data-tables",
|
|
40
|
+
"entity-intel",
|
|
41
|
+
"git",
|
|
42
|
+
"image-intel",
|
|
43
|
+
"images",
|
|
44
|
+
"role-delegation",
|
|
45
|
+
"role-delegation-escalation",
|
|
46
|
+
"security-intel",
|
|
47
|
+
"team-management",
|
|
48
|
+
"videos",
|
|
49
|
+
"web",
|
|
50
|
+
"workspace-fs-write"
|
|
51
|
+
],
|
|
52
|
+
"outputMedium": "artifact",
|
|
53
|
+
"additionalOutputMedia": [
|
|
54
|
+
"task-note"
|
|
55
|
+
]
|
|
56
|
+
},
|
|
57
|
+
"advanceWhen": {
|
|
58
|
+
"file": "{{workPath}}/scope.md",
|
|
59
|
+
"minBytes": 1,
|
|
60
|
+
"sniff": "nonempty",
|
|
61
|
+
"artifact": true
|
|
62
|
+
},
|
|
63
|
+
"gate": {
|
|
64
|
+
"at": "completion",
|
|
65
|
+
"checks": [
|
|
66
|
+
{
|
|
67
|
+
"kind": "minBytes",
|
|
68
|
+
"file": "{{workPath}}/scope.md",
|
|
69
|
+
"bytes": 120,
|
|
70
|
+
"artifact": true
|
|
71
|
+
},
|
|
72
|
+
{
|
|
73
|
+
"kind": "sniff",
|
|
74
|
+
"file": "{{workPath}}/scope.md",
|
|
75
|
+
"sniff": "nonempty",
|
|
76
|
+
"artifact": true
|
|
77
|
+
}
|
|
78
|
+
],
|
|
79
|
+
"onReject": "scope",
|
|
80
|
+
"maxAttempts": 3
|
|
81
|
+
},
|
|
82
|
+
"next": "enrich"
|
|
83
|
+
},
|
|
84
|
+
{
|
|
85
|
+
"id": "enrich",
|
|
86
|
+
"name": "Enrich each lead",
|
|
87
|
+
"description": "derive, look up, normalize, and flag each row",
|
|
88
|
+
"prompt": "Enrich the list to the locked field spec. Step 1: dedup the input by the chosen key first, merging duplicate rows per the merge rule. Step 2: for each lead, derive the deterministic fields (e.g. company_domain from a work email, normalized_country from a location string). Step 3: for lookup fields, use the available capability and constrain outputs to the allowed-value set; when a lookup returns nothing, set the field blank and flag low confidence — do not fabricate. Step 4: compute the lead_score from the agreed signals. Step 5: write the enriched, deduped dataset to the output file in the chosen format and confirm it parses. Call write_task_note with the output path, input vs output row counts, and a per-field fill-rate.",
|
|
89
|
+
"suggestedRole": "developer",
|
|
90
|
+
"toolPolicy": {
|
|
91
|
+
"disallowBuiltinToolsets": [
|
|
92
|
+
"ai-apps",
|
|
93
|
+
"archives",
|
|
94
|
+
"artifacts",
|
|
95
|
+
"audio",
|
|
96
|
+
"browser-automation",
|
|
97
|
+
"craftbooks",
|
|
98
|
+
"data-tables",
|
|
99
|
+
"entity-intel",
|
|
100
|
+
"git",
|
|
101
|
+
"image-intel",
|
|
102
|
+
"images",
|
|
103
|
+
"role-delegation",
|
|
104
|
+
"role-delegation-escalation",
|
|
105
|
+
"security-intel",
|
|
106
|
+
"team-management",
|
|
107
|
+
"videos",
|
|
108
|
+
"web"
|
|
109
|
+
],
|
|
110
|
+
"outputMedium": "workspace",
|
|
111
|
+
"additionalOutputMedia": [
|
|
112
|
+
"task-note"
|
|
113
|
+
]
|
|
114
|
+
},
|
|
115
|
+
"advanceWhen": {
|
|
116
|
+
"file": "leads_enriched.json",
|
|
117
|
+
"minBytes": 1,
|
|
118
|
+
"sniff": "json-valid"
|
|
119
|
+
},
|
|
120
|
+
"gate": {
|
|
121
|
+
"at": "completion",
|
|
122
|
+
"checks": [
|
|
123
|
+
{
|
|
124
|
+
"kind": "minBytes",
|
|
125
|
+
"file": "leads_enriched.json",
|
|
126
|
+
"bytes": 2
|
|
127
|
+
}
|
|
128
|
+
],
|
|
129
|
+
"scripts": [
|
|
130
|
+
{
|
|
131
|
+
"name": "checkJsonValid",
|
|
132
|
+
"scope": "standard",
|
|
133
|
+
"inputs": {
|
|
134
|
+
"file": "leads_enriched.json"
|
|
135
|
+
}
|
|
136
|
+
}
|
|
137
|
+
],
|
|
138
|
+
"onReject": "enrich",
|
|
139
|
+
"maxAttempts": 3
|
|
140
|
+
},
|
|
141
|
+
"next": "verify"
|
|
142
|
+
},
|
|
143
|
+
{
|
|
144
|
+
"id": "verify",
|
|
145
|
+
"name": "Verify the enrichment",
|
|
146
|
+
"description": "coverage, allowed values, no fabrication, dedup",
|
|
147
|
+
"prompt": "Verify the enriched dataset against scope. Step 1: confirm the file parses and every input lead survived (counting merges) — no rows silently dropped. Step 2: confirm enriched fields contain only allowed values (industries, seniority, size bands are in-set). Step 3: confirm unsourced fields are blank-and-flagged rather than guessed — spot-check 5 leads against their evidence (e.g. does the company_domain actually match the email?). Step 4: confirm duplicates were merged by the dedup key and none remain. Step 5: sanity-check the lead_score distribution is plausible. Write PASS/FAIL per criterion to {{workPath}}/verify.md and task notes; on failure, name the rows/fields and loop back to enrich.\n\nThe deliverable `{{workPath}}/verify.md` lands in the project's artifacts drawer — write it with `write_artifact` and read it back with `read_artifact`; the shipped workspace stays untouched.",
|
|
148
|
+
"suggestedRole": "reviewer",
|
|
149
|
+
"toolPolicy": {
|
|
150
|
+
"disallowBuiltinToolsets": [
|
|
151
|
+
"ai-apps",
|
|
152
|
+
"archives",
|
|
153
|
+
"audio",
|
|
154
|
+
"browser-automation",
|
|
155
|
+
"code-execution",
|
|
156
|
+
"craftbooks",
|
|
157
|
+
"data-tables",
|
|
158
|
+
"entity-intel",
|
|
159
|
+
"git",
|
|
160
|
+
"image-intel",
|
|
161
|
+
"images",
|
|
162
|
+
"role-delegation",
|
|
163
|
+
"role-delegation-escalation",
|
|
164
|
+
"security-intel",
|
|
165
|
+
"team-management",
|
|
166
|
+
"videos",
|
|
167
|
+
"web",
|
|
168
|
+
"workspace-fs-write"
|
|
169
|
+
],
|
|
170
|
+
"outputMedium": "artifact",
|
|
171
|
+
"additionalOutputMedia": [
|
|
172
|
+
"task-note"
|
|
173
|
+
]
|
|
174
|
+
},
|
|
175
|
+
"advanceWhen": {
|
|
176
|
+
"file": "{{workPath}}/verify.md",
|
|
177
|
+
"minBytes": 1,
|
|
178
|
+
"sniff": "nonempty",
|
|
179
|
+
"artifact": true
|
|
180
|
+
},
|
|
181
|
+
"gate": {
|
|
182
|
+
"at": "completion",
|
|
183
|
+
"checks": [
|
|
184
|
+
{
|
|
185
|
+
"kind": "minBytes",
|
|
186
|
+
"file": "leads_enriched.json",
|
|
187
|
+
"bytes": 2
|
|
188
|
+
}
|
|
189
|
+
],
|
|
190
|
+
"scripts": [
|
|
191
|
+
{
|
|
192
|
+
"name": "checkJsonValid",
|
|
193
|
+
"scope": "standard",
|
|
194
|
+
"inputs": {
|
|
195
|
+
"file": "leads_enriched.json"
|
|
196
|
+
}
|
|
197
|
+
}
|
|
198
|
+
],
|
|
199
|
+
"onReject": "verify",
|
|
200
|
+
"maxAttempts": 4
|
|
201
|
+
},
|
|
202
|
+
"next": "evaluate"
|
|
203
|
+
},
|
|
204
|
+
{
|
|
205
|
+
"id": "evaluate",
|
|
206
|
+
"name": "Evaluate",
|
|
207
|
+
"description": "Grade the deliverable against every acceptance criterion. All pass → finish; any fail → loop back and fix the gap.",
|
|
208
|
+
"prompt": "Open leads_enriched.json and verify every criterion from {{workPath}}/scope.md. Confirm it parses, every input lead is represented (accounting for merges), enriched fields use only allowed values, unsourced fields are blank+flagged rather than fabricated (spot-check 5 against evidence such as email→domain), duplicates are merged by the dedup key, and the score distribution is plausible. Write PASS/FAIL per criterion; on any failure, name the rows/fields and loop back to enrich.\n\nThen route — this is the whole point of the loop:\n\n- **Every criterion PASSES →** call `advance_task_step({ ref, stepId: \"evaluate\", next: \"finish\" })`.\n- **Any criterion FAILS →** write the specific gaps to notes, then call `advance_task_step({ ref, stepId: \"evaluate\", next: \"enrich\" })` to loop back. The builder fixes exactly those gaps.\n\nNever route to `finish` while any criterion is unmet. The build phase's completion gate already blocked a grossly-incomplete deliverable; your job is the judgment an automated check cannot make (does it actually work, read well, look right). After ~3 unproductive loops, stop and report DONE_WITH_CONCERNS so the user can step in.",
|
|
209
|
+
"suggestedRole": "reviewer",
|
|
210
|
+
"toolPolicy": {
|
|
211
|
+
"disallowBuiltinToolsets": [
|
|
212
|
+
"ai-apps",
|
|
213
|
+
"archives",
|
|
214
|
+
"artifacts",
|
|
215
|
+
"audio",
|
|
216
|
+
"browser-automation",
|
|
217
|
+
"code-execution",
|
|
218
|
+
"craftbooks",
|
|
219
|
+
"data-tables",
|
|
220
|
+
"entity-intel",
|
|
221
|
+
"git",
|
|
222
|
+
"image-intel",
|
|
223
|
+
"images",
|
|
224
|
+
"role-delegation",
|
|
225
|
+
"role-delegation-escalation",
|
|
226
|
+
"security-intel",
|
|
227
|
+
"team-management",
|
|
228
|
+
"videos",
|
|
229
|
+
"web",
|
|
230
|
+
"workspace-fs-write"
|
|
231
|
+
],
|
|
232
|
+
"outputMedium": "task-note"
|
|
233
|
+
},
|
|
234
|
+
"consumes": [
|
|
235
|
+
{
|
|
236
|
+
"file": "leads_enriched.json"
|
|
237
|
+
}
|
|
238
|
+
],
|
|
239
|
+
"next": "enrich"
|
|
240
|
+
},
|
|
241
|
+
{
|
|
242
|
+
"id": "finish",
|
|
243
|
+
"name": "Finish",
|
|
244
|
+
"description": "All acceptance criteria met. Stamp a short summary and report DONE.",
|
|
245
|
+
"prompt": "Every acceptance criterion passed. Write a one-paragraph DONE summary to task notes via `write_task_note`: what was built, the deliverable path(s), and a one-line confirmation that each criterion is met. Then report DONE.",
|
|
246
|
+
"suggestedRole": "developer",
|
|
247
|
+
"toolPolicy": {
|
|
248
|
+
"disallowBuiltinToolsets": [
|
|
249
|
+
"ai-apps",
|
|
250
|
+
"archives",
|
|
251
|
+
"artifacts",
|
|
252
|
+
"audio",
|
|
253
|
+
"browser-automation",
|
|
254
|
+
"code-execution",
|
|
255
|
+
"craftbooks",
|
|
256
|
+
"data-tables",
|
|
257
|
+
"entity-intel",
|
|
258
|
+
"git",
|
|
259
|
+
"image-intel",
|
|
260
|
+
"images",
|
|
261
|
+
"role-delegation",
|
|
262
|
+
"role-delegation-escalation",
|
|
263
|
+
"security-intel",
|
|
264
|
+
"team-management",
|
|
265
|
+
"videos",
|
|
266
|
+
"web",
|
|
267
|
+
"workspace-fs-write"
|
|
268
|
+
],
|
|
269
|
+
"outputMedium": "task-note"
|
|
270
|
+
},
|
|
271
|
+
"terminal": true
|
|
272
|
+
}
|
|
273
|
+
],
|
|
274
|
+
"version": "1.0.4",
|
|
275
|
+
"releasedAt": "2026-09-05T03:30:00Z",
|
|
276
|
+
"minGezelVersion": "1.26233"
|
|
277
|
+
}
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
{
|
|
2
|
+
"schemaVersion": 1,
|
|
3
|
+
"title": "Enrich a Leads List smoke eval",
|
|
4
|
+
"objective": "Self-contained smoke eval for the Enrich a Leads List craftbook using the data generic harness.",
|
|
5
|
+
"tags": [
|
|
6
|
+
"data"
|
|
7
|
+
],
|
|
8
|
+
"prompt": "I dropped our numbers in source/records.csv. Can you look at the data, analyze it, and put togehter some results in analysis.md?",
|
|
9
|
+
"setup": {
|
|
10
|
+
"projectName": "Enrich a Leads List Eval",
|
|
11
|
+
"about": "Self-contained eval project for lead-enrichment. Seeded inputs are under workspace/source or workspace/fixtures; final deliverable is workspace/analysis.md.",
|
|
12
|
+
"missionObjectives": "Use the Enrich a Leads List craftbook/template, read the seeded local fixtures, and write analysis.md without network calls, real credentials, or live services.",
|
|
13
|
+
"files": [
|
|
14
|
+
{
|
|
15
|
+
"path": "source/brief.md",
|
|
16
|
+
"content": "# Enrich a Leads List Eval Brief\n\nClient: Boreal Desk, a home-office accessories company.\nAudience: operations leads who need an artifact they can use this week.\n\nFixed source facts for grounding:\n- The returns desk pilot covered 18 SKUs.\n- Median first response improved from 18 hours to 6 hours.\n- Preventable refund leakage fell from 14.2% to 8.9%.\n- The top unresolved complaint is status silence after photo submission.\n- Required next actions are automated status emails, barcode-exception training, and a weekly Finance exception export.\n\nUse these facts when the task asks for prose, analysis, copy, UI content, or test data. Do not use live web services, real credentials, or current outside data.\n\nCraftbook under test: lead-enrichment - Enrich a Leads List.\n"
|
|
17
|
+
},
|
|
18
|
+
{
|
|
19
|
+
"path": "source/records.csv",
|
|
20
|
+
"content": "customer,region,orders,revenue_usd,churn_risk\nAster,North,12,1440,low\nBoreal,South,8,720,medium\nCedar,West,15,2100,high\nDune,North,5,500,low\n"
|
|
21
|
+
}
|
|
22
|
+
],
|
|
23
|
+
"worker": {
|
|
24
|
+
"name": "Jules",
|
|
25
|
+
"role": "Data Analyst"
|
|
26
|
+
}
|
|
27
|
+
},
|
|
28
|
+
"mocks": [],
|
|
29
|
+
"success": {
|
|
30
|
+
"summary": "analysis.md is a seeded-data analysis with deterministic computed facts.",
|
|
31
|
+
"deliverables": [
|
|
32
|
+
{
|
|
33
|
+
"path": "analysis.md",
|
|
34
|
+
"kind": "markdown-report",
|
|
35
|
+
"minBytes": 900,
|
|
36
|
+
"checks": [
|
|
37
|
+
{
|
|
38
|
+
"kind": "contains",
|
|
39
|
+
"file": "analysis.md",
|
|
40
|
+
"pattern": "(?:^|\\n)#{1,3}\\s+\\S",
|
|
41
|
+
"flags": "i"
|
|
42
|
+
},
|
|
43
|
+
{
|
|
44
|
+
"kind": "contains",
|
|
45
|
+
"file": "analysis.md",
|
|
46
|
+
"pattern": "40\\b",
|
|
47
|
+
"flags": "i",
|
|
48
|
+
"label": "include total orders of 40"
|
|
49
|
+
},
|
|
50
|
+
{
|
|
51
|
+
"kind": "contains",
|
|
52
|
+
"file": "analysis.md",
|
|
53
|
+
"pattern": "\\$?4,?760\\b",
|
|
54
|
+
"flags": "i",
|
|
55
|
+
"label": "include total revenue of 4760"
|
|
56
|
+
},
|
|
57
|
+
{
|
|
58
|
+
"kind": "contains",
|
|
59
|
+
"file": "analysis.md",
|
|
60
|
+
"pattern": "Cedar",
|
|
61
|
+
"flags": "i",
|
|
62
|
+
"label": "identify Cedar as top revenue and high risk"
|
|
63
|
+
},
|
|
64
|
+
{
|
|
65
|
+
"kind": "contains",
|
|
66
|
+
"file": "analysis.md",
|
|
67
|
+
"pattern": "high",
|
|
68
|
+
"flags": "i",
|
|
69
|
+
"label": "include churn-risk finding"
|
|
70
|
+
}
|
|
71
|
+
]
|
|
72
|
+
}
|
|
73
|
+
]
|
|
74
|
+
},
|
|
75
|
+
"rubric": {
|
|
76
|
+
"artifact": {
|
|
77
|
+
"path": "analysis.md",
|
|
78
|
+
"kind": "markdown"
|
|
79
|
+
},
|
|
80
|
+
"axes": [
|
|
81
|
+
{
|
|
82
|
+
"name": "accuracy",
|
|
83
|
+
"description": "Computed figures are correct against the seeded records."
|
|
84
|
+
},
|
|
85
|
+
{
|
|
86
|
+
"name": "grounding",
|
|
87
|
+
"description": "Every claim traces to the seeded data — nothing invented."
|
|
88
|
+
},
|
|
89
|
+
{
|
|
90
|
+
"name": "structure",
|
|
91
|
+
"description": "Tables and sections make the analysis checkable at a glance."
|
|
92
|
+
},
|
|
93
|
+
{
|
|
94
|
+
"name": "actionability",
|
|
95
|
+
"description": "Recommendations follow from the findings and are concretely doable."
|
|
96
|
+
}
|
|
97
|
+
]
|
|
98
|
+
},
|
|
99
|
+
"qualityFocus": [
|
|
100
|
+
"seeded data reading",
|
|
101
|
+
"computed facts",
|
|
102
|
+
"recommendation grounding"
|
|
103
|
+
]
|
|
104
|
+
}
|
|
@@ -97,41 +97,47 @@
|
|
|
97
97
|
"kind": "contains",
|
|
98
98
|
"file": "tasks/eval/report.md",
|
|
99
99
|
"pattern": "(?:^|\\n)#{1,3}\\s+\\S",
|
|
100
|
-
"flags": "i"
|
|
100
|
+
"flags": "i",
|
|
101
|
+
"artifact": true
|
|
101
102
|
},
|
|
102
103
|
{
|
|
103
104
|
"kind": "contains",
|
|
104
105
|
"file": "tasks/eval/report.md",
|
|
105
106
|
"pattern": "Dana",
|
|
106
107
|
"flags": "i",
|
|
107
|
-
"label": "attendees come from the live calendar record"
|
|
108
|
+
"label": "attendees come from the live calendar record",
|
|
109
|
+
"artifact": true
|
|
108
110
|
},
|
|
109
111
|
{
|
|
110
112
|
"kind": "contains",
|
|
111
113
|
"file": "tasks/eval/report.md",
|
|
112
114
|
"pattern": "Marcus",
|
|
113
115
|
"flags": "i",
|
|
114
|
-
"label": "the second Northwind attendee is named"
|
|
116
|
+
"label": "the second Northwind attendee is named",
|
|
117
|
+
"artifact": true
|
|
115
118
|
},
|
|
116
119
|
{
|
|
117
120
|
"kind": "contains",
|
|
118
121
|
"file": "tasks/eval/report.md",
|
|
119
122
|
"pattern": "commit|promis|agreed|owe",
|
|
120
123
|
"flags": "i",
|
|
121
|
-
"label": "last-time commitments appear with status"
|
|
124
|
+
"label": "last-time commitments appear with status",
|
|
125
|
+
"artifact": true
|
|
122
126
|
},
|
|
123
127
|
{
|
|
124
128
|
"kind": "contains",
|
|
125
129
|
"file": "tasks/eval/report.md",
|
|
126
130
|
"pattern": "migration|timeline",
|
|
127
131
|
"flags": "i",
|
|
128
|
-
"label": "the silently-dropped migration timeline resurfaces"
|
|
132
|
+
"label": "the silently-dropped migration timeline resurfaces",
|
|
133
|
+
"artifact": true
|
|
129
134
|
},
|
|
130
135
|
{
|
|
131
136
|
"kind": "contains",
|
|
132
137
|
"file": "tasks/eval/report.md",
|
|
133
138
|
"pattern": "\\?",
|
|
134
|
-
"label": "the brief asks real questions"
|
|
139
|
+
"label": "the brief asks real questions",
|
|
140
|
+
"artifact": true
|
|
135
141
|
},
|
|
136
142
|
{
|
|
137
143
|
"kind": "judge",
|
|
@@ -140,9 +146,11 @@
|
|
|
140
146
|
"sourceFiles": [
|
|
141
147
|
"tasks/eval/vendor-sync-2026-06.md"
|
|
142
148
|
],
|
|
143
|
-
"label": "grounded and elevator-readable"
|
|
149
|
+
"label": "grounded and elevator-readable",
|
|
150
|
+
"artifact": true
|
|
144
151
|
}
|
|
145
|
-
]
|
|
152
|
+
],
|
|
153
|
+
"artifact": true
|
|
146
154
|
}
|
|
147
155
|
],
|
|
148
156
|
"mocks": [
|
|
@@ -14,6 +14,9 @@
|
|
|
14
14
|
"worker": {
|
|
15
15
|
"name": "Roos",
|
|
16
16
|
"role": "Huisautomaat"
|
|
17
|
+
},
|
|
18
|
+
"craftbookParams": {
|
|
19
|
+
"workPath": "tasks/eval"
|
|
17
20
|
}
|
|
18
21
|
},
|
|
19
22
|
"mocks": [
|
|
@@ -152,9 +155,11 @@
|
|
|
152
155
|
"kind": "contains",
|
|
153
156
|
"file": "tasks/eval/states.md",
|
|
154
157
|
"pattern": "dev-\\d{2}",
|
|
155
|
-
"label": "the state evidence carries real device ids from the tools"
|
|
158
|
+
"label": "the state evidence carries real device ids from the tools",
|
|
159
|
+
"artifact": true
|
|
156
160
|
}
|
|
157
|
-
]
|
|
161
|
+
],
|
|
162
|
+
"artifact": true
|
|
158
163
|
},
|
|
159
164
|
{
|
|
160
165
|
"path": "tasks/eval/report.md",
|
|
@@ -165,35 +170,40 @@
|
|
|
165
170
|
"kind": "contains",
|
|
166
171
|
"file": "tasks/eval/report.md",
|
|
167
172
|
"pattern": "(?:^|\\n)#{1,3}\\s+\\S",
|
|
168
|
-
"flags": "i"
|
|
173
|
+
"flags": "i",
|
|
174
|
+
"artifact": true
|
|
169
175
|
},
|
|
170
176
|
{
|
|
171
177
|
"kind": "contains",
|
|
172
178
|
"file": "tasks/eval/report.md",
|
|
173
179
|
"pattern": "attention|unusual|anomal",
|
|
174
180
|
"flags": "i",
|
|
175
|
-
"label": "an anomalies-first section exists"
|
|
181
|
+
"label": "an anomalies-first section exists",
|
|
182
|
+
"artifact": true
|
|
176
183
|
},
|
|
177
184
|
{
|
|
178
185
|
"kind": "contains",
|
|
179
186
|
"file": "tasks/eval/report.md",
|
|
180
187
|
"pattern": "garage",
|
|
181
188
|
"flags": "i",
|
|
182
|
-
"label": "the garage door left open overnight is surfaced"
|
|
189
|
+
"label": "the garage door left open overnight is surfaced",
|
|
190
|
+
"artifact": true
|
|
183
191
|
},
|
|
184
192
|
{
|
|
185
193
|
"kind": "contains",
|
|
186
194
|
"file": "tasks/eval/report.md",
|
|
187
195
|
"pattern": "batter",
|
|
188
196
|
"flags": "i",
|
|
189
|
-
"label": "the low sensor battery is surfaced"
|
|
197
|
+
"label": "the low sensor battery is surfaced",
|
|
198
|
+
"artifact": true
|
|
190
199
|
},
|
|
191
200
|
{
|
|
192
201
|
"kind": "contains",
|
|
193
202
|
"file": "tasks/eval/report.md",
|
|
194
203
|
"pattern": "close|shut|charge|replace|check",
|
|
195
204
|
"flags": "i",
|
|
196
|
-
"label": "a suggested action is offered"
|
|
205
|
+
"label": "a suggested action is offered",
|
|
206
|
+
"artifact": true
|
|
197
207
|
},
|
|
198
208
|
{
|
|
199
209
|
"kind": "judge",
|
|
@@ -202,9 +212,11 @@
|
|
|
202
212
|
"sourceFiles": [
|
|
203
213
|
"tasks/eval/states.md"
|
|
204
214
|
],
|
|
205
|
-
"label": "anomalies-first and glanceable"
|
|
215
|
+
"label": "anomalies-first and glanceable",
|
|
216
|
+
"artifact": true
|
|
206
217
|
}
|
|
207
|
-
]
|
|
218
|
+
],
|
|
219
|
+
"artifact": true
|
|
208
220
|
}
|
|
209
221
|
],
|
|
210
222
|
"mocks": [
|
|
@@ -83,20 +83,20 @@
|
|
|
83
83
|
},
|
|
84
84
|
"axes": [
|
|
85
85
|
{
|
|
86
|
-
"name": "
|
|
87
|
-
"description": "
|
|
86
|
+
"name": "accuracy",
|
|
87
|
+
"description": "Computed figures are correct against the seeded records."
|
|
88
88
|
},
|
|
89
89
|
{
|
|
90
|
-
"name": "
|
|
91
|
-
"description": "
|
|
90
|
+
"name": "grounding",
|
|
91
|
+
"description": "Every claim traces to the seeded data — nothing invented."
|
|
92
92
|
},
|
|
93
93
|
{
|
|
94
|
-
"name": "
|
|
95
|
-
"description": "
|
|
94
|
+
"name": "structure",
|
|
95
|
+
"description": "Tables and sections make the analysis checkable at a glance."
|
|
96
96
|
},
|
|
97
97
|
{
|
|
98
|
-
"name": "
|
|
99
|
-
"description": "
|
|
98
|
+
"name": "actionability",
|
|
99
|
+
"description": "Recommendations follow from the findings and are concretely doable."
|
|
100
100
|
}
|
|
101
101
|
]
|
|
102
102
|
},
|