@bendyline/gilde 0.1.55 → 0.1.56
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/authoring/chat-models/glm-5.3-flash-320b-q2.json +129 -0
- package/authoring/gstack/wave.json +2 -2
- package/data/chat-models/gl/glm-5.3-flash-320b-q2/manifest.json +137 -0
- package/data/chat-models/gl/glm-5.3-flash-320b-q2/versions/1.0.0/manifest.json +23 -0
- package/data/chat-models/index.json +1 -1
- package/data/craftbook-templates/a1/a11y-audit/versions/1.1.3/test.json +16 -8
- package/data/craftbook-templates/ac/accessibility-retrofit/versions/2.0.2/craftbook.json +618 -0
- package/data/craftbook-templates/ac/accessibility-retrofit/versions/2.0.2/test.json +228 -0
- package/data/craftbook-templates/al/album-curate/versions/1.0.3/test.json +6 -3
- package/data/craftbook-templates/al/alert-rules/versions/1.0.3/test.json +8 -8
- package/data/craftbook-templates/an/anniversary-cut/versions/1.0.5/test.json +10 -5
- package/data/craftbook-templates/an/annual-document-review/versions/1.0.3/test.json +16 -8
- package/data/craftbook-templates/an/anomaly-scan/versions/1.0.3/test.json +12 -6
- package/data/craftbook-templates/ap/apply-review-findings/versions/1.0.2/craftbook.json +602 -0
- package/data/craftbook-templates/ap/apply-review-findings/versions/1.0.2/test.json +290 -0
- package/data/craftbook-templates/ar/artifact-integrity-review/versions/1.0.3/test.json +18 -9
- package/data/craftbook-templates/au/audiobook-master-pack/versions/1.0.5/test.json +10 -5
- package/data/craftbook-templates/au/auth-flow/versions/1.0.3/test.json +12 -6
- package/data/craftbook-templates/au/auth-flow/versions/1.0.4/craftbook.json +282 -0
- package/data/craftbook-templates/au/auth-flow/versions/1.0.4/test.json +143 -0
- package/data/craftbook-templates/au/automation-recipe/versions/1.0.3/test.json +12 -6
- package/data/craftbook-templates/ba/backup-routine/versions/1.0.4/craftbook.json +282 -0
- package/data/craftbook-templates/ba/backup-routine/versions/1.0.4/test.json +108 -0
- package/data/craftbook-templates/bl/blog-post/versions/1.0.3/test.json +8 -8
- package/data/craftbook-templates/br/branding-website/versions/1.1.3/craftbook.json +335 -0
- package/data/craftbook-templates/br/branding-website/versions/1.1.3/test.json +164 -0
- package/data/craftbook-templates/br/broadcast-announcement/versions/1.0.3/test.json +8 -8
- package/data/craftbook-templates/br/browser-qa-audit/versions/2.0.8/craftbook.json +621 -0
- package/data/craftbook-templates/br/browser-qa-audit/versions/2.0.8/test.json +376 -0
- package/data/craftbook-templates/bu/bug-fix-tdd/versions/2.0.2/craftbook.json +730 -0
- package/data/craftbook-templates/bu/bug-fix-tdd/versions/2.0.2/test.json +221 -0
- package/data/craftbook-templates/bu/build-loop/versions/1.1.1/test.json +8 -8
- package/data/craftbook-templates/ca/case-study/versions/1.0.3/test.json +8 -8
- package/data/craftbook-templates/ch/chapter-narration-run/versions/1.0.4/test.json +10 -5
- package/data/craftbook-templates/ch/character-turnaround/versions/1.0.3/test.json +25 -11
- package/data/craftbook-templates/ci/ci-pipeline/versions/2.0.2/craftbook.json +587 -0
- package/data/craftbook-templates/ci/ci-pipeline/versions/2.0.2/test.json +225 -0
- package/data/craftbook-templates/ci/citation-audit/versions/1.0.3/test.json +28 -14
- package/data/craftbook-templates/cl/cli-tool/versions/1.1.3/test.json +8 -8
- package/data/craftbook-templates/co/codebase-refactoring-review/versions/1.0.3/test.json +16 -8
- package/data/craftbook-templates/co/codebase-ux-review/versions/1.0.3/test.json +18 -9
- package/data/craftbook-templates/co/codemod-sweep/versions/1.0.2/craftbook.json +605 -0
- package/data/craftbook-templates/co/codemod-sweep/versions/1.0.2/test.json +252 -0
- package/data/craftbook-templates/co/cohort-analysis/versions/1.0.3/test.json +18 -9
- package/data/craftbook-templates/co/competitive-analysis/versions/1.0.3/test.json +23 -15
- package/data/craftbook-templates/co/content-accuracy-review/versions/1.0.3/test.json +28 -14
- package/data/craftbook-templates/co/content-deck/versions/1.2.3/craftbook.json +4 -4
- package/data/craftbook-templates/co/content-deck/versions/1.2.3/test.json +3 -0
- package/data/craftbook-templates/co/content-deck/versions/1.2.4/craftbook.json +341 -0
- package/data/craftbook-templates/co/content-deck/versions/1.2.4/test.json +169 -0
- package/data/craftbook-templates/co/copy-review/versions/1.0.3/test.json +14 -7
- package/data/craftbook-templates/co/corpus-email-digest/versions/1.2.3/craftbook.json +4 -4
- package/data/craftbook-templates/co/corpus-email-digest/versions/1.2.3/test.json +3 -0
- package/data/craftbook-templates/co/corpus-email-digest/versions/1.2.4/craftbook.json +345 -0
- package/data/craftbook-templates/co/corpus-email-digest/versions/1.2.4/test.json +187 -0
- package/data/craftbook-templates/co/corpus-synthesis/versions/1.0.3/test.json +26 -13
- package/data/craftbook-templates/co/cover-letter/versions/1.0.3/test.json +8 -8
- package/data/craftbook-templates/cs/csv-transformer/versions/1.2.4/test.json +7 -7
- package/data/craftbook-templates/da/dashboard-spec/versions/1.0.3/test.json +7 -7
- package/data/craftbook-templates/da/data-export-migrate/versions/1.0.4/craftbook.json +285 -0
- package/data/craftbook-templates/da/data-export-migrate/versions/1.0.4/test.json +104 -0
- package/data/craftbook-templates/da/data-pipeline-etl/versions/1.0.5/craftbook.json +284 -0
- package/data/craftbook-templates/da/data-pipeline-etl/versions/1.0.5/test.json +162 -0
- package/data/craftbook-templates/da/data-quality-audit/versions/1.0.3/test.json +38 -19
- package/data/craftbook-templates/da/data-to-report/versions/1.0.3/test.json +12 -6
- package/data/craftbook-templates/db/db-index-tuning/versions/1.0.4/craftbook.json +286 -0
- package/data/craftbook-templates/db/db-index-tuning/versions/1.0.4/test.json +104 -0
- package/data/craftbook-templates/de/dependency-audit/versions/1.1.3/test.json +14 -7
- package/data/craftbook-templates/de/dependency-upgrade/versions/1.0.2/craftbook.json +621 -0
- package/data/craftbook-templates/de/dependency-upgrade/versions/1.0.2/test.json +235 -0
- package/data/craftbook-templates/de/design-review/versions/1.0.3/test.json +16 -8
- package/data/craftbook-templates/de/design-system-consultation/versions/2.0.8/craftbook.json +616 -0
- package/data/craftbook-templates/de/design-system-consultation/versions/2.0.8/test.json +201 -0
- package/data/craftbook-templates/do/doc-intake-pipeline/versions/1.0.4/craftbook.json +285 -0
- package/data/craftbook-templates/do/doc-intake-pipeline/versions/1.0.4/test.json +104 -0
- package/data/craftbook-templates/do/doc-rewrite/versions/1.0.3/test.json +119 -55
- package/data/craftbook-templates/do/dockerize-app/versions/1.0.4/craftbook.json +285 -0
- package/data/craftbook-templates/do/dockerize-app/versions/1.0.4/test.json +92 -0
- package/data/craftbook-templates/do/documentation-drift-review/versions/1.0.3/test.json +18 -9
- package/data/craftbook-templates/dr/draft-social-post/versions/1.0.2/craftbook.json +324 -0
- package/data/craftbook-templates/dr/draft-social-post/versions/1.0.2/test.json +161 -0
- package/data/craftbook-templates/ed/edit-notes/versions/1.0.3/test.json +11 -4
- package/data/craftbook-templates/en/engineering-retrospective/versions/2.0.8/craftbook.json +566 -0
- package/data/craftbook-templates/en/engineering-retrospective/versions/2.0.8/test.json +191 -0
- package/data/craftbook-templates/ep/episode-plan/versions/1.0.3/test.json +10 -5
- package/data/craftbook-templates/ex/executive-level-review/versions/2.0.7/test.json +7 -10
- package/data/craftbook-templates/ex/executive-level-review/versions/2.0.8/craftbook.json +595 -0
- package/data/craftbook-templates/ex/executive-level-review/versions/2.0.8/test.json +139 -0
- package/data/craftbook-templates/ex/expense-categorize/versions/1.0.4/craftbook.json +277 -0
- package/data/craftbook-templates/ex/expense-categorize/versions/1.0.4/test.json +104 -0
- package/data/craftbook-templates/fe/feature-flag-rollout/versions/1.1.2/craftbook.json +327 -0
- package/data/craftbook-templates/fe/feature-flag-rollout/versions/1.1.2/test.json +162 -0
- package/data/craftbook-templates/fl/flaky-test-fix/versions/1.0.2/craftbook.json +718 -0
- package/data/craftbook-templates/fl/flaky-test-fix/versions/1.0.2/test.json +239 -0
- package/data/craftbook-templates/fo/form-fill-batch/versions/1.0.4/craftbook.json +285 -0
- package/data/craftbook-templates/fo/form-fill-batch/versions/1.0.4/test.json +162 -0
- package/data/craftbook-templates/fo/form-wizard/versions/1.0.3/test.json +8 -8
- package/data/craftbook-templates/ho/holiday-card-run/versions/1.0.3/test.json +11 -4
- package/data/craftbook-templates/ho/hotfix-flow/versions/2.0.2/craftbook.json +610 -0
- package/data/craftbook-templates/ho/hotfix-flow/versions/2.0.2/test.json +220 -0
- package/data/craftbook-templates/ht/html-arcade-game/versions/1.2.3/craftbook.json +343 -0
- package/data/craftbook-templates/ht/html-arcade-game/versions/1.2.3/test.json +176 -0
- package/data/craftbook-templates/id/idea-office-hours/versions/2.0.8/craftbook.json +566 -0
- package/data/craftbook-templates/id/idea-office-hours/versions/2.0.8/test.json +141 -0
- package/data/craftbook-templates/im/image-set-index/versions/1.2.3/craftbook.json +298 -0
- package/data/craftbook-templates/im/image-set-index/versions/1.2.3/test.json +180 -0
- package/data/craftbook-templates/in/insurance-inventory/versions/1.0.3/test.json +11 -4
- package/data/craftbook-templates/in/interactive-quiz/versions/1.0.3/test.json +48 -23
- package/data/craftbook-templates/in/invitation-design/versions/1.0.3/test.json +1 -1
- package/data/craftbook-templates/index.json +1 -1
- package/data/craftbook-templates/it/item-intake/versions/1.0.3/test.json +11 -4
- package/data/craftbook-templates/le/lead-enrichment/versions/1.0.4/craftbook.json +277 -0
- package/data/craftbook-templates/le/lead-enrichment/versions/1.0.4/test.json +104 -0
- package/data/craftbook-templates/me/meeting-prep-brief/versions/1.0.3/test.json +16 -8
- package/data/craftbook-templates/me/memory-prompt-session/versions/1.0.3/test.json +3 -0
- package/data/craftbook-templates/mo/morning-report/versions/1.0.3/test.json +21 -9
- package/data/craftbook-templates/of/office-hours/versions/1.1.1/test.json +8 -8
- package/data/craftbook-templates/pe/perf-optimization/versions/2.0.2/craftbook.json +697 -0
- package/data/craftbook-templates/pe/perf-optimization/versions/2.0.2/test.json +235 -0
- package/data/craftbook-templates/pe/pest-diagnosis/versions/1.0.3/test.json +19 -8
- package/data/craftbook-templates/pl/plan/versions/1.0.2/test.json +8 -8
- package/data/craftbook-templates/pl/playtest-report/versions/1.0.3/test.json +17 -7
- package/data/craftbook-templates/pr/practice-exam/versions/1.0.3/test.json +7 -2
- package/data/craftbook-templates/pr/practice-session/versions/1.0.3/test.json +11 -4
- package/data/craftbook-templates/pu/pull-request-review/versions/1.9.2/craftbook.json +435 -0
- package/data/craftbook-templates/pu/pull-request-review/versions/1.9.2/test.json +141 -0
- package/data/craftbook-templates/qu/quarterly-summary/versions/1.0.3/test.json +16 -8
- package/data/craftbook-templates/re/receipt-intake/versions/1.0.3/test.json +11 -4
- package/data/craftbook-templates/re/receipt-ocr-ledger/versions/1.0.4/craftbook.json +277 -0
- package/data/craftbook-templates/re/receipt-ocr-ledger/versions/1.0.4/test.json +104 -0
- package/data/craftbook-templates/re/recipe-capture/versions/1.0.3/test.json +13 -5
- package/data/craftbook-templates/re/record-a-relative/versions/1.0.3/test.json +3 -0
- package/data/craftbook-templates/re/recurring-invoice-run/versions/1.0.4/craftbook.json +285 -0
- package/data/craftbook-templates/re/recurring-invoice-run/versions/1.0.4/test.json +104 -0
- package/data/craftbook-templates/re/refactor-module/versions/2.0.2/craftbook.json +695 -0
- package/data/craftbook-templates/re/refactor-module/versions/2.0.2/test.json +241 -0
- package/data/craftbook-templates/re/release-artifact-sanity-check/versions/1.0.4/craftbook.json +361 -0
- package/data/craftbook-templates/re/release-artifact-sanity-check/versions/1.0.4/test.json +138 -0
- package/data/craftbook-templates/re/release-notes/versions/1.0.4/test.json +8 -8
- package/data/craftbook-templates/re/release-notes/versions/1.0.5/craftbook.json +339 -0
- package/data/craftbook-templates/re/release-notes/versions/1.0.5/test.json +111 -0
- package/data/craftbook-templates/re/reliability-review/versions/1.0.3/test.json +16 -8
- package/data/craftbook-templates/re/research-report/versions/1.0.3/test.json +16 -8
- package/data/craftbook-templates/ro/root-cause-investigation/versions/2.0.8/craftbook.json +596 -0
- package/data/craftbook-templates/ro/root-cause-investigation/versions/2.0.8/test.json +157 -0
- package/data/craftbook-templates/ro/rough-cut-assembly/versions/1.0.4/test.json +13 -5
- package/data/craftbook-templates/sc/schema-migration/versions/2.0.2/craftbook.json +624 -0
- package/data/craftbook-templates/sc/schema-migration/versions/2.0.2/test.json +225 -0
- package/data/craftbook-templates/sc/script-automation/versions/1.0.4/craftbook.json +284 -0
- package/data/craftbook-templates/sc/script-automation/versions/1.0.4/test.json +162 -0
- package/data/craftbook-templates/se/security-architecture-review/versions/2.0.7/test.json +10 -13
- package/data/craftbook-templates/se/security-architecture-review/versions/2.0.8/craftbook.json +597 -0
- package/data/craftbook-templates/se/security-architecture-review/versions/2.0.8/test.json +157 -0
- package/data/craftbook-templates/so/social-digest/versions/1.0.3/craftbook.json +239 -0
- package/data/craftbook-templates/so/social-digest/versions/1.0.3/test.json +168 -0
- package/data/craftbook-templates/so/source-quality-audit/versions/1.0.3/test.json +25 -11
- package/data/craftbook-templates/sp/spec-authoring/versions/2.0.8/craftbook.json +599 -0
- package/data/craftbook-templates/sp/spec-authoring/versions/2.0.8/test.json +162 -0
- package/data/craftbook-templates/te/technical-documentation/versions/2.0.8/craftbook.json +577 -0
- package/data/craftbook-templates/te/technical-documentation/versions/2.0.8/test.json +174 -0
- package/data/craftbook-templates/te/test-coverage-review/versions/1.2.1/test.json +16 -8
- package/data/craftbook-templates/te/test-suite-backfill/versions/2.0.2/craftbook.json +592 -0
- package/data/craftbook-templates/te/test-suite-backfill/versions/2.0.2/test.json +195 -0
- package/data/craftbook-templates/th/thank-you-batch/versions/1.0.3/test.json +6 -3
- package/data/craftbook-templates/th/thank-you-sweep/versions/1.0.3/test.json +11 -4
- package/data/craftbook-templates/ti/tileset-batch/versions/1.0.3/test.json +9 -3
- package/data/craftbook-templates/tr/transcribe-and-shownotes/versions/1.0.3/test.json +10 -5
- package/data/craftbook-templates/tr/transcribe-index/versions/1.0.3/test.json +6 -3
- package/data/craftbook-templates/tr/translate-content/versions/1.1.3/craftbook.json +140 -0
- package/data/craftbook-templates/tr/translate-content/versions/1.1.3/test.json +141 -0
- package/data/craftbook-templates/ty/type-safety-pass/versions/2.0.2/craftbook.json +594 -0
- package/data/craftbook-templates/ty/type-safety-pass/versions/2.0.2/test.json +221 -0
- package/data/craftbook-templates/ux/ux-update/versions/1.0.2/craftbook.json +606 -0
- package/data/craftbook-templates/ux/ux-update/versions/1.0.2/test.json +206 -0
- package/data/craftbook-templates/we/weak-spot-drill/versions/1.0.3/test.json +11 -4
- package/data/craftbook-templates/we/weekly-walkthrough/versions/1.0.3/test.json +21 -9
- package/data/craftbook-templates/wh/whitepaper/versions/1.0.3/test.json +16 -8
- package/data/craftbook-templates/ye/year-in-review/versions/1.0.3/test.json +8 -4
- package/data/project-types/index.json +1 -1
- package/data/project-types/ju/just-chat/manifest.json +1 -1
- package/data/project-types/ju/just-chat/versions/1.0.1/about.md +5 -0
- package/data/project-types/ju/just-chat/versions/1.0.1/manifest.json +19 -0
- package/package.json +1 -1
- package/schemas/chat-model-identity.schema.json +27 -0
- package/schemas/chat-model-version.schema.json +22 -0
|
@@ -0,0 +1,139 @@
|
|
|
1
|
+
{
|
|
2
|
+
"schemaVersion": 1,
|
|
3
|
+
"title": "Executive-Level Review — analytics expansion decision",
|
|
4
|
+
"objective": "Test whether the Executive-Level Review craftbook challenges a proposal, compares lower-cost alternatives, and produces an evidence-anchored mode verdict with decision conditions.",
|
|
5
|
+
"tags": [
|
|
6
|
+
"workflow",
|
|
7
|
+
"strategy",
|
|
8
|
+
"executive-review",
|
|
9
|
+
"decision"
|
|
10
|
+
],
|
|
11
|
+
"prompt": "Use the Executive-Level Review craftbook to review `source/analytics-proposal.md`. Challenge whether the proposed scope is the best way to reach the outcome; compare at least three alternatives including the two-day manual export and waiting. Produce all craftbook artifacts and a decision-ready `tasks/eval/reviews/executive-level-review.md`. Use exactly one allowed mode verdict, score every required dimension from 0–10 with a factual basis, separate unknowns from evidence, and state explicit approve, pause, and stop conditions. Do not invent customer or financial evidence.",
|
|
12
|
+
"setup": {
|
|
13
|
+
"projectName": "Analytics expansion review",
|
|
14
|
+
"about": "A hermetic executive decision review. The seeded proposal is the complete evidence set.",
|
|
15
|
+
"missionObjectives": "Decide what, if anything, should be approved now while preserving the fastest credible route to learning.",
|
|
16
|
+
"files": [
|
|
17
|
+
{
|
|
18
|
+
"path": "source/analytics-proposal.md",
|
|
19
|
+
"content": "# Embedded analytics proposal\n\n## Requested decision\nApprove an eight-week build of customer-facing dashboards with scheduled PDF delivery and CSV export.\n\n## Evidence\nTwelve of 86 active customers requested better reporting in support conversations. Three customers agreed to a design pilot, but none signed a paid commitment. Support currently creates weekly exports for five customers.\n\n## Cost and constraints\n- Estimate: two engineers for 8 weeks.\n- New charting service: $4,000 per month at current volume.\n- Security review is complete; accessibility and data-retention reviews are not.\n- The enterprise renewal decision is in six weeks.\n- A manual scheduled-export workflow can be built in two engineer-days and would test delivery cadence, but not dashboard engagement.\n\n## Unknowns\nWillingness to pay, which metrics matter, PDF accessibility, and whether scheduled delivery or interactive dashboards drives retention.\n"
|
|
20
|
+
}
|
|
21
|
+
],
|
|
22
|
+
"craftbookParams": {
|
|
23
|
+
"workPath": "tasks/eval"
|
|
24
|
+
}
|
|
25
|
+
},
|
|
26
|
+
"mocks": [],
|
|
27
|
+
"success": {
|
|
28
|
+
"summary": "The named craftbook produces a grounded executive verdict, alternatives, scorecard, and terminal task record.",
|
|
29
|
+
"deliverables": [
|
|
30
|
+
{
|
|
31
|
+
"path": "tasks/eval/reviews/executive-level-review.md",
|
|
32
|
+
"kind": "markdown-report",
|
|
33
|
+
"artifact": true,
|
|
34
|
+
"minBytes": 1400,
|
|
35
|
+
"checks": [
|
|
36
|
+
{
|
|
37
|
+
"kind": "contains",
|
|
38
|
+
"file": "tasks/eval/reviews/executive-level-review.md",
|
|
39
|
+
"pattern": "^#{1,3}\\s+Executive verdict\\b[\\s\\S]*(SCOPE EXPANSION|SELECTIVE|HOLD|SCOPE REDUCTION)[\\s\\S]*^#{1,3}\\s+Scorecard\\b[\\s\\S]*^#{1,3}\\s+Alternatives\\b[\\s\\S]*^#{1,3}\\s+Recommendation\\b[\\s\\S]*^#{1,3}\\s+Decision conditions\\b",
|
|
40
|
+
"flags": "im",
|
|
41
|
+
"label": "verdict, scorecard, alternatives, recommendation, and conditions"
|
|
42
|
+
},
|
|
43
|
+
{
|
|
44
|
+
"kind": "contains",
|
|
45
|
+
"file": "tasks/eval/reviews/executive-level-review.md",
|
|
46
|
+
"pattern": "(user value).*?(strategic fit).*?(evidence strength).*?(feasibility).*?(sequencing).*?(reversibility).*?(operating cost).*?(downside)",
|
|
47
|
+
"flags": "is",
|
|
48
|
+
"label": "complete executive scorecard"
|
|
49
|
+
},
|
|
50
|
+
{
|
|
51
|
+
"kind": "valueGrounding",
|
|
52
|
+
"file": "tasks/eval/reviews/executive-level-review.md",
|
|
53
|
+
"facts": [
|
|
54
|
+
{
|
|
55
|
+
"id": "customer-requests",
|
|
56
|
+
"required": [
|
|
57
|
+
"12\\s+(?:of\\s+86\\s+)?(?:active\\s+)?customers"
|
|
58
|
+
]
|
|
59
|
+
},
|
|
60
|
+
{
|
|
61
|
+
"id": "pilot-evidence",
|
|
62
|
+
"required": [
|
|
63
|
+
"3\\s+customers?[^.]{0,60}(pilot|agreed)"
|
|
64
|
+
]
|
|
65
|
+
},
|
|
66
|
+
{
|
|
67
|
+
"id": "build-duration",
|
|
68
|
+
"required": [
|
|
69
|
+
"8\\s+weeks?"
|
|
70
|
+
]
|
|
71
|
+
},
|
|
72
|
+
{
|
|
73
|
+
"id": "operating-cost",
|
|
74
|
+
"required": [
|
|
75
|
+
"\\$4,?000\\s+per\\s+month"
|
|
76
|
+
]
|
|
77
|
+
},
|
|
78
|
+
{
|
|
79
|
+
"id": "cheap-alternative",
|
|
80
|
+
"required": [
|
|
81
|
+
"(two|2)[ -]?(engineer[- ])?days?[\\s\\S]{0,120}(manual|scheduled)[ -]export"
|
|
82
|
+
]
|
|
83
|
+
}
|
|
84
|
+
]
|
|
85
|
+
},
|
|
86
|
+
{
|
|
87
|
+
"kind": "citationsResolve",
|
|
88
|
+
"file": "tasks/eval/reviews/executive-level-review.md",
|
|
89
|
+
"minCitations": 1
|
|
90
|
+
}
|
|
91
|
+
]
|
|
92
|
+
}
|
|
93
|
+
],
|
|
94
|
+
"taskNotes": {
|
|
95
|
+
"minBytes": 160,
|
|
96
|
+
"checks": [
|
|
97
|
+
{
|
|
98
|
+
"kind": "contains",
|
|
99
|
+
"file": "task-notes.md",
|
|
100
|
+
"pattern": "\\bDONE\\b[\\s\\S]*tasks/eval/reviews/executive-level-review\\.md",
|
|
101
|
+
"label": "terminal craftbook note names the review"
|
|
102
|
+
}
|
|
103
|
+
],
|
|
104
|
+
"requireCraftbookTask": true
|
|
105
|
+
},
|
|
106
|
+
"taskGraph": {
|
|
107
|
+
"requireCraftbookTask": true,
|
|
108
|
+
"requireTerminalStep": true
|
|
109
|
+
},
|
|
110
|
+
"unchangedFixtures": [
|
|
111
|
+
"source/analytics-proposal.md"
|
|
112
|
+
]
|
|
113
|
+
},
|
|
114
|
+
"rubric": {
|
|
115
|
+
"artifact": {
|
|
116
|
+
"path": "tasks/eval/reviews/executive-level-review.md",
|
|
117
|
+
"kind": "markdown"
|
|
118
|
+
},
|
|
119
|
+
"axes": [
|
|
120
|
+
{
|
|
121
|
+
"name": "Premise challenge",
|
|
122
|
+
"description": "The review tests whether action is warranted now and compares materially different ways to learn or deliver value."
|
|
123
|
+
},
|
|
124
|
+
{
|
|
125
|
+
"name": "Evidence discipline",
|
|
126
|
+
"description": "Scores and conclusions trace to supplied facts, with unsupported assumptions and missing decisions called out."
|
|
127
|
+
},
|
|
128
|
+
{
|
|
129
|
+
"name": "Executive usefulness",
|
|
130
|
+
"description": "The verdict is decisive, sequenced, reversible where possible, and paired with clear decision-changing conditions."
|
|
131
|
+
}
|
|
132
|
+
]
|
|
133
|
+
},
|
|
134
|
+
"qualityFocus": [
|
|
135
|
+
"One allowed mode verdict is selected",
|
|
136
|
+
"All scorecard dimensions have evidence-backed scores",
|
|
137
|
+
"The two-day alternative and missing willingness-to-pay evidence influence the decision"
|
|
138
|
+
]
|
|
139
|
+
}
|
|
@@ -0,0 +1,277 @@
|
|
|
1
|
+
{
|
|
2
|
+
"id": "expense-categorize",
|
|
3
|
+
"name": "Categorize Transactions",
|
|
4
|
+
"description": "Classify a list of bank/card transactions into a fixed chart-of-accounts category set, with confidence and a review queue for the uncertain ones, producing a clean categorized ledger ready for bookkeeping or tax. Locks the category taxonomy and the matching rules (merchant patterns, amount/sign heuristics) FIRST, then categorizes every transaction deterministically-first then by inference, then reviews that the taxonomy is closed, low-confidence rows are queued not force-fit, and totals reconcile to the input. Rules-before-categorize is the wisdom: a defined taxonomy plus rule precedence makes results auditable and repeatable, not a one-off guess.\n\nA gallery craftbook generated from an archetype spec. It runs\n`phase → (per-phase gate) → … → evaluate → (loop) → finish`. Each build\nphase that produces a checkable artifact is followed by a **runtime\ngate-checkpoint** — the runtime verifies the artifact and routes with no\nmodel turn, looping back to redo the phase on a miss. The final `evaluate`\nstep holds a static deliverable gate plus a reviewer QA pass. What it adds\nover the generic `build-loop`: a specialist role per phase, a\ndomain-correct ordering, and a concrete per-phase quality bar.\n\nDeliverables marked \"artifact\" land in the project's artifacts drawer (`write_artifact` / `read_artifact`), not the shipped workspace — review output is not product source.\n\nPhases:\n\n1. Scope the rules (planner) — category taxonomy + matching rules + precedence → gated on artifact `{{workPath}}/rules-scope.md` (markdown-notes)\n2. Categorize transactions (developer) — apply rules, assign category + confidence, queue doubts → gated on `ledger.json` (json)\n3. Review the ledger (reviewer) — closed taxonomy, reconciliation, review queue → gated on artifact `{{workPath}}/review.md` (markdown-notes)\n\nThe gates never advance with an unmet criterion, and loop back to the\nowning phase to fix named gaps.\n",
|
|
5
|
+
"entryStepId": "rules-scope",
|
|
6
|
+
"triggers": [
|
|
7
|
+
"categorize my transactions",
|
|
8
|
+
"sort expenses by category",
|
|
9
|
+
"clean up bank transactions",
|
|
10
|
+
"bookkeeping categorization",
|
|
11
|
+
"tag spending"
|
|
12
|
+
],
|
|
13
|
+
"paramSchema": {
|
|
14
|
+
"type": "object",
|
|
15
|
+
"properties": {
|
|
16
|
+
"workPath": {
|
|
17
|
+
"type": "string",
|
|
18
|
+
"title": "Working folder",
|
|
19
|
+
"description": "Per-task working folder in the artifacts drawer. Defaults to this task's own folder so runs never collide; override with a stable name when you deliberately want runs to share files.",
|
|
20
|
+
"default": "{{task.dir}}"
|
|
21
|
+
}
|
|
22
|
+
}
|
|
23
|
+
},
|
|
24
|
+
"steps": [
|
|
25
|
+
{
|
|
26
|
+
"id": "rules-scope",
|
|
27
|
+
"name": "Scope the rules",
|
|
28
|
+
"description": "category taxonomy + matching rules + precedence",
|
|
29
|
+
"prompt": "Define the taxonomy and rules before categorizing. Step 1: load the transactions (date, description/merchant, amount, sign) and list the columns. Step 2: define the closed category set (the chart of accounts, e.g. Income, Office, Software, Travel, Meals, Utilities, Transfers, Uncategorized) — every transaction must land in exactly one. Step 3: define matching rules with precedence: explicit merchant→category patterns first, then heuristics (sign for income vs expense, amount bands, recurring-amount detection), and an Uncategorized fallback. Step 4: define the confidence rule and the review threshold (below which a row goes to a review queue rather than being force-categorized). Step 5: write an acceptance-criteria checklist ('every transaction categorized into exactly one in-set category', 'rule precedence applied deterministically', 'low-confidence rows queued for review', 'per-category totals reconcile to the input sum', 'no category invented outside the taxonomy'). Call write_task_note with taxonomy + rules + checklist and write {{workPath}}/rules-scope.md. No categorizing yet.\n\nThe deliverable `{{workPath}}/rules-scope.md` lands in the project's artifacts drawer — write it with `write_artifact` and read it back with `read_artifact`; the shipped workspace stays untouched.",
|
|
30
|
+
"suggestedRole": "planner",
|
|
31
|
+
"toolPolicy": {
|
|
32
|
+
"disallowBuiltinToolsets": [
|
|
33
|
+
"ai-apps",
|
|
34
|
+
"archives",
|
|
35
|
+
"audio",
|
|
36
|
+
"browser-automation",
|
|
37
|
+
"code-execution",
|
|
38
|
+
"craftbooks",
|
|
39
|
+
"data-tables",
|
|
40
|
+
"entity-intel",
|
|
41
|
+
"git",
|
|
42
|
+
"image-intel",
|
|
43
|
+
"images",
|
|
44
|
+
"role-delegation",
|
|
45
|
+
"role-delegation-escalation",
|
|
46
|
+
"security-intel",
|
|
47
|
+
"team-management",
|
|
48
|
+
"videos",
|
|
49
|
+
"web",
|
|
50
|
+
"workspace-fs-write"
|
|
51
|
+
],
|
|
52
|
+
"outputMedium": "artifact",
|
|
53
|
+
"additionalOutputMedia": [
|
|
54
|
+
"task-note"
|
|
55
|
+
]
|
|
56
|
+
},
|
|
57
|
+
"advanceWhen": {
|
|
58
|
+
"file": "{{workPath}}/rules-scope.md",
|
|
59
|
+
"minBytes": 1,
|
|
60
|
+
"sniff": "nonempty",
|
|
61
|
+
"artifact": true
|
|
62
|
+
},
|
|
63
|
+
"gate": {
|
|
64
|
+
"at": "completion",
|
|
65
|
+
"checks": [
|
|
66
|
+
{
|
|
67
|
+
"kind": "minBytes",
|
|
68
|
+
"file": "{{workPath}}/rules-scope.md",
|
|
69
|
+
"bytes": 120,
|
|
70
|
+
"artifact": true
|
|
71
|
+
},
|
|
72
|
+
{
|
|
73
|
+
"kind": "sniff",
|
|
74
|
+
"file": "{{workPath}}/rules-scope.md",
|
|
75
|
+
"sniff": "nonempty",
|
|
76
|
+
"artifact": true
|
|
77
|
+
}
|
|
78
|
+
],
|
|
79
|
+
"onReject": "rules-scope",
|
|
80
|
+
"maxAttempts": 3
|
|
81
|
+
},
|
|
82
|
+
"next": "categorize"
|
|
83
|
+
},
|
|
84
|
+
{
|
|
85
|
+
"id": "categorize",
|
|
86
|
+
"name": "Categorize transactions",
|
|
87
|
+
"description": "apply rules, assign category + confidence, queue doubts",
|
|
88
|
+
"prompt": "Categorize every transaction to the locked taxonomy. Step 1: for each transaction, apply the rules in precedence order — explicit merchant match first, then heuristics, then Uncategorized. Step 2: assign exactly one in-set category plus a confidence and a short reason (which rule fired). Step 3: route rows below the review threshold to a review queue flag rather than forcing a category. Step 4: produce the categorized ledger (the input columns plus category, confidence, reason, needs_review) in the chosen format. Step 5: compute per-category totals and a grand total and confirm the grand total equals the sum of the input amounts. Call write_task_note with the output path, the per-category totals, and the count of rows queued for review.",
|
|
89
|
+
"suggestedRole": "developer",
|
|
90
|
+
"toolPolicy": {
|
|
91
|
+
"disallowBuiltinToolsets": [
|
|
92
|
+
"ai-apps",
|
|
93
|
+
"archives",
|
|
94
|
+
"artifacts",
|
|
95
|
+
"audio",
|
|
96
|
+
"browser-automation",
|
|
97
|
+
"craftbooks",
|
|
98
|
+
"data-tables",
|
|
99
|
+
"entity-intel",
|
|
100
|
+
"git",
|
|
101
|
+
"image-intel",
|
|
102
|
+
"images",
|
|
103
|
+
"role-delegation",
|
|
104
|
+
"role-delegation-escalation",
|
|
105
|
+
"security-intel",
|
|
106
|
+
"team-management",
|
|
107
|
+
"videos",
|
|
108
|
+
"web"
|
|
109
|
+
],
|
|
110
|
+
"outputMedium": "workspace",
|
|
111
|
+
"additionalOutputMedia": [
|
|
112
|
+
"task-note"
|
|
113
|
+
]
|
|
114
|
+
},
|
|
115
|
+
"advanceWhen": {
|
|
116
|
+
"file": "ledger.json",
|
|
117
|
+
"minBytes": 1,
|
|
118
|
+
"sniff": "json-valid"
|
|
119
|
+
},
|
|
120
|
+
"gate": {
|
|
121
|
+
"at": "completion",
|
|
122
|
+
"checks": [
|
|
123
|
+
{
|
|
124
|
+
"kind": "minBytes",
|
|
125
|
+
"file": "ledger.json",
|
|
126
|
+
"bytes": 2
|
|
127
|
+
}
|
|
128
|
+
],
|
|
129
|
+
"scripts": [
|
|
130
|
+
{
|
|
131
|
+
"name": "checkJsonValid",
|
|
132
|
+
"scope": "standard",
|
|
133
|
+
"inputs": {
|
|
134
|
+
"file": "ledger.json"
|
|
135
|
+
}
|
|
136
|
+
}
|
|
137
|
+
],
|
|
138
|
+
"onReject": "categorize",
|
|
139
|
+
"maxAttempts": 3
|
|
140
|
+
},
|
|
141
|
+
"next": "review"
|
|
142
|
+
},
|
|
143
|
+
{
|
|
144
|
+
"id": "review",
|
|
145
|
+
"name": "Review the ledger",
|
|
146
|
+
"description": "closed taxonomy, reconciliation, review queue",
|
|
147
|
+
"prompt": "Review the categorized ledger against the rules scope. Step 1: confirm the file parses and every transaction has exactly one category drawn only from the closed taxonomy — no invented categories. Step 2: confirm rule precedence was applied — spot-check that explicit merchant matches beat heuristics. Step 3: confirm low-confidence rows are flagged needs_review rather than force-fit, and that the queue is non-empty if genuinely ambiguous rows exist. Step 4: confirm reconciliation — per-category totals sum to the grand total and the grand total equals the input sum (flag any penny drift). Step 5: spot-check 5 categorizations for correctness. Write PASS/FAIL per criterion to {{workPath}}/review.md and task notes; on failure, name the rows and loop back to categorize.\n\nThe deliverable `{{workPath}}/review.md` lands in the project's artifacts drawer — write it with `write_artifact` and read it back with `read_artifact`; the shipped workspace stays untouched.",
|
|
148
|
+
"suggestedRole": "reviewer",
|
|
149
|
+
"toolPolicy": {
|
|
150
|
+
"disallowBuiltinToolsets": [
|
|
151
|
+
"ai-apps",
|
|
152
|
+
"archives",
|
|
153
|
+
"audio",
|
|
154
|
+
"browser-automation",
|
|
155
|
+
"code-execution",
|
|
156
|
+
"craftbooks",
|
|
157
|
+
"data-tables",
|
|
158
|
+
"entity-intel",
|
|
159
|
+
"git",
|
|
160
|
+
"image-intel",
|
|
161
|
+
"images",
|
|
162
|
+
"role-delegation",
|
|
163
|
+
"role-delegation-escalation",
|
|
164
|
+
"security-intel",
|
|
165
|
+
"team-management",
|
|
166
|
+
"videos",
|
|
167
|
+
"web",
|
|
168
|
+
"workspace-fs-write"
|
|
169
|
+
],
|
|
170
|
+
"outputMedium": "artifact",
|
|
171
|
+
"additionalOutputMedia": [
|
|
172
|
+
"task-note"
|
|
173
|
+
]
|
|
174
|
+
},
|
|
175
|
+
"advanceWhen": {
|
|
176
|
+
"file": "{{workPath}}/review.md",
|
|
177
|
+
"minBytes": 1,
|
|
178
|
+
"sniff": "nonempty",
|
|
179
|
+
"artifact": true
|
|
180
|
+
},
|
|
181
|
+
"gate": {
|
|
182
|
+
"at": "completion",
|
|
183
|
+
"checks": [
|
|
184
|
+
{
|
|
185
|
+
"kind": "minBytes",
|
|
186
|
+
"file": "ledger.json",
|
|
187
|
+
"bytes": 2
|
|
188
|
+
}
|
|
189
|
+
],
|
|
190
|
+
"scripts": [
|
|
191
|
+
{
|
|
192
|
+
"name": "checkJsonValid",
|
|
193
|
+
"scope": "standard",
|
|
194
|
+
"inputs": {
|
|
195
|
+
"file": "ledger.json"
|
|
196
|
+
}
|
|
197
|
+
}
|
|
198
|
+
],
|
|
199
|
+
"onReject": "review",
|
|
200
|
+
"maxAttempts": 4
|
|
201
|
+
},
|
|
202
|
+
"next": "evaluate"
|
|
203
|
+
},
|
|
204
|
+
{
|
|
205
|
+
"id": "evaluate",
|
|
206
|
+
"name": "Evaluate",
|
|
207
|
+
"description": "Grade the deliverable against every acceptance criterion. All pass → finish; any fail → loop back and fix the gap.",
|
|
208
|
+
"prompt": "Open ledger.json and verify every criterion from {{workPath}}/rules-scope.md. Confirm it parses, every transaction has exactly one category from the closed taxonomy (none invented), rule precedence held (explicit merchant matches beat heuristics — spot-check), low-confidence rows are flagged for review rather than force-fit, and totals reconcile (per-category sums equal the grand total which equals the input sum). Write PASS/FAIL per criterion; on any failure, name the rows and loop back to categorize.\n\nThen route — this is the whole point of the loop:\n\n- **Every criterion PASSES →** call `advance_task_step({ ref, stepId: \"evaluate\", next: \"finish\" })`.\n- **Any criterion FAILS →** write the specific gaps to notes, then call `advance_task_step({ ref, stepId: \"evaluate\", next: \"categorize\" })` to loop back. The builder fixes exactly those gaps.\n\nNever route to `finish` while any criterion is unmet. The build phase's completion gate already blocked a grossly-incomplete deliverable; your job is the judgment an automated check cannot make (does it actually work, read well, look right). After ~3 unproductive loops, stop and report DONE_WITH_CONCERNS so the user can step in.",
|
|
209
|
+
"suggestedRole": "reviewer",
|
|
210
|
+
"toolPolicy": {
|
|
211
|
+
"disallowBuiltinToolsets": [
|
|
212
|
+
"ai-apps",
|
|
213
|
+
"archives",
|
|
214
|
+
"artifacts",
|
|
215
|
+
"audio",
|
|
216
|
+
"browser-automation",
|
|
217
|
+
"code-execution",
|
|
218
|
+
"craftbooks",
|
|
219
|
+
"data-tables",
|
|
220
|
+
"entity-intel",
|
|
221
|
+
"git",
|
|
222
|
+
"image-intel",
|
|
223
|
+
"images",
|
|
224
|
+
"role-delegation",
|
|
225
|
+
"role-delegation-escalation",
|
|
226
|
+
"security-intel",
|
|
227
|
+
"team-management",
|
|
228
|
+
"videos",
|
|
229
|
+
"web",
|
|
230
|
+
"workspace-fs-write"
|
|
231
|
+
],
|
|
232
|
+
"outputMedium": "task-note"
|
|
233
|
+
},
|
|
234
|
+
"consumes": [
|
|
235
|
+
{
|
|
236
|
+
"file": "ledger.json"
|
|
237
|
+
}
|
|
238
|
+
],
|
|
239
|
+
"next": "categorize"
|
|
240
|
+
},
|
|
241
|
+
{
|
|
242
|
+
"id": "finish",
|
|
243
|
+
"name": "Finish",
|
|
244
|
+
"description": "All acceptance criteria met. Stamp a short summary and report DONE.",
|
|
245
|
+
"prompt": "Every acceptance criterion passed. Write a one-paragraph DONE summary to task notes via `write_task_note`: what was built, the deliverable path(s), and a one-line confirmation that each criterion is met. Then report DONE.",
|
|
246
|
+
"suggestedRole": "developer",
|
|
247
|
+
"toolPolicy": {
|
|
248
|
+
"disallowBuiltinToolsets": [
|
|
249
|
+
"ai-apps",
|
|
250
|
+
"archives",
|
|
251
|
+
"artifacts",
|
|
252
|
+
"audio",
|
|
253
|
+
"browser-automation",
|
|
254
|
+
"code-execution",
|
|
255
|
+
"craftbooks",
|
|
256
|
+
"data-tables",
|
|
257
|
+
"entity-intel",
|
|
258
|
+
"git",
|
|
259
|
+
"image-intel",
|
|
260
|
+
"images",
|
|
261
|
+
"role-delegation",
|
|
262
|
+
"role-delegation-escalation",
|
|
263
|
+
"security-intel",
|
|
264
|
+
"team-management",
|
|
265
|
+
"videos",
|
|
266
|
+
"web",
|
|
267
|
+
"workspace-fs-write"
|
|
268
|
+
],
|
|
269
|
+
"outputMedium": "task-note"
|
|
270
|
+
},
|
|
271
|
+
"terminal": true
|
|
272
|
+
}
|
|
273
|
+
],
|
|
274
|
+
"version": "1.0.4",
|
|
275
|
+
"releasedAt": "2026-09-05T03:30:00Z",
|
|
276
|
+
"minGezelVersion": "1.26233"
|
|
277
|
+
}
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
{
|
|
2
|
+
"schemaVersion": 1,
|
|
3
|
+
"title": "Categorize Transactions smoke eval",
|
|
4
|
+
"objective": "Self-contained smoke eval for the Categorize Transactions craftbook using the data generic harness.",
|
|
5
|
+
"tags": [
|
|
6
|
+
"data"
|
|
7
|
+
],
|
|
8
|
+
"prompt": "I dropped our numbers in source/records.csv. Can you look at the data, analyze it, and put togehter some results in analysis.md?",
|
|
9
|
+
"setup": {
|
|
10
|
+
"projectName": "Categorize Transactions Eval",
|
|
11
|
+
"about": "Self-contained eval project for expense-categorize. Seeded inputs are under workspace/source or workspace/fixtures; final deliverable is workspace/analysis.md.",
|
|
12
|
+
"missionObjectives": "Use the Categorize Transactions craftbook/template, read the seeded local fixtures, and write analysis.md without network calls, real credentials, or live services.",
|
|
13
|
+
"files": [
|
|
14
|
+
{
|
|
15
|
+
"path": "source/brief.md",
|
|
16
|
+
"content": "# Categorize Transactions Eval Brief\n\nClient: Boreal Desk, a home-office accessories company.\nAudience: operations leads who need an artifact they can use this week.\n\nFixed source facts for grounding:\n- The returns desk pilot covered 18 SKUs.\n- Median first response improved from 18 hours to 6 hours.\n- Preventable refund leakage fell from 14.2% to 8.9%.\n- The top unresolved complaint is status silence after photo submission.\n- Required next actions are automated status emails, barcode-exception training, and a weekly Finance exception export.\n\nUse these facts when the task asks for prose, analysis, copy, UI content, or test data. Do not use live web services, real credentials, or current outside data.\n\nCraftbook under test: expense-categorize - Categorize Transactions.\n"
|
|
17
|
+
},
|
|
18
|
+
{
|
|
19
|
+
"path": "source/records.csv",
|
|
20
|
+
"content": "customer,region,orders,revenue_usd,churn_risk\nAster,North,12,1440,low\nBoreal,South,8,720,medium\nCedar,West,15,2100,high\nDune,North,5,500,low\n"
|
|
21
|
+
}
|
|
22
|
+
],
|
|
23
|
+
"worker": {
|
|
24
|
+
"name": "Jules",
|
|
25
|
+
"role": "Data Analyst"
|
|
26
|
+
}
|
|
27
|
+
},
|
|
28
|
+
"mocks": [],
|
|
29
|
+
"success": {
|
|
30
|
+
"summary": "analysis.md is a seeded-data analysis with deterministic computed facts.",
|
|
31
|
+
"deliverables": [
|
|
32
|
+
{
|
|
33
|
+
"path": "analysis.md",
|
|
34
|
+
"kind": "markdown-report",
|
|
35
|
+
"minBytes": 900,
|
|
36
|
+
"checks": [
|
|
37
|
+
{
|
|
38
|
+
"kind": "contains",
|
|
39
|
+
"file": "analysis.md",
|
|
40
|
+
"pattern": "(?:^|\\n)#{1,3}\\s+\\S",
|
|
41
|
+
"flags": "i"
|
|
42
|
+
},
|
|
43
|
+
{
|
|
44
|
+
"kind": "contains",
|
|
45
|
+
"file": "analysis.md",
|
|
46
|
+
"pattern": "40\\b",
|
|
47
|
+
"flags": "i",
|
|
48
|
+
"label": "include total orders of 40"
|
|
49
|
+
},
|
|
50
|
+
{
|
|
51
|
+
"kind": "contains",
|
|
52
|
+
"file": "analysis.md",
|
|
53
|
+
"pattern": "\\$?4,?760\\b",
|
|
54
|
+
"flags": "i",
|
|
55
|
+
"label": "include total revenue of 4760"
|
|
56
|
+
},
|
|
57
|
+
{
|
|
58
|
+
"kind": "contains",
|
|
59
|
+
"file": "analysis.md",
|
|
60
|
+
"pattern": "Cedar",
|
|
61
|
+
"flags": "i",
|
|
62
|
+
"label": "identify Cedar as top revenue and high risk"
|
|
63
|
+
},
|
|
64
|
+
{
|
|
65
|
+
"kind": "contains",
|
|
66
|
+
"file": "analysis.md",
|
|
67
|
+
"pattern": "high",
|
|
68
|
+
"flags": "i",
|
|
69
|
+
"label": "include churn-risk finding"
|
|
70
|
+
}
|
|
71
|
+
]
|
|
72
|
+
}
|
|
73
|
+
]
|
|
74
|
+
},
|
|
75
|
+
"rubric": {
|
|
76
|
+
"artifact": {
|
|
77
|
+
"path": "analysis.md",
|
|
78
|
+
"kind": "markdown"
|
|
79
|
+
},
|
|
80
|
+
"axes": [
|
|
81
|
+
{
|
|
82
|
+
"name": "accuracy",
|
|
83
|
+
"description": "Computed figures are correct against the seeded records."
|
|
84
|
+
},
|
|
85
|
+
{
|
|
86
|
+
"name": "grounding",
|
|
87
|
+
"description": "Every claim traces to the seeded data — nothing invented."
|
|
88
|
+
},
|
|
89
|
+
{
|
|
90
|
+
"name": "structure",
|
|
91
|
+
"description": "Tables and sections make the analysis checkable at a glance."
|
|
92
|
+
},
|
|
93
|
+
{
|
|
94
|
+
"name": "actionability",
|
|
95
|
+
"description": "Recommendations follow from the findings and are concretely doable."
|
|
96
|
+
}
|
|
97
|
+
]
|
|
98
|
+
},
|
|
99
|
+
"qualityFocus": [
|
|
100
|
+
"seeded data reading",
|
|
101
|
+
"computed facts",
|
|
102
|
+
"recommendation grounding"
|
|
103
|
+
]
|
|
104
|
+
}
|