@bendyline/gilde 0.1.53 → 0.1.55

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (34) hide show
  1. package/authoring/chat-models/qwen3.5-27b-q4.json +147 -0
  2. package/authoring/gstack/evals/cso.json +21 -18
  3. package/authoring/gstack/evals/investigate.json +25 -22
  4. package/authoring/gstack/evals/plan-ceo-review.json +19 -16
  5. package/authoring/gstack/overlays/retro.json +2 -2
  6. package/authoring/gstack/wave.json +2 -2
  7. package/data/chat-models/index.json +1 -1
  8. package/data/chat-models/qw/qwen3.5-27b-q4/manifest.json +217 -0
  9. package/data/chat-models/qw/qwen3.5-27b-q4/versions/1.0.0/manifest.json +90 -0
  10. package/data/craftbook-templates/br/browser-qa-audit/versions/2.0.7/craftbook.json +620 -0
  11. package/data/craftbook-templates/br/browser-qa-audit/versions/2.0.7/test.json +376 -0
  12. package/data/craftbook-templates/de/design-system-consultation/versions/2.0.7/craftbook.json +616 -0
  13. package/data/craftbook-templates/de/design-system-consultation/versions/2.0.7/test.json +201 -0
  14. package/data/craftbook-templates/en/engineering-retrospective/versions/2.0.7/craftbook.json +566 -0
  15. package/data/craftbook-templates/en/engineering-retrospective/versions/2.0.7/test.json +191 -0
  16. package/data/craftbook-templates/ex/executive-level-review/versions/2.0.7/craftbook.json +595 -0
  17. package/data/craftbook-templates/ex/executive-level-review/versions/2.0.7/test.json +139 -0
  18. package/data/craftbook-templates/id/idea-office-hours/versions/2.0.7/craftbook.json +566 -0
  19. package/data/craftbook-templates/id/idea-office-hours/versions/2.0.7/test.json +141 -0
  20. package/data/craftbook-templates/index.json +1 -1
  21. package/data/craftbook-templates/pl/plan/versions/1.0.2/craftbook.json +222 -0
  22. package/data/craftbook-templates/pl/plan/versions/1.0.2/test.json +91 -0
  23. package/data/craftbook-templates/ro/root-cause-investigation/versions/2.0.7/craftbook.json +595 -0
  24. package/data/craftbook-templates/ro/root-cause-investigation/versions/2.0.7/test.json +157 -0
  25. package/data/craftbook-templates/se/security-architecture-review/versions/2.0.7/craftbook.json +597 -0
  26. package/data/craftbook-templates/se/security-architecture-review/versions/2.0.7/test.json +157 -0
  27. package/data/craftbook-templates/sh/ship/versions/1.1.2/craftbook.json +346 -0
  28. package/data/craftbook-templates/sh/ship/versions/1.1.2/test.json +228 -0
  29. package/data/craftbook-templates/sp/spec-authoring/versions/2.0.7/craftbook.json +599 -0
  30. package/data/craftbook-templates/sp/spec-authoring/versions/2.0.7/test.json +162 -0
  31. package/data/craftbook-templates/te/technical-documentation/versions/2.0.7/craftbook.json +577 -0
  32. package/data/craftbook-templates/te/technical-documentation/versions/2.0.7/test.json +174 -0
  33. package/package.json +1 -1
  34. package/schemas/craftbook-test.schema.json +7 -1
@@ -0,0 +1,139 @@
1
+ {
2
+ "schemaVersion": 1,
3
+ "title": "Executive-Level Review — analytics expansion decision",
4
+ "objective": "Test whether the Executive-Level Review craftbook challenges a proposal, compares lower-cost alternatives, and produces an evidence-anchored mode verdict with decision conditions.",
5
+ "tags": [
6
+ "workflow",
7
+ "strategy",
8
+ "executive-review",
9
+ "decision"
10
+ ],
11
+ "prompt": "Use the Executive-Level Review craftbook to review `source/analytics-proposal.md`. Challenge whether the proposed scope is the best way to reach the outcome; compare at least three alternatives including the two-day manual export and waiting. Produce all craftbook artifacts and a decision-ready `tasks/eval/reviews/executive-level-review.md`. Use exactly one allowed mode verdict, score every required dimension from 0–10 with a factual basis, separate unknowns from evidence, and state explicit approve, pause, and stop conditions. Do not invent customer or financial evidence.",
12
+ "setup": {
13
+ "projectName": "Analytics expansion review",
14
+ "about": "A hermetic executive decision review. The seeded proposal is the complete evidence set.",
15
+ "missionObjectives": "Decide what, if anything, should be approved now while preserving the fastest credible route to learning.",
16
+ "files": [
17
+ {
18
+ "path": "source/analytics-proposal.md",
19
+ "content": "# Embedded analytics proposal\n\n## Requested decision\nApprove an eight-week build of customer-facing dashboards with scheduled PDF delivery and CSV export.\n\n## Evidence\nTwelve of 86 active customers requested better reporting in support conversations. Three customers agreed to a design pilot, but none signed a paid commitment. Support currently creates weekly exports for five customers.\n\n## Cost and constraints\n- Estimate: two engineers for 8 weeks.\n- New charting service: $4,000 per month at current volume.\n- Security review is complete; accessibility and data-retention reviews are not.\n- The enterprise renewal decision is in six weeks.\n- A manual scheduled-export workflow can be built in two engineer-days and would test delivery cadence, but not dashboard engagement.\n\n## Unknowns\nWillingness to pay, which metrics matter, PDF accessibility, and whether scheduled delivery or interactive dashboards drives retention.\n"
20
+ }
21
+ ],
22
+ "craftbookParams": {
23
+ "workPath": "tasks/eval"
24
+ }
25
+ },
26
+ "mocks": [],
27
+ "success": {
28
+ "summary": "The named craftbook produces a grounded executive verdict, alternatives, scorecard, and terminal task record.",
29
+ "deliverables": [
30
+ {
31
+ "path": "tasks/eval/reviews/executive-level-review.md",
32
+ "kind": "markdown-report",
33
+ "artifact": true,
34
+ "minBytes": 1400,
35
+ "checks": [
36
+ {
37
+ "kind": "contains",
38
+ "file": "tasks/eval/reviews/executive-level-review.md",
39
+ "pattern": "^#{1,3}\\s+Executive verdict\\b[\\s\\S]*(SCOPE EXPANSION|SELECTIVE|HOLD|SCOPE REDUCTION)[\\s\\S]*^#{1,3}\\s+Scorecard\\b[\\s\\S]*^#{1,3}\\s+Alternatives\\b[\\s\\S]*^#{1,3}\\s+Recommendation\\b[\\s\\S]*^#{1,3}\\s+Decision conditions\\b",
40
+ "flags": "im",
41
+ "label": "verdict, scorecard, alternatives, recommendation, and conditions"
42
+ },
43
+ {
44
+ "kind": "contains",
45
+ "file": "tasks/eval/reviews/executive-level-review.md",
46
+ "pattern": "(user value).*?(strategic fit).*?(evidence strength).*?(feasibility).*?(sequencing).*?(reversibility).*?(operating cost).*?(downside)",
47
+ "flags": "is",
48
+ "label": "complete executive scorecard"
49
+ },
50
+ {
51
+ "kind": "valueGrounding",
52
+ "file": "tasks/eval/reviews/executive-level-review.md",
53
+ "facts": [
54
+ {
55
+ "id": "customer-requests",
56
+ "required": [
57
+ "12\\s+(?:of\\s+86\\s+)?(?:active\\s+)?customers"
58
+ ]
59
+ },
60
+ {
61
+ "id": "pilot-evidence",
62
+ "required": [
63
+ "3\\s+customers?[^.]{0,60}(pilot|agreed)"
64
+ ]
65
+ },
66
+ {
67
+ "id": "build-duration",
68
+ "required": [
69
+ "8\\s+weeks?"
70
+ ]
71
+ },
72
+ {
73
+ "id": "operating-cost",
74
+ "required": [
75
+ "\\$4,?000\\s+per\\s+month"
76
+ ]
77
+ },
78
+ {
79
+ "id": "cheap-alternative",
80
+ "required": [
81
+ "(two|2)[ -]?(engineer[- ])?days?[\\s\\S]{0,120}(manual|scheduled)[ -]export"
82
+ ]
83
+ }
84
+ ]
85
+ },
86
+ {
87
+ "kind": "citationsResolve",
88
+ "file": "tasks/eval/reviews/executive-level-review.md",
89
+ "minCitations": 1
90
+ }
91
+ ]
92
+ }
93
+ ],
94
+ "taskNotes": {
95
+ "minBytes": 160,
96
+ "checks": [
97
+ {
98
+ "kind": "contains",
99
+ "file": "task-notes.md",
100
+ "pattern": "\\bDONE\\b[\\s\\S]*tasks/eval/reviews/executive-level-review\\.md",
101
+ "label": "terminal craftbook note names the review"
102
+ }
103
+ ],
104
+ "requireCraftbookTask": true
105
+ },
106
+ "taskGraph": {
107
+ "requireCraftbookTask": true,
108
+ "requireTerminalStep": true
109
+ },
110
+ "unchangedFixtures": [
111
+ "source/analytics-proposal.md"
112
+ ]
113
+ },
114
+ "rubric": {
115
+ "artifact": {
116
+ "path": "tasks/eval/reviews/executive-level-review.md",
117
+ "kind": "markdown"
118
+ },
119
+ "axes": [
120
+ {
121
+ "name": "Premise challenge",
122
+ "description": "The review tests whether action is warranted now and compares materially different ways to learn or deliver value."
123
+ },
124
+ {
125
+ "name": "Evidence discipline",
126
+ "description": "Scores and conclusions trace to supplied facts, with unsupported assumptions and missing decisions called out."
127
+ },
128
+ {
129
+ "name": "Executive usefulness",
130
+ "description": "The verdict is decisive, sequenced, reversible where possible, and paired with clear decision-changing conditions."
131
+ }
132
+ ]
133
+ },
134
+ "qualityFocus": [
135
+ "One allowed mode verdict is selected",
136
+ "All scorecard dimensions have evidence-backed scores",
137
+ "The two-day alternative and missing willingness-to-pay evidence influence the decision"
138
+ ]
139
+ }