@bendyline/gilde 0.1.4 → 0.1.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -0
- package/data/chat-models/de/deepseek-v4-flash-284b-q2/manifest.json +4 -3
- package/data/chat-models/de/deepseek-v4-flash-284b-q4/manifest.json +4 -3
- package/data/chat-models/ge/gemma4-12b-q4/manifest.json +20 -11
- package/data/chat-models/ge/gemma4-12b-q4/versions/1.1.0/manifest.json +2 -8
- package/data/chat-models/ge/gemma4-12b-q4/versions/1.1.2/manifest.json +78 -0
- package/data/chat-models/ge/gemma4-12b-q8/manifest.json +23 -14
- package/data/chat-models/ge/gemma4-12b-q8/versions/1.0.0/manifest.json +2 -8
- package/data/chat-models/ge/gemma4-12b-q8/versions/1.0.2/manifest.json +78 -0
- package/data/chat-models/ge/gemma4-26b-q4/manifest.json +55 -46
- package/data/chat-models/ge/gemma4-26b-q4/versions/1.2.0/manifest.json +2 -8
- package/data/chat-models/ge/gemma4-26b-q4/versions/1.2.1/manifest.json +81 -0
- package/data/chat-models/ge/gemma4-31b-q4/manifest.json +36 -28
- package/data/chat-models/ge/gemma4-31b-q4/versions/1.2.0/manifest.json +2 -8
- package/data/chat-models/ge/gemma4-31b-q4/versions/1.2.1/manifest.json +87 -0
- package/data/chat-models/ge/gemma4-e2b-q8/manifest.json +76 -69
- package/data/chat-models/ge/gemma4-e2b-q8/versions/1.1.0/manifest.json +2 -8
- package/data/chat-models/ge/gemma4-e2b-q8/versions/1.1.2/manifest.json +71 -0
- package/data/chat-models/ge/gemma4-e4b-q8/manifest.json +71 -78
- package/data/chat-models/ge/gemma4-e4b-q8/versions/1.1.0/manifest.json +3 -9
- package/data/chat-models/ge/gemma4-e4b-q8/versions/1.1.2/manifest.json +71 -0
- package/data/chat-models/index.json +1 -1
- package/data/chat-models/la/laguna-s-2.1-118b-q4/manifest.json +17 -17
- package/data/chat-models/la/laguna-s-2.1-118b-q4/versions/1.0.1/manifest.json +124 -0
- package/data/chat-models/la/laguna-s-2.1-118b-q8/manifest.json +7 -7
- package/data/chat-models/la/laguna-s-2.1-118b-q8/versions/1.0.1/manifest.json +174 -0
- package/data/chat-models/mi/mistral-medium-3.5-128b-q4/manifest.json +4 -17
- package/data/chat-models/mi/mistral-medium-3.5-128b-q4/versions/1.0.0/manifest.json +2 -12
- package/data/chat-models/ne/nemotron3-nano-30b-q4/manifest.json +5 -0
- package/data/chat-models/qw/qwen3.5-122b-a10b-q4/manifest.json +27 -19
- package/data/chat-models/qw/qwen3.5-122b-a10b-q4/versions/1.0.1/manifest.json +164 -0
- package/data/chat-models/qw/qwen3.5-2b-q4/manifest.json +79 -77
- package/data/chat-models/qw/qwen3.5-2b-q4/versions/1.1.0/manifest.json +2 -8
- package/data/chat-models/qw/qwen3.5-2b-q4/versions/1.1.2/manifest.json +76 -0
- package/data/chat-models/qw/qwen3.5-4b-q4/manifest.json +79 -77
- package/data/chat-models/qw/qwen3.5-4b-q4/versions/1.1.0/manifest.json +2 -8
- package/data/chat-models/qw/qwen3.5-4b-q4/versions/1.1.2/manifest.json +76 -0
- package/data/chat-models/qw/qwen3.5-9b-q4/manifest.json +87 -85
- package/data/chat-models/qw/qwen3.5-9b-q4/versions/1.1.0/manifest.json +2 -8
- package/data/chat-models/qw/qwen3.5-9b-q4/versions/1.1.2/manifest.json +81 -0
- package/data/chat-models/qw/qwen3.6-27b-q4/manifest.json +99 -97
- package/data/chat-models/qw/qwen3.6-27b-q4/versions/1.1.0/manifest.json +2 -8
- package/data/chat-models/qw/qwen3.6-27b-q4/versions/1.1.4/manifest.json +91 -0
- package/data/chat-models/qw/qwen3.6-27b-q8/manifest.json +16 -12
- package/data/chat-models/qw/qwen3.6-27b-q8/versions/1.0.0/manifest.json +2 -8
- package/data/chat-models/qw/qwen3.6-27b-q8/versions/1.0.2/manifest.json +103 -0
- package/data/chat-models/qw/qwen3.6-35b-a3b-q4/manifest.json +18 -14
- package/data/chat-models/qw/qwen3.6-35b-a3b-q4/versions/1.0.0/manifest.json +2 -8
- package/data/chat-models/qw/qwen3.6-35b-a3b-q4/versions/1.0.1/manifest.json +93 -0
- package/data/chat-models/qw/qwen3.6-35b-a3b-q8/manifest.json +16 -12
- package/data/chat-models/qw/qwen3.6-35b-a3b-q8/versions/1.0.0/manifest.json +2 -8
- package/data/chat-models/qw/qwen3.6-35b-a3b-q8/versions/1.0.1/manifest.json +113 -0
- package/data/craftbook-templates/bu/bug-fix-tdd/versions/1.0.1/test.json +0 -5
- package/data/craftbook-templates/bu/build-loop/versions/1.1.0/craftbook.json +65 -0
- package/data/craftbook-templates/bu/build-loop/versions/1.1.0/test.json +121 -0
- package/data/craftbook-templates/ch/character-sheet/versions/1.0.0/test.json +1 -2
- package/data/craftbook-templates/ch/character-turnaround/versions/1.0.0/test.json +1 -2
- package/data/craftbook-templates/cl/cli-tool/versions/1.1.0/craftbook.json +139 -0
- package/data/craftbook-templates/cl/cli-tool/versions/1.1.0/test.json +128 -0
- package/data/craftbook-templates/cr/crossword-forge/manifest.json +1 -2
- package/data/craftbook-templates/de/deep-security-review/versions/1.1.0/craftbook.json +201 -0
- package/data/craftbook-templates/de/deep-security-review/versions/1.1.0/test.json +157 -0
- package/data/craftbook-templates/do/dockerize-app/versions/1.0.0/test.json +0 -5
- package/data/craftbook-templates/fr/freeze-scope/manifest.json +1 -2
- package/data/craftbook-templates/fr/freeze-scope/versions/{1.0.1 → 1.1.0}/craftbook.json +5 -17
- package/data/craftbook-templates/gr/graphql-api/versions/1.0.1/test.json +0 -5
- package/data/craftbook-templates/gr/grpc-service/versions/1.0.0/test.json +0 -5
- package/data/craftbook-templates/ho/hotfix-flow/versions/1.0.0/test.json +0 -5
- package/data/craftbook-templates/in/investigate/versions/1.1.0/craftbook.json +81 -0
- package/data/craftbook-templates/in/investigate/versions/1.1.0/test.json +122 -0
- package/data/craftbook-templates/in/invoice-run/manifest.json +1 -2
- package/data/craftbook-templates/in/invoice-run/versions/{1.0.1 → 1.1.0}/craftbook.json +4 -4
- package/data/craftbook-templates/index.json +1 -1
- package/data/craftbook-templates/li/library-package/versions/1.0.0/test.json +0 -5
- package/data/craftbook-templates/me/memory-prompt-session/manifest.json +1 -2
- package/data/craftbook-templates/me/message-queue-consumer/versions/1.0.0/test.json +0 -5
- package/data/craftbook-templates/of/office-hours/versions/1.1.0/craftbook.json +48 -0
- package/data/craftbook-templates/of/office-hours/versions/1.1.0/test.json +118 -0
- package/data/craftbook-templates/pa/page-spread/manifest.json +1 -2
- package/data/craftbook-templates/pa/parser-grammar/versions/1.0.0/test.json +0 -5
- package/data/craftbook-templates/pe/perf-optimization/versions/1.0.0/test.json +0 -5
- package/data/craftbook-templates/pl/plan/manifest.json +1 -2
- package/data/craftbook-templates/po/powerpoint-deck/versions/1.1.0/craftbook.json +128 -0
- package/data/craftbook-templates/{ro/root-cause-investigation/versions/1.0.1 → po/powerpoint-deck/versions/1.1.0}/test.json +6 -6
- package/data/craftbook-templates/pu/pull-request-review/versions/1.1.0/craftbook.json +118 -0
- package/data/craftbook-templates/pu/pull-request-review/versions/1.1.0/test.json +160 -0
- package/data/craftbook-templates/re/refactor-module/versions/1.0.1/test.json +0 -5
- package/data/craftbook-templates/re/regex-builder/versions/1.0.0/test.json +0 -5
- package/data/craftbook-templates/re/research-to-document/versions/1.1.0/craftbook.json +158 -0
- package/data/craftbook-templates/re/research-to-document/versions/1.1.0/test.json +97 -0
- package/data/craftbook-templates/ro/root-cause-investigation/manifest.json +1 -2
- package/data/craftbook-templates/sd/sdk-wrapper/versions/1.0.1/test.json +0 -5
- package/data/craftbook-templates/se/security-architecture-review/versions/1.1.0/craftbook.json +28 -0
- package/data/craftbook-templates/{te/technical-documentation/versions/1.0.1 → se/security-architecture-review/versions/1.1.0}/test.json +7 -7
- package/data/craftbook-templates/sh/ship/versions/1.1.0/craftbook.json +137 -0
- package/data/craftbook-templates/sh/ship/versions/1.1.0/test.json +225 -0
- package/data/craftbook-templates/st/state-machine/versions/1.0.0/test.json +0 -5
- package/data/craftbook-templates/te/technical-documentation/manifest.json +1 -2
- package/data/craftbook-templates/te/test-suite-backfill/versions/1.0.0/test.json +0 -5
- package/data/craftbook-templates/ti/tileset-batch/versions/1.0.0/test.json +1 -2
- package/data/craftbook-templates/ty/type-safety-pass/versions/1.0.0/test.json +0 -5
- package/data/craftbook-templates/ve/version-bump/versions/1.0.0/test.json +0 -5
- package/data/gezel-templates/ch/chess-player/manifest.json +20 -0
- package/data/gezel-templates/ch/chess-player/versions/1.0.0/about.md +21 -0
- package/data/gezel-templates/ch/chess-player/versions/1.0.0/manifest.json +10 -0
- package/data/gezel-templates/go/go-player/manifest.json +22 -0
- package/data/gezel-templates/go/go-player/versions/1.0.0/about.md +22 -0
- package/data/gezel-templates/go/go-player/versions/1.0.0/manifest.json +10 -0
- package/data/gezel-templates/index.json +1 -1
- package/data/project-types/ch/chess/manifest.json +20 -0
- package/data/project-types/ch/chess/versions/1.0.0/about.md +9 -0
- package/data/project-types/ch/chess/versions/1.0.0/game.json +127 -0
- package/data/project-types/ch/chess/versions/1.0.0/manifest.json +180 -0
- package/data/project-types/ch/chess/versions/1.0.0/mission.md +8 -0
- package/data/project-types/ch/chess/versions/1.0.0/pages/board/index.html +574 -0
- package/data/project-types/go/go/manifest.json +22 -0
- package/data/project-types/go/go/versions/1.0.0/about.md +9 -0
- package/data/project-types/go/go/versions/1.0.0/game.json +103 -0
- package/data/project-types/go/go/versions/1.0.0/manifest.json +166 -0
- package/data/project-types/go/go/versions/1.0.0/mission.md +8 -0
- package/data/project-types/go/go/versions/1.0.0/pages/board/index.html +637 -0
- package/data/project-types/index.json +1 -1
- package/package.json +2 -1
- package/schemas/chat-model-version.schema.json +22 -0
- package/schemas/craftbook-doc.schema.json +12 -0
- package/schemas/craftbook-template-version.schema.json +12 -0
- package/schemas/craftbook-test.schema.json +12 -0
- package/data/chat-models/gl/glm-5.2-754b-q2/manifest.json +0 -66
- package/data/chat-models/gl/glm-5.2-754b-q2/versions/1.0.0/manifest.json +0 -18
- package/data/craftbook-templates/cr/crossword-forge/versions/1.0.1/craftbook.json +0 -147
- package/data/craftbook-templates/cr/crossword-forge/versions/1.0.1/test.json +0 -135
- package/data/craftbook-templates/me/memory-prompt-session/versions/1.0.1/craftbook.json +0 -116
- package/data/craftbook-templates/me/memory-prompt-session/versions/1.0.1/test.json +0 -112
- package/data/craftbook-templates/pa/page-spread/versions/1.0.1/craftbook.json +0 -94
- package/data/craftbook-templates/pa/page-spread/versions/1.0.1/test.json +0 -133
- package/data/craftbook-templates/pl/plan/versions/1.0.1/craftbook.json +0 -104
- package/data/craftbook-templates/pl/plan/versions/1.0.1/test.json +0 -91
- package/data/craftbook-templates/ro/root-cause-investigation/versions/1.0.1/craftbook.json +0 -126
- package/data/craftbook-templates/te/technical-documentation/versions/1.0.1/craftbook.json +0 -152
- /package/data/craftbook-templates/fr/freeze-scope/versions/{1.0.1 → 1.1.0}/test.json +0 -0
- /package/data/craftbook-templates/in/invoice-run/versions/{1.0.1 → 1.1.0}/test.json +0 -0
|
@@ -0,0 +1,201 @@
|
|
|
1
|
+
{
|
|
2
|
+
"id": "deep-security-review",
|
|
3
|
+
"name": "Deep Security Review",
|
|
4
|
+
"description": "Run a slow, thorough, WHOLE-CODEBASE security audit that looks for SYSTEMIC weaknesses — a vulnerability class repeated across many routes, a missing control at an architectural layer, inconsistent authorization, ad-hoc secret handling, a trust-boundary violation — not just isolated issues in one file. First reconnoiters the whole system using gezel's static security index (dependency inventory, attack surface, source/sink findings, taint reachability), then builds a threat model to find the highest-blast-radius areas, then audits those areas holistically confirming real source-to-sink paths and consistency of controls across every entry point, then clusters findings into root-cause systemic themes, then writes a prioritized report with a machine-readable findings file. Use this for a deep or systemic security review, a codebase security audit, a nightly security sweep, or finding security issues across a whole project — it is intentionally broader and deeper than a PR-time diff review (see pr-security-review for that).\n\nA gallery craftbook generated from an archetype spec. It runs\n`phase → (per-phase gate) → … → evaluate → (loop) → finish`. Each build\nphase that produces a checkable artifact is followed by a **runtime\ngate-checkpoint** — the runtime verifies the artifact and routes with no\nmodel turn, looping back to redo the phase on a miss. The final `evaluate`\nstep holds a static deliverable gate plus a reviewer QA pass. What it adds\nover the generic `build-loop`: a specialist role per phase, a\ndomain-correct ordering, and a concrete per-phase quality bar.\n\nPhases:\n\n1. Reconnoiter the whole system (reviewer) — build a whole-system security map from the static index → gated on `notes/security/surface-map.md` (markdown-notes)\n2. Threat-model the system (reviewer) — assets, actors, STRIDE per boundary, highest blast radius → gated on `notes/security/threat-model.md` (markdown-notes)\n3. Audit for systemic weaknesses (reviewer) — confirm real source→sink paths and control consistency across all entry points → gated on `notes/security/findings.md` (markdown-notes)\n4. Cluster findings into systemic themes (reviewer) — group findings by root cause; rank by blast radius → gated on `notes/security/themes.md` (markdown-notes)\n5. Write the security report (reviewer) — systemic report + machine-readable findings, all citing real files → gated on `security-review/REPORT.md` (security-report)\n\nThe gates never advance with an unmet criterion, and loop back to the\nowning phase to fix named gaps.\n",
|
|
5
|
+
"entryStepId": "recon",
|
|
6
|
+
"triggers": [
|
|
7
|
+
"deep security review",
|
|
8
|
+
"audit the codebase for vulnerabilities",
|
|
9
|
+
"systemic security review",
|
|
10
|
+
"find security issues across the whole codebase",
|
|
11
|
+
"whole-codebase security audit",
|
|
12
|
+
"nightly security audit"
|
|
13
|
+
],
|
|
14
|
+
"toolsets": [
|
|
15
|
+
{
|
|
16
|
+
"toolsetId": "builtin.security-intel",
|
|
17
|
+
"autoAllow": true,
|
|
18
|
+
"reason": "static security analysis over the index — scan, findings, attack surface, dependencies, taint reachability"
|
|
19
|
+
},
|
|
20
|
+
{
|
|
21
|
+
"toolsetId": "builtin.code-intel",
|
|
22
|
+
"autoAllow": true,
|
|
23
|
+
"reason": "navigate and read the code to confirm each lead"
|
|
24
|
+
}
|
|
25
|
+
],
|
|
26
|
+
"steps": [
|
|
27
|
+
{
|
|
28
|
+
"id": "recon",
|
|
29
|
+
"name": "Reconnoiter the whole system",
|
|
30
|
+
"description": "build a whole-system security map from the static index",
|
|
31
|
+
"prompt": "You are starting a deep, whole-codebase security review — take your time and look across the ENTIRE system, not one file. Step 1: Call `security_scan` to build the dependency inventory and run any installed OSS scanners (semgrep / osv-scanner / gitleaks). Step 2: Call `map_repo`, `map_attack_surface`, `security_overview`, and `list_dependencies` to assemble a whole-system picture: languages and top areas; entry points; route/handler files; auth & middleware boundaries; secret touchpoints (config/env/credential files); files that read untrusted input (taint sources); external integrations and data stores; and the dependency inventory with any advisories. Step 3: Identify the trust boundaries — where untrusted data crosses into the system and where privileged/state-changing operations live. Step 4: Record the CANDIDATE SYSTEMIC THEMES that `security_overview` surfaced (categories recurring across many files) as leads to investigate. Write a whole-system surface map — assets, entry points, trust boundaries, the auth model, data stores, external dependencies, and the candidate themes — to `notes/security/surface-map.md`. No findings yet; this phase orients the review across the whole codebase.",
|
|
32
|
+
"suggestedRole": "reviewer",
|
|
33
|
+
"advanceWhen": {
|
|
34
|
+
"file": "notes/security/surface-map.md",
|
|
35
|
+
"minBytes": 1,
|
|
36
|
+
"sniff": "nonempty"
|
|
37
|
+
},
|
|
38
|
+
"gate": {
|
|
39
|
+
"at": "completion",
|
|
40
|
+
"checks": [
|
|
41
|
+
{
|
|
42
|
+
"kind": "minBytes",
|
|
43
|
+
"file": "notes/security/surface-map.md",
|
|
44
|
+
"bytes": 120
|
|
45
|
+
},
|
|
46
|
+
{
|
|
47
|
+
"kind": "sniff",
|
|
48
|
+
"file": "notes/security/surface-map.md",
|
|
49
|
+
"sniff": "nonempty"
|
|
50
|
+
}
|
|
51
|
+
],
|
|
52
|
+
"onReject": "recon",
|
|
53
|
+
"maxAttempts": 3
|
|
54
|
+
},
|
|
55
|
+
"next": "threatmodel"
|
|
56
|
+
},
|
|
57
|
+
{
|
|
58
|
+
"id": "threatmodel",
|
|
59
|
+
"name": "Threat-model the system",
|
|
60
|
+
"description": "assets, actors, STRIDE per boundary, highest blast radius",
|
|
61
|
+
"prompt": "Using the surface map, build the system's threat model to decide WHERE systemic weaknesses would hurt most. Step 1: List the assets worth protecting (user data, secrets, money/state-changing operations, admin capability) and the actors/threats (unauthenticated user, authenticated user, insider, compromised dependency). Step 2: For each trust boundary, walk STRIDE (spoofing, tampering, repudiation, information-disclosure, denial-of-service, elevation-of-privilege) and note where a weakness would have the highest BLAST RADIUS across the system. Step 3: Prioritize — which boundaries or assets, if weak, would compromise many parts of the app at once? Those are where systemic issues hide and where the audit should spend its time. Write the threat model — assets, actors, trust zones, STRIDE-per-boundary, and the ranked high-blast-radius areas to focus on — to `notes/security/threat-model.md`.",
|
|
62
|
+
"suggestedRole": "reviewer",
|
|
63
|
+
"advanceWhen": {
|
|
64
|
+
"file": "notes/security/threat-model.md",
|
|
65
|
+
"minBytes": 1,
|
|
66
|
+
"sniff": "nonempty"
|
|
67
|
+
},
|
|
68
|
+
"gate": {
|
|
69
|
+
"at": "completion",
|
|
70
|
+
"checks": [
|
|
71
|
+
{
|
|
72
|
+
"kind": "minBytes",
|
|
73
|
+
"file": "notes/security/threat-model.md",
|
|
74
|
+
"bytes": 120
|
|
75
|
+
},
|
|
76
|
+
{
|
|
77
|
+
"kind": "sniff",
|
|
78
|
+
"file": "notes/security/threat-model.md",
|
|
79
|
+
"sniff": "nonempty"
|
|
80
|
+
}
|
|
81
|
+
],
|
|
82
|
+
"onReject": "threatmodel",
|
|
83
|
+
"maxAttempts": 3
|
|
84
|
+
},
|
|
85
|
+
"next": "audit"
|
|
86
|
+
},
|
|
87
|
+
{
|
|
88
|
+
"id": "audit",
|
|
89
|
+
"name": "Audit for systemic weaknesses",
|
|
90
|
+
"description": "confirm real source→sink paths and control consistency across all entry points",
|
|
91
|
+
"prompt": "Investigate the high-blast-radius areas for CONCRETE, SYSTEMIC weaknesses — not one-off style nits. Step 1: Use `scan_findings` (filter by severity/category) to pull the static leads, then VERIFY each in the code with `read_symbol` / `read_file` — a regex lead is not a finding until you confirm the vulnerable path. Step 2: For untrusted inputs, use `trace_taint` on the relevant files to see which sinks are reachable, then confirm the actual source→sink path by reading it; flag any path that reaches a dangerous sink without validation, escaping, or parameterization. Step 3: Assess the auth model HOLISTICALLY — is every route behind an auth boundary? Is authorization applied consistently across ALL entry points, or is the same check missing or re-implemented differently in several places (a systemic broken-access / IDOR pattern)? Step 4: Check secrets posture (hardcoded or logged secrets, ad-hoc env reads), crypto usage (weak algorithms, predictable randomness), and dependency risk (advisory-bearing packages from `list_dependencies`). Step 5: For every CONFIRMED finding capture file:line, the vulnerability class, an exploit sketch, severity and likelihood, and a concrete remediation. Prefer findings that RECUR across the codebase — a class of bug appearing in many places is the systemic signal. Stage the running findings list to `notes/security/findings.md`; also note the threat classes you checked and found clean.",
|
|
92
|
+
"suggestedRole": "reviewer",
|
|
93
|
+
"advanceWhen": {
|
|
94
|
+
"file": "notes/security/findings.md",
|
|
95
|
+
"minBytes": 1,
|
|
96
|
+
"sniff": "nonempty"
|
|
97
|
+
},
|
|
98
|
+
"gate": {
|
|
99
|
+
"at": "completion",
|
|
100
|
+
"checks": [
|
|
101
|
+
{
|
|
102
|
+
"kind": "minBytes",
|
|
103
|
+
"file": "notes/security/findings.md",
|
|
104
|
+
"bytes": 120
|
|
105
|
+
},
|
|
106
|
+
{
|
|
107
|
+
"kind": "sniff",
|
|
108
|
+
"file": "notes/security/findings.md",
|
|
109
|
+
"sniff": "nonempty"
|
|
110
|
+
}
|
|
111
|
+
],
|
|
112
|
+
"onReject": "audit",
|
|
113
|
+
"maxAttempts": 3
|
|
114
|
+
},
|
|
115
|
+
"next": "synthesize"
|
|
116
|
+
},
|
|
117
|
+
{
|
|
118
|
+
"id": "synthesize",
|
|
119
|
+
"name": "Cluster findings into systemic themes",
|
|
120
|
+
"description": "group findings by root cause; rank by blast radius",
|
|
121
|
+
"prompt": "Step back from the individual findings and cluster them into SYSTEMIC THEMES — the root-cause patterns that explain many findings at once (e.g. 'input validation is inconsistent across the API routes', 'secrets are loaded ad hoc in several modules with no central config', 'there is no shared authorization layer so each endpoint re-checks access differently'). Step 1: Group the findings by ROOT CAUSE, not by file. Step 2: For each theme, state the root cause (the architectural gap), the BLAST RADIUS (how much of the system it touches and how many findings it explains), and the systemic FIX (the single change that would prevent the whole class). Step 3: Rank the themes by blast radius × severity. Write the themes to `notes/security/themes.md`. A clean codebase with no material findings can say so plainly rather than inventing themes.",
|
|
122
|
+
"suggestedRole": "reviewer",
|
|
123
|
+
"advanceWhen": {
|
|
124
|
+
"file": "notes/security/themes.md",
|
|
125
|
+
"minBytes": 1,
|
|
126
|
+
"sniff": "nonempty"
|
|
127
|
+
},
|
|
128
|
+
"gate": {
|
|
129
|
+
"at": "completion",
|
|
130
|
+
"checks": [
|
|
131
|
+
{
|
|
132
|
+
"kind": "minBytes",
|
|
133
|
+
"file": "notes/security/themes.md",
|
|
134
|
+
"bytes": 120
|
|
135
|
+
},
|
|
136
|
+
{
|
|
137
|
+
"kind": "sniff",
|
|
138
|
+
"file": "notes/security/themes.md",
|
|
139
|
+
"sniff": "nonempty"
|
|
140
|
+
}
|
|
141
|
+
],
|
|
142
|
+
"onReject": "synthesize",
|
|
143
|
+
"maxAttempts": 3
|
|
144
|
+
},
|
|
145
|
+
"next": "report"
|
|
146
|
+
},
|
|
147
|
+
{
|
|
148
|
+
"id": "report",
|
|
149
|
+
"name": "Write the security report",
|
|
150
|
+
"description": "systemic report + machine-readable findings, all citing real files",
|
|
151
|
+
"prompt": "Write the final deliverable as TWO files in the `security-review/` folder. Step 1: Write `security-review/findings.json` — a JSON array where each finding is an object with: `file` (a REAL workspace path), `line` (a number), `severity` (one of critical, high, medium, low, info), `category`, `title`, `remediation` (a concrete fix), and `theme` (the systemic theme it belongs to). Every `file` MUST be a real path in this workspace and every finding MUST have a line, a severity, and a concrete remediation — no fabricated paths. Step 2: Write `security-review/REPORT.md` with these sections, using those exact headings: `## Verdict` — a one-line top verdict (e.g. Block / Remediate-then-ship / Acceptable) with the count of findings by severity; do NOT say 'safe to merge' while any critical/high finding is open. `## Systemic Themes` — each theme from synthesize with its root cause and blast radius (this is the heart of a deep review). `## Findings` — each finding as `### [SEVERITY] class — file:line` with the exploit sketch, impact, and remediation, ordered by severity then blast radius. `## Verified Safe` — the threat classes you checked and found clean, so coverage is auditable. `## Not Statically Verifiable` — anything needing a pentest or runtime check. On a loop-back, fix ONLY the gaps the evaluator named without dropping findings that already passed.",
|
|
152
|
+
"suggestedRole": "reviewer",
|
|
153
|
+
"advanceWhen": {
|
|
154
|
+
"file": "security-review/REPORT.md",
|
|
155
|
+
"minBytes": 1,
|
|
156
|
+
"sniff": "nonempty"
|
|
157
|
+
},
|
|
158
|
+
"gate": {
|
|
159
|
+
"at": "completion",
|
|
160
|
+
"checks": [
|
|
161
|
+
{
|
|
162
|
+
"kind": "minBytes",
|
|
163
|
+
"file": "security-review/REPORT.md",
|
|
164
|
+
"bytes": 1500
|
|
165
|
+
}
|
|
166
|
+
],
|
|
167
|
+
"scripts": [
|
|
168
|
+
{
|
|
169
|
+
"name": "checkSecurityReport",
|
|
170
|
+
"scope": "standard",
|
|
171
|
+
"inputs": {
|
|
172
|
+
"file": "security-review/REPORT.md",
|
|
173
|
+
"findings": "security-review/findings.json"
|
|
174
|
+
}
|
|
175
|
+
}
|
|
176
|
+
],
|
|
177
|
+
"onReject": "report",
|
|
178
|
+
"maxAttempts": 4
|
|
179
|
+
},
|
|
180
|
+
"next": "evaluate"
|
|
181
|
+
},
|
|
182
|
+
{
|
|
183
|
+
"id": "evaluate",
|
|
184
|
+
"name": "Evaluate",
|
|
185
|
+
"description": "Grade the deliverable against every acceptance criterion. All pass → finish; any fail → loop back and fix the gap.",
|
|
186
|
+
"prompt": "Open `security-review/REPORT.md` and `security-review/findings.json` and grade the review against its acceptance criteria. Check EACH and write PASS/FAIL with a one-line reason: (1) every finding is pinned to a REAL file:line (no fabricated paths) with a severity and a concrete remediation; (2) the Systemic Themes section names the root-cause patterns with their blast radius — a deep review is about systemic issues, not a list of isolated nits; (3) untrusted-input findings trace a real source→sink path rather than restating a raw scanner hit; (4) the auth model was assessed HOLISTICALLY (consistency across all entry points), not file-by-file; (5) a Verified-safe section makes the review's coverage auditable; (6) the verdict matches the open severity (no 'safe' with an open critical/high); (7) no vague filler like 'consider validating inputs' — every finding is concrete. If any criterion fails, name the exact gap and loop back.\n\nThen route — this is the whole point of the loop:\n\n- **Every criterion PASSES →** call `advance_task_step({ ref, stepId: \"evaluate\", next: \"finish\" })`.\n- **Any criterion FAILS →** write the specific gaps to notes, then call `advance_task_step({ ref, stepId: \"evaluate\", next: \"report\" })` to loop back. The builder fixes exactly those gaps.\n\nNever route to `finish` while any criterion is unmet. The build phase's completion gate already blocked a grossly-incomplete deliverable; your job is the judgment an automated check cannot make (does it actually work, read well, look right). After ~3 unproductive loops, stop and report DONE_WITH_CONCERNS so the user can step in.",
|
|
187
|
+
"suggestedRole": "reviewer",
|
|
188
|
+
"next": "report"
|
|
189
|
+
},
|
|
190
|
+
{
|
|
191
|
+
"id": "finish",
|
|
192
|
+
"name": "Finish",
|
|
193
|
+
"description": "All acceptance criteria met. Stamp a short summary and report DONE.",
|
|
194
|
+
"prompt": "The security review passed every acceptance criterion. Write a one-paragraph DONE summary to task notes via `write_task_note`: the top-line verdict, the count of findings by severity, the systemic themes, and the deliverable paths (`security-review/REPORT.md` + `security-review/findings.json`). Then report DONE.",
|
|
195
|
+
"suggestedRole": "developer",
|
|
196
|
+
"terminal": true
|
|
197
|
+
}
|
|
198
|
+
],
|
|
199
|
+
"version": "1.1.0",
|
|
200
|
+
"releasedAt": "2026-07-28T00:00:00Z"
|
|
201
|
+
}
|
|
@@ -0,0 +1,157 @@
|
|
|
1
|
+
{
|
|
2
|
+
"schemaVersion": 1,
|
|
3
|
+
"title": "Deep Security Review systemic eval",
|
|
4
|
+
"objective": "Measure whether the review confirms seeded source-to-sink vulnerabilities, identifies systemic root causes, and emits valid machine-readable findings.",
|
|
5
|
+
"tags": [
|
|
6
|
+
"corpus"
|
|
7
|
+
],
|
|
8
|
+
"prompt": "Use the Deep Security Review craftbook for a whole-codebase security sweep. The fixture is offline and self-contained. Write security-review/REPORT.md and security-review/findings.json; do not modify source.",
|
|
9
|
+
"setup": {
|
|
10
|
+
"projectName": "Security Maintenance Eval",
|
|
11
|
+
"about": "A tiny admin service with deliberate command injection, SQL injection, and inconsistent authorization.",
|
|
12
|
+
"missionObjectives": "Confirm real source-to-sink paths, cluster root causes, and leave report plus structured findings without inventing files or advisories.",
|
|
13
|
+
"files": [
|
|
14
|
+
{
|
|
15
|
+
"path": "src/admin.ts",
|
|
16
|
+
"content": "import { exec } from 'node:child_process';\nexport function register(app, db) {\n app.get('/admin/export', (req, res) => {\n exec('zip -r /tmp/' + req.query.name + '.zip data', (_e, out) => res.send(out));\n });\n app.get('/admin/users', async (req, res) => {\n const rows = await db.query(`SELECT * FROM users WHERE team = '${req.query.team}'`);\n res.json(rows);\n });\n}\n"
|
|
17
|
+
},
|
|
18
|
+
{
|
|
19
|
+
"path": "src/profile.ts",
|
|
20
|
+
"content": "export function registerProfile(app, db, requireUser) {\n app.get('/profile', requireUser, async (req, res) => {\n const row = await db.query('SELECT * FROM users WHERE id = ?', [req.user.id]);\n res.json(row);\n });\n}\n"
|
|
21
|
+
},
|
|
22
|
+
{
|
|
23
|
+
"path": "docs/security.md",
|
|
24
|
+
"content": "# Security model\n\nAll /admin routes must require the requireAdmin middleware before business logic runs. SQL must use placeholders. Shell commands must not include request data.\n"
|
|
25
|
+
}
|
|
26
|
+
],
|
|
27
|
+
"worker": {
|
|
28
|
+
"name": "Rae",
|
|
29
|
+
"role": "Reviewer"
|
|
30
|
+
}
|
|
31
|
+
},
|
|
32
|
+
"mocks": [],
|
|
33
|
+
"success": {
|
|
34
|
+
"summary": "The report and findings JSON confirm both injection paths and the missing admin boundary with systemic remediation.",
|
|
35
|
+
"deliverables": [
|
|
36
|
+
{
|
|
37
|
+
"path": "security-review/REPORT.md",
|
|
38
|
+
"kind": "security-report",
|
|
39
|
+
"minBytes": 1500,
|
|
40
|
+
"checks": [
|
|
41
|
+
{
|
|
42
|
+
"kind": "contains",
|
|
43
|
+
"file": "security-review/REPORT.md",
|
|
44
|
+
"pattern": "src/admin\\.ts:[0-9]+",
|
|
45
|
+
"flags": "i",
|
|
46
|
+
"label": "real file and line"
|
|
47
|
+
},
|
|
48
|
+
{
|
|
49
|
+
"kind": "contains",
|
|
50
|
+
"file": "security-review/REPORT.md",
|
|
51
|
+
"pattern": "command injection|exec",
|
|
52
|
+
"flags": "i"
|
|
53
|
+
},
|
|
54
|
+
{
|
|
55
|
+
"kind": "contains",
|
|
56
|
+
"file": "security-review/REPORT.md",
|
|
57
|
+
"pattern": "SQL injection|parameter",
|
|
58
|
+
"flags": "i"
|
|
59
|
+
},
|
|
60
|
+
{
|
|
61
|
+
"kind": "contains",
|
|
62
|
+
"file": "security-review/REPORT.md",
|
|
63
|
+
"pattern": "authorization|requireAdmin|auth",
|
|
64
|
+
"flags": "i"
|
|
65
|
+
},
|
|
66
|
+
{
|
|
67
|
+
"kind": "contains",
|
|
68
|
+
"file": "security-review/REPORT.md",
|
|
69
|
+
"pattern": "Systemic Themes|root cause|blast radius",
|
|
70
|
+
"flags": "i"
|
|
71
|
+
},
|
|
72
|
+
{
|
|
73
|
+
"kind": "contains",
|
|
74
|
+
"file": "security-review/REPORT.md",
|
|
75
|
+
"pattern": "Verified Safe|Not Statically Verifiable",
|
|
76
|
+
"flags": "i"
|
|
77
|
+
}
|
|
78
|
+
]
|
|
79
|
+
},
|
|
80
|
+
{
|
|
81
|
+
"path": "security-review/findings.json",
|
|
82
|
+
"kind": "json",
|
|
83
|
+
"minBytes": 300,
|
|
84
|
+
"checks": [
|
|
85
|
+
{
|
|
86
|
+
"kind": "sniff",
|
|
87
|
+
"file": "security-review/findings.json",
|
|
88
|
+
"sniff": "json-valid"
|
|
89
|
+
},
|
|
90
|
+
{
|
|
91
|
+
"kind": "recordSchema",
|
|
92
|
+
"file": "security-review/findings.json",
|
|
93
|
+
"format": "json",
|
|
94
|
+
"minRows": 3,
|
|
95
|
+
"fields": [
|
|
96
|
+
{
|
|
97
|
+
"name": "file",
|
|
98
|
+
"type": "nonempty",
|
|
99
|
+
"required": true
|
|
100
|
+
},
|
|
101
|
+
{
|
|
102
|
+
"name": "line",
|
|
103
|
+
"type": "integer",
|
|
104
|
+
"required": true
|
|
105
|
+
},
|
|
106
|
+
{
|
|
107
|
+
"name": "severity",
|
|
108
|
+
"type": "^(critical|high|medium|low|info)$",
|
|
109
|
+
"required": true
|
|
110
|
+
},
|
|
111
|
+
{
|
|
112
|
+
"name": "category",
|
|
113
|
+
"type": "nonempty",
|
|
114
|
+
"required": true
|
|
115
|
+
},
|
|
116
|
+
{
|
|
117
|
+
"name": "remediation",
|
|
118
|
+
"type": "nonempty",
|
|
119
|
+
"required": true
|
|
120
|
+
},
|
|
121
|
+
{
|
|
122
|
+
"name": "theme",
|
|
123
|
+
"type": "nonempty",
|
|
124
|
+
"required": true
|
|
125
|
+
}
|
|
126
|
+
]
|
|
127
|
+
}
|
|
128
|
+
]
|
|
129
|
+
}
|
|
130
|
+
]
|
|
131
|
+
},
|
|
132
|
+
"rubric": {
|
|
133
|
+
"artifact": {
|
|
134
|
+
"path": "security-review/REPORT.md",
|
|
135
|
+
"kind": "markdown"
|
|
136
|
+
},
|
|
137
|
+
"axes": [
|
|
138
|
+
{
|
|
139
|
+
"name": "verification",
|
|
140
|
+
"description": "Raw leads are confirmed as real source-to-sink or authorization paths."
|
|
141
|
+
},
|
|
142
|
+
{
|
|
143
|
+
"name": "systemic analysis",
|
|
144
|
+
"description": "Findings are grouped by root cause and blast radius."
|
|
145
|
+
},
|
|
146
|
+
{
|
|
147
|
+
"name": "remediation",
|
|
148
|
+
"description": "The report and JSON contain concrete, severity-consistent fixes."
|
|
149
|
+
}
|
|
150
|
+
]
|
|
151
|
+
},
|
|
152
|
+
"qualityFocus": [
|
|
153
|
+
"confirmed vulnerabilities",
|
|
154
|
+
"systemic themes",
|
|
155
|
+
"machine-readable findings"
|
|
156
|
+
]
|
|
157
|
+
}
|
|
@@ -18,7 +18,7 @@
|
|
|
18
18
|
"hooks": [
|
|
19
19
|
{
|
|
20
20
|
"phase": "PreToolUse",
|
|
21
|
-
"matcher": "^(
|
|
21
|
+
"matcher": "^(write_file|append_to_file|replace_in_file|replace_lines|apply_patch|insert_at_marker|rename|rm|mkdir)$",
|
|
22
22
|
"script": {
|
|
23
23
|
"name": "check-freeze",
|
|
24
24
|
"scope": "craftbook"
|
|
@@ -31,30 +31,18 @@
|
|
|
31
31
|
"id": "set-boundary",
|
|
32
32
|
"name": "Agree the frozen directory",
|
|
33
33
|
"prompt": "Confirm with the user which single workspace directory edits are allowed in — use `ask_user_question` if the kickoff didn’t name one. Then record it by writing `.gezel/freeze.json` with exactly:\n\n```json\n{ \"dir\": \"<workspace-relative-directory>\" }\n```\n\nUse a relative path with no leading `./` and no trailing slash (e.g. `src/billing`). Tell the user the freeze is on and advance.",
|
|
34
|
-
"next": "frozen"
|
|
35
|
-
"suggestedRole": "developer",
|
|
36
|
-
"gate": {
|
|
37
|
-
"maxAttempts": 2,
|
|
38
|
-
"checks": [
|
|
39
|
-
{
|
|
40
|
-
"kind": "minBytes",
|
|
41
|
-
"file": ".gezel/freeze.json",
|
|
42
|
-
"bytes": 10
|
|
43
|
-
}
|
|
44
|
-
]
|
|
45
|
-
}
|
|
34
|
+
"next": "frozen"
|
|
46
35
|
},
|
|
47
36
|
{
|
|
48
37
|
"id": "frozen",
|
|
49
38
|
"name": "Freeze active",
|
|
50
39
|
"prompt": "The write boundary is active: any write outside the frozen directory is denied automatically — you will see the denial as a tool error naming the boundary. Do the scoped work normally inside the directory. If the work genuinely needs to touch something outside, do NOT fight the boundary: tell the user what needs changing and why, and let them either widen `.gezel/freeze.json` deliberately or end this task to lift the freeze.",
|
|
51
|
-
"terminal": true
|
|
52
|
-
"suggestedRole": "developer"
|
|
40
|
+
"terminal": true
|
|
53
41
|
}
|
|
54
42
|
],
|
|
55
43
|
"scripts": {
|
|
56
44
|
"check-freeze": "import { defineScript, gezel } from '@bendyline/gezel-sdk';\n\nexport const meta = defineScript({\n name: 'check-freeze',\n description: 'Block workspace writes outside the frozen directory (freeze scope).',\n inputs: {},\n outputs: {\n decision: { type: 'string', description: 'allow | deny' },\n message: { type: 'string', description: 'Reason shown on deny.' },\n },\n requires: ['workspace.read'],\n});\n\nconst STATE_FILE = '.gezel/freeze.json';\n\nfunction normalize(p: string): string {\n return p.replace(/\\\\/g, '/').replace(/\\/{2,}/g, '/').replace(/^\\.\\//, '').replace(/\\/$/, '');\n}\n\nasync function main(): Promise<void> {\n const input = gezel.input as { args?: Record<string, unknown> };\n const args = input.args ?? {};\n\n let frozenDir = '';\n try {\n const raw = await gezel.fs.read(STATE_FILE);\n frozenDir = normalize(String(JSON.parse(raw).dir ?? ''));\n } catch {\n // Not configured yet (the setup step writes it) — allow everything.\n gezel.output({ decision: 'allow', message: '' });\n return;\n }\n if (!frozenDir) {\n gezel.output({ decision: 'allow', message: '' });\n return;\n }\n\n // Every path-like argument the write-tool family carries.\n const candidates = ['path', 'from', 'to', 'file']\n .map((key) => (args as Record<string, unknown>)[key])\n .filter((value): value is string => typeof value === 'string' && value.length > 0)\n .map(normalize);\n if (candidates.length === 0) {\n gezel.output({ decision: 'allow', message: '' });\n return;\n }\n\n for (const candidate of candidates) {\n const inside = candidate === frozenDir || candidate.startsWith(frozenDir + '/');\n const isStateFile = candidate === normalize(STATE_FILE);\n if (candidate.includes('..')) {\n gezel.output({ decision: 'deny', message: '[freeze] Path escapes are blocked while freeze scope is active.' });\n return;\n }\n if (!inside && !isStateFile) {\n gezel.output({\n decision: 'deny',\n message:\n '[freeze] Writes are frozen to \"' + frozenDir + '/\" — \"' + candidate + '\" is outside it. Finish the scoped work first, or end the freeze-scope task to lift the boundary.',\n });\n return;\n }\n }\n gezel.output({ decision: 'allow', message: '' });\n}\n\nawait main();\n"
|
|
57
45
|
},
|
|
58
|
-
"version": "1.0
|
|
59
|
-
"releasedAt": "2026-07-
|
|
46
|
+
"version": "1.1.0",
|
|
47
|
+
"releasedAt": "2026-07-28T00:00:00Z"
|
|
60
48
|
}
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
{
|
|
2
|
+
"id": "investigate",
|
|
3
|
+
"name": "Investigate",
|
|
4
|
+
"description": "Systematic root-cause debugging procedure. Six steps:\n\n1. **Gather symptoms** — observable signals, no theorizing yet\n2. **Search prior** — has this project seen this before? (memory)\n3. **Reproduce** — deterministic repro before any tracing\n4. **Trace to root cause** — keep asking 'and why is THAT?'\n5. **Propose fix** — state it; don't apply unless asked\n6. **Save learning** — stamp (symptom → root cause → fix) into memory\n\nPorted from gstack's `/investigate`. The principle is the Iron Law:\n**no fixes without investigation**. Jumping to a fix without root cause\neither ships a coincidence (the bug comes back), papers over the real\ncause (introduces tech debt), or fixes a symptom you didn't actually\nunderstand.\n\nWhere gstack used `~/.gstack/projects/{slug}/learnings.jsonl`, this\ncraftbook uses the per-gezel/project memory subsystem. The\n`search-prior.ts` script wraps `search_memory` with the symptom-shape\nquery; `save-learning.ts` wraps `save_memory` with structured metadata\nso future `search-prior` queries find the match.\n\nThe Iron Law is not literal: sometimes a one-line typo fix is the right\nmove. The principle is \"if you can't say *why* this fix works, you're\nguessing.\" The craftbook's structure makes that guessing visible.\n",
|
|
5
|
+
"entryStepId": "gather-symptoms",
|
|
6
|
+
"triggers": [
|
|
7
|
+
"investigate this",
|
|
8
|
+
"debug this",
|
|
9
|
+
"why is this broken",
|
|
10
|
+
"find the bug",
|
|
11
|
+
"root cause"
|
|
12
|
+
],
|
|
13
|
+
"steps": [
|
|
14
|
+
{
|
|
15
|
+
"id": "gather-symptoms",
|
|
16
|
+
"name": "Gather symptoms",
|
|
17
|
+
"description": "Pull together every observable signal: error messages, stack traces, logs, repro steps, when it started.",
|
|
18
|
+
"prompt": "**For this task you are an investigator, not an author.** The Iron Law: no fixes without investigation. Resist the urge to `write_file` a fix — even if your default persona is a developer, this craftbook is about *understanding what broke* and the fix comes later (separate task).\n\n**Your first action this turn:** call `ask_user_question` with the four bullets below if any signal is missing from the chat. If the user already supplied the symptom in their message, skip to `write_task_note`.\n\nElicit (if missing):\n- What's the observable failure? (exact error message, screenshot, stack trace, log line)\n- When did it start? (recent commit? deploy? environment change?)\n- Is it reproducible? (every time? intermittent? specific input?)\n- What's the smallest input that triggers it?\n\nThen `write_task_note({ ref, content: <3-5 line symptom summary> })`. Do NOT theorize yet. Do NOT propose causes — that's the next step. Do NOT call `read_task_notes` to find the procedure; it's right here.",
|
|
19
|
+
"suggestedRole": "developer",
|
|
20
|
+
"next": "search-prior"
|
|
21
|
+
},
|
|
22
|
+
{
|
|
23
|
+
"id": "search-prior",
|
|
24
|
+
"name": "Search prior",
|
|
25
|
+
"description": "Have we seen this before? Search memory for prior investigations in this project.",
|
|
26
|
+
"prompt": "Run the `search-prior` script. It searches project memory for prior investigations matching the symptom signature. If we've seen this before, the model returns the prior learning + the fix that landed.\n\nIf there's a match: state what the prior investigation found, and ask the user if this looks like the same bug (with `ask_user_question`). If yes, jump straight to the fix. If no, proceed to fresh investigation.\n\nIf there's no match: note 'no prior matches' to task notes and advance.",
|
|
27
|
+
"suggestedRole": "developer",
|
|
28
|
+
"onExit": {
|
|
29
|
+
"name": "search-prior",
|
|
30
|
+
"autoAdvanceWhen": {
|
|
31
|
+
"op": "ok"
|
|
32
|
+
}
|
|
33
|
+
},
|
|
34
|
+
"next": "reproduce"
|
|
35
|
+
},
|
|
36
|
+
{
|
|
37
|
+
"id": "reproduce",
|
|
38
|
+
"name": "Reproduce",
|
|
39
|
+
"description": "Get a deterministic repro. Often the longest step. Cannot fix what can't be reproduced.",
|
|
40
|
+
"prompt": "Get the bug to fire on demand. Strategies:\n\n1. **Read the failing path** — find the function/route from the stack trace (use `read_file` / `search_files`).\n2. **Walk inputs backward** — what user action triggered this? What's the request shape?\n3. **Try the minimal trigger** — strip the user's repro down. Half the bugs disappear when you isolate.\n4. **Run a small script if needed** — `run_nodejs_script` against a synthetic input.\n5. **If it won't reproduce**: state that explicitly. Either get more info from the user (`ask_user_question`) or escalate.\n\nWhen you have a repro, write the minimal trigger to task notes. Do not move on without one.",
|
|
41
|
+
"suggestedRole": "developer",
|
|
42
|
+
"next": "trace"
|
|
43
|
+
},
|
|
44
|
+
{
|
|
45
|
+
"id": "trace",
|
|
46
|
+
"name": "Trace to root cause",
|
|
47
|
+
"description": "Follow the data from the failure point back to the smallest 'this is the thing that's wrong.'",
|
|
48
|
+
"prompt": "Trace backward from the failure. Read the code path. The Iron Law: **the cause is not 'the function returned the wrong thing,' it's WHY the function returned the wrong thing.** Keep asking 'and why is THAT?' until you hit one of:\n\n- A coding bug (assignment swapped, condition inverted, type mismatch)\n- A bad invariant (assumed X but the data has Y)\n- A wrong assumption about a dependency (the API returns null in this case)\n- An environment / config issue (the staging DB doesn't have this column)\n\nDocument the chain in task notes. Bullet form is fine. End with: 'Root cause: <one sentence>.'",
|
|
49
|
+
"suggestedRole": "developer",
|
|
50
|
+
"next": "propose-fix"
|
|
51
|
+
},
|
|
52
|
+
{
|
|
53
|
+
"id": "propose-fix",
|
|
54
|
+
"name": "Propose fix",
|
|
55
|
+
"description": "State the fix. Don't apply it yet — let the user (or a separate craftbook) decide.",
|
|
56
|
+
"prompt": "Now the easy part. State the fix in plain language:\n\n- What file/function changes?\n- What's the new behavior?\n- What did NOT change (so the user knows you didn't gold-plate)?\n- What test would catch this regressing?\n\nDo NOT apply the fix unless the user explicitly asks. Investigation lives separate from edit-the-code so the diff stays minimal and reviewable. Advance.",
|
|
57
|
+
"suggestedRole": "developer",
|
|
58
|
+
"next": "save-learning"
|
|
59
|
+
},
|
|
60
|
+
{
|
|
61
|
+
"id": "save-learning",
|
|
62
|
+
"name": "Save learning",
|
|
63
|
+
"description": "Stamp the (symptom → root cause → fix) triple into project memory so future investigations find it.",
|
|
64
|
+
"prompt": "Run the `save-learning` script with the symptom, root cause, and proposed fix. It saves them to project memory with appropriate tags so `search-prior` finds them next time.\n\nThen report DONE with a one-line summary: 'Investigated <symptom>; root cause: <cause>; fix proposed: <fix>.'",
|
|
65
|
+
"suggestedRole": "developer",
|
|
66
|
+
"onExit": {
|
|
67
|
+
"name": "save-learning",
|
|
68
|
+
"autoAdvanceWhen": {
|
|
69
|
+
"op": "ok"
|
|
70
|
+
}
|
|
71
|
+
},
|
|
72
|
+
"terminal": true
|
|
73
|
+
}
|
|
74
|
+
],
|
|
75
|
+
"scripts": {
|
|
76
|
+
"search-prior": "import { defineScript, gezel } from '@bendyline/gezel-sdk';\n\nexport const meta = defineScript({\n name: 'search-prior',\n description:\n 'Search project memory for prior investigations matching a symptom signature. Stamps {matches: number, hits: [{text, score}]}.',\n inputs: {\n symptom: {\n type: 'string',\n description: 'Short symptom description (1-3 sentences). Used as the search query.',\n required: true,\n },\n },\n outputs: {\n matches: { type: 'number', description: 'Number of memory hits returned.' },\n hits: {\n type: 'array',\n description: 'Top matches, each with text + score.',\n itemType: 'object',\n },\n },\n requires: ['memory.read'],\n});\n\ninterface MemoryHit {\n text: string;\n score: number;\n}\n\nasync function main(): Promise<void> {\n const input = gezel.input as { symptom: string };\n const query = `investigation: ${input.symptom}`;\n const results = (await gezel.memory.search(query)) as MemoryHit[];\n const filtered = (results ?? []).filter(\n (r): r is MemoryHit => Boolean(r) && typeof r.text === 'string' && typeof r.score === 'number',\n );\n gezel.log(`Memory search for \"${query}\" returned ${filtered.length} hit(s)`);\n gezel.output({ matches: filtered.length, hits: filtered.slice(0, 5) });\n}\n\nawait main();\n",
|
|
77
|
+
"save-learning": "import { defineScript, gezel } from '@bendyline/gezel-sdk';\n\nexport const meta = defineScript({\n name: 'save-learning',\n description:\n 'Stamp a (symptom → root cause → fix) investigation triple into project memory so future `search-prior` queries find it.',\n inputs: {\n symptom: { type: 'string', description: 'Observable failure.', required: true },\n rootCause: { type: 'string', description: 'Root cause in one sentence.', required: true },\n fix: { type: 'string', description: 'Proposed fix.', required: true },\n files: {\n type: 'string',\n description: 'Comma-separated list of files touched (or expected to be touched).',\n },\n },\n outputs: {\n ok: { type: 'boolean', description: 'True on successful save.' },\n },\n requires: ['memory.write'],\n});\n\nasync function main(): Promise<void> {\n const input = gezel.input as {\n symptom: string;\n rootCause: string;\n fix: string;\n files?: string;\n };\n const text = [\n 'investigation:',\n `symptom: ${input.symptom}`,\n `root cause: ${input.rootCause}`,\n `fix: ${input.fix}`,\n ...(input.files ? [`files: ${input.files}`] : []),\n ].join('\\n');\n await gezel.memory.save(text, {\n kind: 'investigation',\n ...(input.files ? { files: input.files } : {}),\n });\n gezel.log('Saved investigation to memory.');\n gezel.output({ ok: true });\n}\n\nawait main();\n"
|
|
78
|
+
},
|
|
79
|
+
"version": "1.1.0",
|
|
80
|
+
"releasedAt": "2026-07-28T00:00:00Z"
|
|
81
|
+
}
|