@bendyline/gilde 0.1.71 → 0.1.73
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/data/craftbook-templates/ap/apply-review-findings/versions/1.0.3/craftbook.json +604 -0
- package/data/craftbook-templates/ap/apply-review-findings/versions/1.0.3/test.json +290 -0
- package/data/craftbook-templates/co/code-review/versions/1.0.2/craftbook.json +230 -0
- package/data/craftbook-templates/co/code-review/versions/1.0.2/test.json +252 -0
- package/data/craftbook-templates/co/content-accuracy-review/versions/1.1.0/craftbook.json +311 -0
- package/data/craftbook-templates/co/content-accuracy-review/versions/1.1.0/test.json +167 -0
- package/data/craftbook-templates/eb/ebook-compile/versions/1.1.0/craftbook.json +332 -0
- package/data/craftbook-templates/eb/ebook-compile/versions/1.1.0/test.json +111 -0
- package/data/craftbook-templates/fa/faq-from-docs/versions/1.1.0/craftbook.json +245 -0
- package/data/craftbook-templates/fa/faq-from-docs/versions/1.1.0/test.json +113 -0
- package/data/craftbook-templates/index.json +1 -1
- package/data/craftbook-templates/me/meeting-minutes/versions/1.1.0/craftbook.json +246 -0
- package/data/craftbook-templates/me/meeting-minutes/versions/1.1.0/test.json +113 -0
- package/data/craftbook-templates/me/meeting-notes-to-actions/versions/1.1.0/craftbook.json +245 -0
- package/data/craftbook-templates/me/meeting-notes-to-actions/versions/1.1.0/test.json +113 -0
- package/data/craftbook-templates/ni/nightly-fix-sweep/versions/1.0.2/craftbook.json +243 -0
- package/data/craftbook-templates/ni/nightly-fix-sweep/versions/1.0.2/test.json +120 -0
- package/data/craftbook-templates/po/powerpoint-deck/versions/1.7.12/craftbook.json +581 -0
- package/data/craftbook-templates/po/powerpoint-deck/versions/1.7.12/test.json +240 -0
- package/data/craftbook-templates/pr/press-release/versions/1.0.4/craftbook.json +281 -0
- package/data/craftbook-templates/pr/press-release/versions/1.0.4/test.json +356 -0
- package/data/craftbook-templates/st/standup-summary/versions/1.1.0/craftbook.json +249 -0
- package/data/craftbook-templates/st/standup-summary/versions/1.1.0/test.json +113 -0
- package/data/craftbook-templates/we/weekly-review/versions/1.0.4/craftbook.json +223 -0
- package/data/craftbook-templates/we/weekly-review/versions/1.0.4/test.json +110 -0
- package/package.json +1 -1
|
@@ -0,0 +1,252 @@
|
|
|
1
|
+
{
|
|
2
|
+
"schemaVersion": 1,
|
|
3
|
+
"title": "Snapshot-driven code review with a gated artifact report",
|
|
4
|
+
"objective": "Measure the snapshot-review class under volume: read a staged change-set snapshot from the artifacts drawer, ground findings in the actual diff, catch every seeded defect, raise nothing against the constructs that only look like defects, and produce a severity-ranked, file-cited report artifact whose verdict follows from the findings.",
|
|
5
|
+
"tags": [],
|
|
6
|
+
"prompt": "A change-set snapshot for review rev-eval-1 is staged in this project's artifacts drawer: reviews/rev-eval-1/manifest.json and reviews/rev-eval-1/changes.diff. Run the code-review craftbook against it and write the gated report artifact to reviews/rev-eval-1/report.md with write_artifact. The diff is the change set under review; the matching source files are in the workspace, and docs/review-conventions.md is the house style this codebase is reviewed against. Also leave the machine-readable finding list at the workspace path review-output/findings.json with write_file - a JSON array with one record per finding, each record naming the file (the path as it appears in the diff), the line, a severity of critical, major, minor or nit, a category, and a recommendation - so the follow-up fix pass can pick it up. Do not change any source file: review-output/findings.json is the only workspace file you write.",
|
|
7
|
+
"setup": {
|
|
8
|
+
"projectName": "Code Review Eval",
|
|
9
|
+
"about": "A small payments service. A seven-file change set is staged for review in the artifacts drawer; the house style it is reviewed against is documented in docs/review-conventions.md.",
|
|
10
|
+
"worker": {
|
|
11
|
+
"name": "Rex",
|
|
12
|
+
"role": "Reviewer"
|
|
13
|
+
},
|
|
14
|
+
"craftbookParams": {
|
|
15
|
+
"reviewId": "rev-eval-1"
|
|
16
|
+
},
|
|
17
|
+
"files": [
|
|
18
|
+
{
|
|
19
|
+
"path": "src/payment.js",
|
|
20
|
+
"content": "const db = require('./db');\n\nasync function chargeCustomer(req) {\n const amount = req.body.amount;\n const customerId = req.body.customerId;\n // TODO: validate amount server-side\n const rows = await db.query(\"SELECT * FROM customers WHERE id = '\" + customerId + \"'\");\n const customer = rows[0];\n console.log('charging', customerId, amount);\n const receipt = await db.query(\n \"INSERT INTO charges (customer_id, amount) VALUES ('\" + customerId + \"', \" + amount + \") RETURNING id\",\n );\n return { chargeId: receipt[0].id, customer: customer.name };\n}\n\nmodule.exports = { chargeCustomer };\n"
|
|
21
|
+
},
|
|
22
|
+
{
|
|
23
|
+
"path": "src/util.js",
|
|
24
|
+
"content": "function formatCents(cents) {\n return `$${(cents / 100).toFixed(2)}`;\n}\n\nmodule.exports = { formatCents };\n",
|
|
25
|
+
"modelInput": false
|
|
26
|
+
},
|
|
27
|
+
{
|
|
28
|
+
"path": "src/pagination.js",
|
|
29
|
+
"content": "const DEFAULT_PAGE_SIZE = 25;\n\nfunction pageOf(items, page, pageSize = DEFAULT_PAGE_SIZE) {\n const start = (page - 1) * pageSize;\n const end = start + pageSize + 1;\n return items.slice(start, end);\n}\n\nfunction pageCount(items, pageSize = DEFAULT_PAGE_SIZE) {\n return Math.ceil(items.length / pageSize);\n}\n\nmodule.exports = { pageOf, pageCount, DEFAULT_PAGE_SIZE };\n",
|
|
30
|
+
"modelInput": false
|
|
31
|
+
},
|
|
32
|
+
{
|
|
33
|
+
"path": "src/settings.js",
|
|
34
|
+
"content": "const fs = require('fs');\n\nfunction loadSettings(path) {\n try {\n return { ok: true, settings: JSON.parse(fs.readFileSync(path, 'utf8')) };\n } catch (err) {\n return { ok: true, settings: {} };\n }\n}\n\nmodule.exports = { loadSettings };\n",
|
|
35
|
+
"modelInput": false
|
|
36
|
+
},
|
|
37
|
+
{
|
|
38
|
+
"path": "src/export.js",
|
|
39
|
+
"content": "const fs = require('fs/promises');\n\nasync function exportCharges(db, outPath) {\n const handle = await fs.open(outPath, 'w');\n const rows = await db.query('SELECT id, customer_id, amount FROM charges ORDER BY id');\n for (const row of rows) {\n await handle.write(`${row.id},${row.customer_id},${row.amount}\\n`);\n }\n await handle.close();\n}\n\nmodule.exports = { exportCharges };\n",
|
|
40
|
+
"modelInput": false
|
|
41
|
+
},
|
|
42
|
+
{
|
|
43
|
+
"path": "src/customer.js",
|
|
44
|
+
"content": "// `== null` is the one loose comparison the house style allows: it matches\n// null and undefined together and nothing else, which is exactly what an\n// optional field needs. See docs/review-conventions.md.\nfunction displayName(customer) {\n if (customer.nickname == null) return customer.legalName;\n return customer.nickname;\n}\n\nfunction billingEmail(customer) {\n return customer.billingEmail == null ? customer.email : customer.billingEmail;\n}\n\nmodule.exports = { displayName, billingEmail };\n",
|
|
45
|
+
"modelInput": false
|
|
46
|
+
},
|
|
47
|
+
{
|
|
48
|
+
"path": "src/audit.js",
|
|
49
|
+
"content": "const logger = require('./logger');\n\n// Audit writes are best-effort: they must never delay or break the charge\n// that triggered them, so this call is deliberately not awaited. The\n// attached .catch is what keeps a rejected write from becoming an\n// unhandled rejection. See docs/review-conventions.md.\nfunction recordAudit(db, event) {\n db.insertAudit(event).catch((err) => logger.warn('audit write skipped', err));\n}\n\nmodule.exports = { recordAudit };\n",
|
|
50
|
+
"modelInput": false
|
|
51
|
+
},
|
|
52
|
+
{
|
|
53
|
+
"path": "src/errors.js",
|
|
54
|
+
"content": "const logger = require('./logger');\n\n// Express selects error middleware by ARITY: a handler is only registered\n// as an error handler when it declares exactly four parameters. `req` and\n// `next` are required by that signature even though this terminal handler\n// uses neither. See docs/review-conventions.md.\nfunction errorHandler(err, req, res, next) {\n logger.error('unhandled request error', err);\n res.status(500).json({ error: 'internal_error' });\n}\n\nfunction register(app) {\n app.use(errorHandler);\n}\n\nmodule.exports = { errorHandler, register };\n",
|
|
55
|
+
"modelInput": false
|
|
56
|
+
},
|
|
57
|
+
{
|
|
58
|
+
"path": "src/logger.js",
|
|
59
|
+
"content": "function warn(message, err) {\n process.stderr.write(`[warn] ${message}: ${err && err.message}\\n`);\n}\n\nfunction error(message, err) {\n process.stderr.write(`[error] ${message}: ${err && err.stack}\\n`);\n}\n\nmodule.exports = { warn, error };\n",
|
|
60
|
+
"modelInput": false
|
|
61
|
+
},
|
|
62
|
+
{
|
|
63
|
+
"path": "docs/review-conventions.md",
|
|
64
|
+
"content": "# Review conventions\n\nThe house rules this codebase is reviewed against.\n\n- Every SQL statement uses placeholders. A request value is never\n concatenated or interpolated into a query string.\n- Every acquired resource (file handle, connection, lock) is released on\n every path out of the function, including the error path.\n- A caught error is either handled or re-thrown. A catch block never\n reports success.\n- Slice and range bounds are half-open: `slice(start, start + size)`.\n- Debug logging does not ship.\n\n## Deliberate patterns - correct as written\n\nThese three shapes look like common smells and are not. They are house\nstyle, and a review must not raise them as findings.\n\n- `== null` is the one loose comparison the style allows. It matches null\n and undefined together and nothing else, which is precisely what an\n optional field needs. It is not to be tightened to `===`.\n- Best-effort background writes (audit, telemetry) are deliberately not\n awaited, so they cannot delay or break the request that triggered them.\n A `.catch` handler attached at the call site is the required form; a\n promise carrying a `.catch` is not a floating promise.\n- Express selects error middleware by arity: a handler is only registered\n as an error handler when it declares exactly four parameters. The\n parameters such a handler does not use are required by that signature\n and stay.\n"
|
|
65
|
+
},
|
|
66
|
+
{
|
|
67
|
+
"path": "reviews/rev-eval-1/manifest.json",
|
|
68
|
+
"surface": "artifact",
|
|
69
|
+
"content": "{\n \"version\": 1,\n \"reviewId\": \"rev-eval-1\",\n \"kind\": \"commit\",\n \"createdAt\": \"2026-07-29T09:00:00Z\",\n \"branch\": \"main\",\n \"headSha\": \"3d0c1a2\",\n \"baseRef\": \"HEAD\",\n \"files\": [\n {\n \"path\": \"src/payment.js\",\n \"kind\": \"modified\",\n \"additions\": 8,\n \"deletions\": 2\n },\n {\n \"path\": \"src/audit.js\",\n \"kind\": \"added\",\n \"additions\": 11,\n \"deletions\": 0\n },\n {\n \"path\": \"src/customer.js\",\n \"kind\": \"added\",\n \"additions\": 13,\n \"deletions\": 0\n },\n {\n \"path\": \"src/errors.js\",\n \"kind\": \"added\",\n \"additions\": 16,\n \"deletions\": 0\n },\n {\n \"path\": \"src/export.js\",\n \"kind\": \"added\",\n \"additions\": 12,\n \"deletions\": 0\n },\n {\n \"path\": \"src/pagination.js\",\n \"kind\": \"added\",\n \"additions\": 13,\n \"deletions\": 0\n },\n {\n \"path\": \"src/settings.js\",\n \"kind\": \"added\",\n \"additions\": 11,\n \"deletions\": 0\n }\n ],\n \"totalFiles\": 7,\n \"filesTruncated\": false,\n \"diffFile\": \"changes.diff\",\n \"diffChars\": 4396,\n \"diffTruncated\": false,\n \"notes\": []\n}\n"
|
|
70
|
+
},
|
|
71
|
+
{
|
|
72
|
+
"path": "reviews/rev-eval-1/changes.diff",
|
|
73
|
+
"surface": "artifact",
|
|
74
|
+
"content": "diff --git a/src/payment.js b/src/payment.js\nindex 3d0c1a2..9f4b7e1 100644\n--- a/src/payment.js\n+++ b/src/payment.js\n@@ -1,12 +1,16 @@\n const db = require('./db');\n \n async function chargeCustomer(req) {\n const amount = req.body.amount;\n const customerId = req.body.customerId;\n- const rows = await db.query('SELECT * FROM customers WHERE id = $1', [customerId]);\n+ // TODO: validate amount server-side\n+ const rows = await db.query(\"SELECT * FROM customers WHERE id = '\" + customerId + \"'\");\n const customer = rows[0];\n- const receipt = await db.query('INSERT INTO charges (customer_id, amount) VALUES ($1, $2) RETURNING id', [customerId, amount]);\n+ console.log('charging', customerId, amount);\n+ const receipt = await db.query(\n+ \"INSERT INTO charges (customer_id, amount) VALUES ('\" + customerId + \"', \" + amount + \") RETURNING id\",\n+ );\n return { chargeId: receipt[0].id, customer: customer.name };\n }\n \n module.exports = { chargeCustomer };\ndiff --git a/src/audit.js b/src/audit.js\nnew file mode 100644\nindex 0000000..297df83\n--- /dev/null\n+++ b/src/audit.js\n@@ -0,0 +1,11 @@\n+const logger = require('./logger');\n+\n+// Audit writes are best-effort: they must never delay or break the charge\n+// that triggered them, so this call is deliberately not awaited. The\n+// attached .catch is what keeps a rejected write from becoming an\n+// unhandled rejection. See docs/review-conventions.md.\n+function recordAudit(db, event) {\n+ db.insertAudit(event).catch((err) => logger.warn('audit write skipped', err));\n+}\n+\n+module.exports = { recordAudit };\ndiff --git a/src/customer.js b/src/customer.js\nnew file mode 100644\nindex 0000000..c722780\n--- /dev/null\n+++ b/src/customer.js\n@@ -0,0 +1,13 @@\n+// `== null` is the one loose comparison the house style allows: it matches\n+// null and undefined together and nothing else, which is exactly what an\n+// optional field needs. See docs/review-conventions.md.\n+function displayName(customer) {\n+ if (customer.nickname == null) return customer.legalName;\n+ return customer.nickname;\n+}\n+\n+function billingEmail(customer) {\n+ return customer.billingEmail == null ? customer.email : customer.billingEmail;\n+}\n+\n+module.exports = { displayName, billingEmail };\ndiff --git a/src/errors.js b/src/errors.js\nnew file mode 100644\nindex 0000000..7107b89\n--- /dev/null\n+++ b/src/errors.js\n@@ -0,0 +1,16 @@\n+const logger = require('./logger');\n+\n+// Express selects error middleware by ARITY: a handler is only registered\n+// as an error handler when it declares exactly four parameters. `req` and\n+// `next` are required by that signature even though this terminal handler\n+// uses neither. See docs/review-conventions.md.\n+function errorHandler(err, req, res, next) {\n+ logger.error('unhandled request error', err);\n+ res.status(500).json({ error: 'internal_error' });\n+}\n+\n+function register(app) {\n+ app.use(errorHandler);\n+}\n+\n+module.exports = { errorHandler, register };\ndiff --git a/src/export.js b/src/export.js\nnew file mode 100644\nindex 0000000..01f5dcf\n--- /dev/null\n+++ b/src/export.js\n@@ -0,0 +1,12 @@\n+const fs = require('fs/promises');\n+\n+async function exportCharges(db, outPath) {\n+ const handle = await fs.open(outPath, 'w');\n+ const rows = await db.query('SELECT id, customer_id, amount FROM charges ORDER BY id');\n+ for (const row of rows) {\n+ await handle.write(`${row.id},${row.customer_id},${row.amount}\\n`);\n+ }\n+ await handle.close();\n+}\n+\n+module.exports = { exportCharges };\ndiff --git a/src/pagination.js b/src/pagination.js\nnew file mode 100644\nindex 0000000..008b78a\n--- /dev/null\n+++ b/src/pagination.js\n@@ -0,0 +1,13 @@\n+const DEFAULT_PAGE_SIZE = 25;\n+\n+function pageOf(items, page, pageSize = DEFAULT_PAGE_SIZE) {\n+ const start = (page - 1) * pageSize;\n+ const end = start + pageSize + 1;\n+ return items.slice(start, end);\n+}\n+\n+function pageCount(items, pageSize = DEFAULT_PAGE_SIZE) {\n+ return Math.ceil(items.length / pageSize);\n+}\n+\n+module.exports = { pageOf, pageCount, DEFAULT_PAGE_SIZE };\ndiff --git a/src/settings.js b/src/settings.js\nnew file mode 100644\nindex 0000000..b7c463b\n--- /dev/null\n+++ b/src/settings.js\n@@ -0,0 +1,11 @@\n+const fs = require('fs');\n+\n+function loadSettings(path) {\n+ try {\n+ return { ok: true, settings: JSON.parse(fs.readFileSync(path, 'utf8')) };\n+ } catch (err) {\n+ return { ok: true, settings: {} };\n+ }\n+}\n+\n+module.exports = { loadSettings };\n"
|
|
75
|
+
},
|
|
76
|
+
{
|
|
77
|
+
"path": "tests/code-review-oracle.mjs",
|
|
78
|
+
"content": "// Precision/recall oracle for the snapshot code review.\n//\n// Run by the harness with the reviewed workspace as cwd. It grades the two\n// halves of a real review separately, because recall alone is passed by a\n// review that flags everything:\n//\n// RECALL - each of the four seeded defects in the change set is\n// reported, against the right file and as the right class.\n// PRECISION - no finding repeats one of the three shotgun smells the\n// change set deliberately baits, and no finding is raised\n// against a file outside the change set.\n//\n// Two rules keep this from punishing correct work.\n//\n// First, defect classes are matched by the ordinary vocabulary of the\n// class, never by wording this fixture invented: \"26 items per page\",\n// \"the end index is one too high\" and \"half-open range\" all count as the\n// pagination defect. The bar is \"a real review of this class\", not this\n// scenario's phrasing.\n//\n// Second, the precision half is scoped to the SMELL, not to the file.\n// src/customer.js, src/audit.js and src/errors.js are baited with a\n// documented house-style construct that must not be flagged - but they\n// are still ordinary code, and a reviewer may legitimately find\n// something else in them (a missing res.headersSent guard in the error\n// handler, say, or a synchronous throw escaping the best-effort audit\n// call). Failing those would punish the exact deep reading this eval is\n// trying to reward, so only a finding that IS the baited smell counts\n// against precision. Whole-file scope is reserved for the two files that\n// are not in the change set at all, where any finding is out of scope by\n// the book's own rule.\n//\n// Precision is graded on the finding records only: a report that\n// discusses a correct construct and explicitly clears it is good\n// reviewing, and is not penalised here.\nimport { readFileSync } from 'node:fs';\n\n/**\n * Fail with the message ALONE on stderr.\n *\n * A thrown Error would work, but the harness feeds this script's stderr\n * into the model's repair nudge verbatim, and an uncaught throw pads that\n * nudge with a Node stack trace and a version banner. The actionable\n * sentence is what the next attempt needs; the stack is noise that pushes\n * it out of view on a small model.\n */\nfunction fail(message) {\n console.error(message);\n process.exit(1);\n}\n\n\nconst FINDINGS = 'review-output/findings.json';\n\nlet raw;\ntry {\n raw = readFileSync(FINDINGS, 'utf8');\n} catch {\n fail(FINDINGS + ' does not exist - the machine-readable finding list is required');\n}\n\nlet findings;\ntry {\n findings = JSON.parse(raw.replace(/^\\uFEFF/, ''));\n} catch (err) {\n fail(FINDINGS + ' is not valid JSON: ' + err.message);\n}\nif (!Array.isArray(findings)) {\n const nested =\n findings && typeof findings === 'object'\n ? Object.values(findings).find((value) => Array.isArray(value))\n : null;\n if (!nested) fail(FINDINGS + ' must be an array of finding records');\n findings = nested;\n}\nfindings = findings.filter((finding) => finding && typeof finding === 'object');\n\n// Every PROSE string anywhere in the record, nested objects and arrays\n// included - a reviewer who nests the explanation under `detail` or\n// splits it across an array is saying the same thing. Citation and\n// identifier fields are excluded: they are addresses, not description,\n// and letting them through means a line number or a path fragment can\n// satisfy a defect-class matcher by coincidence.\nconst CITATION_KEYS = new Set([\n 'file',\n 'path',\n 'filename',\n 'location',\n 'line',\n 'lines',\n 'endline',\n 'startline',\n 'column',\n 'col',\n 'id',\n '#',\n]);\nconst collectStrings = (value, out) => {\n if (typeof value === 'string') out.push(value);\n else if (Array.isArray(value)) for (const item of value) collectStrings(item, out);\n else if (value && typeof value === 'object') {\n for (const [key, item] of Object.entries(value)) {\n if (CITATION_KEYS.has(key.toLowerCase())) continue;\n collectStrings(item, out);\n }\n }\n return out;\n};\nconst text = (finding) => collectStrings(finding, []).join(' ').toLowerCase();\n\nconst fileOf = (finding) =>\n String(finding.file ?? finding.path ?? finding.filename ?? finding.location ?? '')\n .replace(/\\\\/g, '/')\n .replace(/^\\.\\//, '')\n .replace(/^[ab]\\//, '')\n .replace(/\\s*\\([^()]*\\)\\s*$/, '')\n .replace(/[:#]\\s*L?\\d+(?:\\s*-\\s*L?\\d+)?\\s*$/, '')\n .trim();\n\n// Basenames are unique across every seeded file, so a bare `payment.js`\n// citation resolves the same as `src/payment.js`.\nconst onFile = (path) => {\n const base = path.slice(path.lastIndexOf('/') + 1);\n const full = new RegExp('(^|/)' + path.replace(/\\./g, '\\\\.') + '$');\n const bare = new RegExp('^' + base.replace(/\\./g, '\\\\.') + '$');\n return findings.filter((finding) => {\n const cited = fileOf(finding);\n return full.test(cited) || bare.test(cited);\n });\n};\n\nconst REQUIRED = [\n {\n file: 'src/payment.js',\n label: 'the customer and charge queries build SQL by string concatenation from req.body',\n match: (t) =>\n /injection|sqli\\b|parameteri|parametri|placeholder|prepared statement|\\bbind\\b|\\bbinding\\b|bound parameter|concatenat|interpolat|escap|sanitiz|quoting/.test(\n t,\n ) ||\n (/\\bsql\\b|\\bquery\\b|\\bqueries\\b|statement/.test(t) &&\n /untrusted|user[- ](?:input|supplied|controlled|data)|req\\.body|request (?:value|data|input|body)|unsafe|unsanitiz|unvalidated|string (?:concat|building|built)|built from|splic/.test(\n t,\n )),\n },\n {\n file: 'src/pagination.js',\n label: 'pageOf is off by one on its slice end bound, so pages overlap by a row',\n match: (t) =>\n /off.?by.?one|\\bslice\\b|\\bbound(?:ary|aries|s)?\\b|half.?open|exclusive|inclusive|overlap|duplicat|repeat|reappear|one too many|one (?:extra|more)|too many|extra (?:item|row|record|element|entry|result)|\\+\\s?1\\b|pagesize \\+|\\+ pagesize|increment|\\b26\\b|end (?:index|bound|offset|position)|last (?:item|row|record|element)/.test(\n t,\n ),\n },\n {\n file: 'src/settings.js',\n label: 'loadSettings swallows the parse error and still reports ok: true',\n match: (t) =>\n /swallow|silent|suppress|\\bmask|\\bhid(?:e|es|den|ing)|\\bignor|empty catch|catch block|\\bok\\b|\\btrue\\b|success|succeed|re-?throw|rethrow|propagat|bubble|surfac|indistinguishable|no indication|cannot tell|unaware|misleading|misreport|false (?:success|positive)|as if|pretend/.test(\n t,\n ),\n },\n {\n file: 'src/export.js',\n label: 'exportCharges leaks the file handle when the query or a write throws',\n match: (t) =>\n /leak|\\bclos(?:e|es|ed|ing)\\b|\\bfinally\\b|clean.?up|releas|dispos|\\bhandle\\b|descriptor|\\bfd\\b|resource|left open|remains open|stays open|still open|dangling|exhaust|unclosed/.test(\n t,\n ),\n },\n];\n\nconst missing = REQUIRED.filter(\n (required) => !onFile(required.file).some((finding) => required.match(text(finding))),\n);\nif (missing.length > 0) {\n const reported = findings.map((finding) => fileOf(finding) || '?').join(', ') || '(nothing)';\n fail(\n 'RECALL SHORTFALL - the change set holds four separate defects and these were not reported: ' +\n missing.map((entry) => entry.file + ' (' + entry.label + ')').join('; ') +\n '. Files carrying a reported finding: ' +\n reported +\n '. Re-walk changes.diff hunk by hunk; every defect above is visible in the added lines.',\n );\n}\n\n// Baited smells. Each of these three files carries a construct that looks\n// like a common lint smell and is documented house style in\n// docs/review-conventions.md. Only a finding that raises THAT construct\n// is a precision failure; any other observation about the file is fair\n// review work and passes.\nconst BAITED = [\n {\n file: 'src/customer.js',\n smell: 'loose equality',\n matches: (t) =>\n /[=!]==?|loose equal|strict equal|\\bequality\\b|double equals|triple equals|coerc|nullish|\\?\\?/.test(\n t,\n ),\n why: '`== null` is the documented house idiom for a null-or-undefined check (docs/review-conventions.md) and is not to be tightened',\n },\n {\n file: 'src/audit.js',\n smell: 'floating / unawaited promise',\n matches: (t) =>\n /floating|un-?awaited|not awaited|missing await|no await|without await|should (?:be )?await|add (?:an )?await|\\bawait\\b|fire.?and.?forget|dangling promise|unhandled (?:promise )?rejection/.test(\n t,\n ) &&\n // A reviewer who spots that a SYNCHRONOUS throw from insertAudit\n // still escapes into the charge path has found something real, not\n // the bait. That finding is welcome.\n !/synchronous|synchronously|thenable|non-?promise|not (?:return )?a promise|does not return|returns undefined|before the promise/.test(\n t,\n ),\n why: 'the audit write is deliberately not awaited and carries a .catch, which is the required form for a best-effort write (docs/review-conventions.md)',\n },\n {\n file: 'src/errors.js',\n smell: 'unused parameters',\n matches: (t) =>\n /unused|not used|never used|un-?referenced|no-unused-vars|underscore|_req|_next|dead (?:parameter|argument)|(?:remove|drop|omit) (?:the )?(?:`?req`?|`?next`?|parameter|argument)|unnecessary (?:parameter|argument)/.test(\n t,\n ),\n why: \"Express registers error middleware by arity, so the four-parameter signature and the parameters it does not use are required (docs/review-conventions.md)\",\n },\n];\n\n// Not in manifest.json's seven-file list and not in changes.diff. The\n// book says to discuss only files in the change set, so any finding here\n// is out of scope whatever it says.\nconst OUT_OF_SCOPE = [\n { file: 'src/util.js', why: 'it is not one of the seven files in manifest.json' },\n { file: 'src/logger.js', why: 'it is not one of the seven files in manifest.json' },\n];\n\nconst nameOf = (finding) =>\n String(finding.category ?? finding.finding ?? finding.title ?? 'unnamed');\n\nconst wrong = [];\nfor (const bait of BAITED) {\n for (const finding of onFile(bait.file)) {\n if (!bait.matches(text(finding))) continue;\n wrong.push(bait.file + ' (' + nameOf(finding) + ' - ' + bait.smell + ') - ' + bait.why);\n }\n}\nfor (const out of OUT_OF_SCOPE) {\n for (const finding of onFile(out.file)) {\n wrong.push(out.file + ' (' + nameOf(finding) + ') - ' + out.why);\n }\n}\nif (wrong.length > 0) {\n fail(\n 'PRECISION SHORTFALL - drop these records from ' +\n FINDINGS +\n ': ' +\n wrong.join('; ') +\n '. A review that raises everything is not usable. Re-read docs/review-conventions.md, ' +\n 'which names the shapes that are deliberate house style, and re-read manifest.json for ' +\n 'which files are actually under review. Discussing a construct in report.md and clearing ' +\n 'it is fine - listing it as a finding is not. Any OTHER defect you find in these files is ' +\n 'welcome; only the constructs named above are off limits.',\n );\n}\n\nconsole.log(\n 'CODEREVIEW_ORACLE ok - all 4 seeded defects reported (src/payment.js, src/pagination.js, ' +\n 'src/settings.js, src/export.js) and none of the 3 baited house-style smells or 2 ' +\n 'out-of-scope files carries a finding.',\n);\n",
|
|
79
|
+
"surface": "harness",
|
|
80
|
+
"modelInput": false
|
|
81
|
+
}
|
|
82
|
+
]
|
|
83
|
+
},
|
|
84
|
+
"mocks": [],
|
|
85
|
+
"success": {
|
|
86
|
+
"summary": "reviews/rev-eval-1/report.md exists in the artifacts drawer with Summary/Findings/Verdict sections, a severity table citing the flawed files, and a request-changes verdict; review-output/findings.json holds one record per finding carrying at least file, line, severity, category and recommendation (extra context such as a title or a cwe is welcome), lists every seeded defect (unparameterized SQL, the off-by-one page bound, the swallowed settings error, the leaked export file handle), repeats none of the three documented house-style smells, and raises nothing against the two files outside the change set.",
|
|
87
|
+
"deliverables": [
|
|
88
|
+
{
|
|
89
|
+
"path": "reviews/rev-eval-1/report.md",
|
|
90
|
+
"kind": "markdown-report",
|
|
91
|
+
"artifact": true,
|
|
92
|
+
"minBytes": 400,
|
|
93
|
+
"checks": [
|
|
94
|
+
{
|
|
95
|
+
"kind": "contains",
|
|
96
|
+
"file": "reviews/rev-eval-1/report.md",
|
|
97
|
+
"pattern": "##\\s+Summary",
|
|
98
|
+
"label": "Summary section",
|
|
99
|
+
"artifact": true
|
|
100
|
+
},
|
|
101
|
+
{
|
|
102
|
+
"kind": "contains",
|
|
103
|
+
"file": "reviews/rev-eval-1/report.md",
|
|
104
|
+
"pattern": "##\\s+Findings",
|
|
105
|
+
"label": "Findings section",
|
|
106
|
+
"artifact": true
|
|
107
|
+
},
|
|
108
|
+
{
|
|
109
|
+
"kind": "contains",
|
|
110
|
+
"file": "reviews/rev-eval-1/report.md",
|
|
111
|
+
"pattern": "Verdict:\\s*request-changes",
|
|
112
|
+
"label": "request-changes verdict (SQL concat is at least major)",
|
|
113
|
+
"artifact": true
|
|
114
|
+
},
|
|
115
|
+
{
|
|
116
|
+
"kind": "contains",
|
|
117
|
+
"file": "reviews/rev-eval-1/report.md",
|
|
118
|
+
"pattern": "src/payment\\.js",
|
|
119
|
+
"label": "cites the flawed file",
|
|
120
|
+
"artifact": true
|
|
121
|
+
},
|
|
122
|
+
{
|
|
123
|
+
"kind": "tableShape",
|
|
124
|
+
"file": "reviews/rev-eval-1/report.md",
|
|
125
|
+
"requiredColumns": [
|
|
126
|
+
"Severity",
|
|
127
|
+
"File",
|
|
128
|
+
"Finding"
|
|
129
|
+
],
|
|
130
|
+
"minRows": 1,
|
|
131
|
+
"artifact": true
|
|
132
|
+
}
|
|
133
|
+
]
|
|
134
|
+
},
|
|
135
|
+
{
|
|
136
|
+
"path": "review-output/findings.json",
|
|
137
|
+
"kind": "json",
|
|
138
|
+
"minBytes": 300,
|
|
139
|
+
"checks": [
|
|
140
|
+
{
|
|
141
|
+
"kind": "sniff",
|
|
142
|
+
"file": "review-output/findings.json",
|
|
143
|
+
"sniff": "json-valid"
|
|
144
|
+
},
|
|
145
|
+
{
|
|
146
|
+
"kind": "recordSchema",
|
|
147
|
+
"file": "review-output/findings.json",
|
|
148
|
+
"format": "json",
|
|
149
|
+
"minRows": 4,
|
|
150
|
+
"fields": [
|
|
151
|
+
{
|
|
152
|
+
"name": "file",
|
|
153
|
+
"type": "nonempty",
|
|
154
|
+
"required": true
|
|
155
|
+
},
|
|
156
|
+
{
|
|
157
|
+
"name": "line",
|
|
158
|
+
"type": "^\\d+(\\s*-\\s*\\d+)?$",
|
|
159
|
+
"required": true
|
|
160
|
+
},
|
|
161
|
+
{
|
|
162
|
+
"name": "severity",
|
|
163
|
+
"type": "^([Cc]ritical|[Mm]ajor|[Mm]inor|[Nn]it|CRITICAL|MAJOR|MINOR|NIT)$",
|
|
164
|
+
"required": true
|
|
165
|
+
},
|
|
166
|
+
{
|
|
167
|
+
"name": "category",
|
|
168
|
+
"type": "nonempty",
|
|
169
|
+
"required": true
|
|
170
|
+
},
|
|
171
|
+
{
|
|
172
|
+
"name": "recommendation",
|
|
173
|
+
"type": "nonempty",
|
|
174
|
+
"required": true
|
|
175
|
+
}
|
|
176
|
+
],
|
|
177
|
+
"allowExtraFields": true
|
|
178
|
+
},
|
|
179
|
+
{
|
|
180
|
+
"kind": "nodeScriptPasses",
|
|
181
|
+
"script": "tests/code-review-oracle.mjs",
|
|
182
|
+
"timeoutMs": 30000,
|
|
183
|
+
"requiredOutput": [
|
|
184
|
+
{
|
|
185
|
+
"pattern": "CODEREVIEW_ORACLE ok",
|
|
186
|
+
"label": "every seeded defect reported, and no correct-as-written construct reported"
|
|
187
|
+
}
|
|
188
|
+
]
|
|
189
|
+
}
|
|
190
|
+
]
|
|
191
|
+
}
|
|
192
|
+
],
|
|
193
|
+
"taskNotes": {
|
|
194
|
+
"requireCraftbookTask": true,
|
|
195
|
+
"checks": [
|
|
196
|
+
{
|
|
197
|
+
"kind": "contains",
|
|
198
|
+
"file": "task-notes.md",
|
|
199
|
+
"pattern": "Verdict:",
|
|
200
|
+
"flags": "i"
|
|
201
|
+
}
|
|
202
|
+
]
|
|
203
|
+
},
|
|
204
|
+
"unchangedFixtures": [
|
|
205
|
+
"src/payment.js",
|
|
206
|
+
"src/util.js",
|
|
207
|
+
"src/pagination.js",
|
|
208
|
+
"src/settings.js",
|
|
209
|
+
"src/export.js",
|
|
210
|
+
"src/customer.js",
|
|
211
|
+
"src/audit.js",
|
|
212
|
+
"src/errors.js",
|
|
213
|
+
"src/logger.js",
|
|
214
|
+
"docs/review-conventions.md"
|
|
215
|
+
]
|
|
216
|
+
},
|
|
217
|
+
"rubric": {
|
|
218
|
+
"artifact": {
|
|
219
|
+
"path": "reviews/rev-eval-1/report.md",
|
|
220
|
+
"kind": "markdown"
|
|
221
|
+
},
|
|
222
|
+
"axes": [
|
|
223
|
+
{
|
|
224
|
+
"name": "grounding",
|
|
225
|
+
"description": "Findings reference code that actually appears in changes.diff; nothing is invented and cited lines exist."
|
|
226
|
+
},
|
|
227
|
+
{
|
|
228
|
+
"name": "precision",
|
|
229
|
+
"description": "The three documented house-style constructs (the `== null` check, the unawaited best-effort audit write with its .catch, the four-parameter Express error handler) are not raised as findings, and the two files outside the change set (src/util.js, src/logger.js) carry no findings at all; if a house-style construct is discussed it is to clear it. A different, real defect found in one of those three files is good review work and counts in its favour, not against it."
|
|
230
|
+
},
|
|
231
|
+
{
|
|
232
|
+
"name": "severity-judgment",
|
|
233
|
+
"description": "The SQL string concatenation, the off-by-one page bound, the swallowed settings error and the leaked export file handle are rated critical or major; the console.log and TODO are minor or nit - severities proportionate to impact."
|
|
234
|
+
},
|
|
235
|
+
{
|
|
236
|
+
"name": "actionability",
|
|
237
|
+
"description": "Every finding carries a concrete, minimal recommendation (e.g. return to parameterized queries, close the handle in a finally), not generic advice."
|
|
238
|
+
},
|
|
239
|
+
{
|
|
240
|
+
"name": "verdict-consistency",
|
|
241
|
+
"description": "The verdict follows mechanically from the findings table: any critical or major finding yields request-changes."
|
|
242
|
+
}
|
|
243
|
+
],
|
|
244
|
+
"contextNote": "The report is an artifacts-drawer file and review-output/findings.json is the single workspace file the reviewer writes - no source file may change. Some of the reviewed code only looks wrong: docs/review-conventions.md records which shapes are deliberate house style. src/util.js and src/logger.js are in the workspace as context but are not in the change set - manifest.json's seven files are."
|
|
245
|
+
},
|
|
246
|
+
"qualityFocus": [
|
|
247
|
+
"evidence-grounded review",
|
|
248
|
+
"review precision",
|
|
249
|
+
"file citations",
|
|
250
|
+
"artifact deliverable discipline"
|
|
251
|
+
]
|
|
252
|
+
}
|
|
@@ -0,0 +1,311 @@
|
|
|
1
|
+
{
|
|
2
|
+
"id": "content-accuracy-review",
|
|
3
|
+
"name": "Content Accuracy Review",
|
|
4
|
+
"description": "Fact-check a piece of content for accuracy and produce a claim-by-claim verification report. First scopes the content and extracts every checkable claim (statistics, dates, quotes, attributions, causal and factual statements) and locks a verification standard, then verifies each claim against authoritative sources rating it true/false/misleading/unverifiable with the supporting evidence, then writes a report flagging every inaccuracy with the correction and source plus an overall trust rating. Use this for a fact-check, accuracy review, content verification, checking claims and statistics, source-checking an article, or vetting content before publishing — it returns a claim-level verdict backed by cited sources rather than a vague 'seems accurate'.\n\nIt works on the content you choose when you launch it: a file in this project, or one you pick from your computer.\n\nA gallery craftbook generated from an archetype spec. It runs\n`phase → (per-phase gate) → … → evaluate → (loop) → finish`. Each build\nphase that produces a checkable artifact is followed by a **runtime\ngate-checkpoint** — the runtime verifies the artifact and routes with no\nmodel turn, looping back to redo the phase on a miss. The final `evaluate`\nstep holds a static deliverable gate plus a reviewer QA pass. What it adds\nover the generic `build-loop`: a specialist role per phase, a\ndomain-correct ordering, and a concrete per-phase quality bar.\n\nDeliverables marked \"artifact\" land in the project's artifacts drawer (`write_artifact` / `read_artifact`), not the shipped workspace — review output is not product source.\n\nPhases:\n\n1. Scope + extract claims (reviewer) — extract every checkable claim, lock the standard → gated on artifact `{{workPath}}/scope.md` (markdown-notes)\n2. Verify each claim (researcher) — check claims against sources, rate each → gated on artifact `{{workPath}}/verify.md` (markdown-notes)\n3. Write the accuracy report (reviewer) — claim-by-claim verdicts + corrections → gated on artifact `{{workPath}}/accuracy-review.md` (markdown-report)\n\nSource discipline:\n\n- Extract only checkable claims that appear in the supplied content.\n- Verify against the supplied or fetched sources; never create extra claims, companies, metrics, quotes, or source sections to make the report longer.\n- Preserve enough of the original claim text that a reader can trace each verdict back to the draft.\n- Mark unsupported but plausible claims as `UNVERIFIABLE` or `FALSE` according to the evidence, not as true by assumption.\n\nThe gates never advance with an unmet criterion, and loop back to the\nowning phase to fix named gaps.\n",
|
|
5
|
+
"entryStepId": "scope",
|
|
6
|
+
"triggers": [
|
|
7
|
+
"fact check this",
|
|
8
|
+
"accuracy review",
|
|
9
|
+
"verify these claims",
|
|
10
|
+
"is this content accurate",
|
|
11
|
+
"source-check this article",
|
|
12
|
+
"vet this before publishing"
|
|
13
|
+
],
|
|
14
|
+
"paramSchema": {
|
|
15
|
+
"type": "object",
|
|
16
|
+
"required": [
|
|
17
|
+
"source"
|
|
18
|
+
],
|
|
19
|
+
"properties": {
|
|
20
|
+
"source": {
|
|
21
|
+
"type": "string",
|
|
22
|
+
"title": "Content to review",
|
|
23
|
+
"description": "The article, post, or document whose claims should be checked. A file in this project, or one you pick from your computer.",
|
|
24
|
+
"input": {
|
|
25
|
+
"kind": "file",
|
|
26
|
+
"accept": [
|
|
27
|
+
".md",
|
|
28
|
+
".markdown",
|
|
29
|
+
".txt",
|
|
30
|
+
".html",
|
|
31
|
+
".htm",
|
|
32
|
+
".docx",
|
|
33
|
+
".pdf",
|
|
34
|
+
".vtt",
|
|
35
|
+
".srt"
|
|
36
|
+
]
|
|
37
|
+
}
|
|
38
|
+
},
|
|
39
|
+
"workPath": {
|
|
40
|
+
"type": "string",
|
|
41
|
+
"title": "Working folder",
|
|
42
|
+
"description": "Per-task working folder in the artifacts drawer. Defaults to this task's own folder so runs never collide; override with a stable name when you deliberately want runs to share files.",
|
|
43
|
+
"default": "{{task.dir}}"
|
|
44
|
+
}
|
|
45
|
+
}
|
|
46
|
+
},
|
|
47
|
+
"steps": [
|
|
48
|
+
{
|
|
49
|
+
"id": "scope",
|
|
50
|
+
"name": "Scope + extract claims",
|
|
51
|
+
"description": "extract every checkable claim, lock the standard",
|
|
52
|
+
"prompt": "Step 1: Read the content under review (the `source` input) end to end and EXTRACT every checkable claim into a numbered list — statistics and numbers, dates, named quotes and their attributions, causal statements ('X causes Y'), definitions, and any assertion presented as fact. Step 2: For each claim, note what kind it is and what evidence would settle it. Step 3: Separate verifiable facts from opinion/analysis (opinions are out of scope for accuracy but note unsupported opinions stated as fact). Step 4: Write the acceptance-criteria checklist / verification standard, e.g. 'every factual claim is extracted', 'each claim is checked against at least one authoritative source', 'a claim with no reliable source is marked unverifiable, not assumed true', 'corrections cite their source'. Step 5: Note the source quality bar (primary/official sources over secondary). Write the numbered claim list + standard to `write_task_note` AND `{{workPath}}/scope.md`. No verification yet.\n\nThe deliverable `{{workPath}}/scope.md` lands in the project's artifacts drawer — write it with `write_artifact` and read it back with `read_artifact`; the shipped workspace stays untouched.",
|
|
53
|
+
"suggestedRole": "reviewer",
|
|
54
|
+
"advanceWhen": {
|
|
55
|
+
"file": "{{workPath}}/scope.md",
|
|
56
|
+
"minBytes": 1,
|
|
57
|
+
"sniff": "nonempty",
|
|
58
|
+
"artifact": true
|
|
59
|
+
},
|
|
60
|
+
"gate": {
|
|
61
|
+
"at": "completion",
|
|
62
|
+
"checks": [
|
|
63
|
+
{
|
|
64
|
+
"kind": "minBytes",
|
|
65
|
+
"file": "{{workPath}}/scope.md",
|
|
66
|
+
"bytes": 120,
|
|
67
|
+
"artifact": true
|
|
68
|
+
},
|
|
69
|
+
{
|
|
70
|
+
"kind": "sniff",
|
|
71
|
+
"file": "{{workPath}}/scope.md",
|
|
72
|
+
"sniff": "nonempty",
|
|
73
|
+
"artifact": true
|
|
74
|
+
}
|
|
75
|
+
],
|
|
76
|
+
"onReject": "scope",
|
|
77
|
+
"maxAttempts": 3
|
|
78
|
+
},
|
|
79
|
+
"next": "verify",
|
|
80
|
+
"toolPolicy": {
|
|
81
|
+
"disallowBuiltinToolsets": [
|
|
82
|
+
"ai-apps",
|
|
83
|
+
"archives",
|
|
84
|
+
"audio",
|
|
85
|
+
"browser-automation",
|
|
86
|
+
"code-execution",
|
|
87
|
+
"craftbooks",
|
|
88
|
+
"data-tables",
|
|
89
|
+
"entity-intel",
|
|
90
|
+
"git",
|
|
91
|
+
"image-intel",
|
|
92
|
+
"images",
|
|
93
|
+
"role-delegation",
|
|
94
|
+
"role-delegation-escalation",
|
|
95
|
+
"security-intel",
|
|
96
|
+
"team-management",
|
|
97
|
+
"videos",
|
|
98
|
+
"web",
|
|
99
|
+
"workspace-fs-write"
|
|
100
|
+
],
|
|
101
|
+
"outputMedium": "artifact",
|
|
102
|
+
"additionalOutputMedia": [
|
|
103
|
+
"task-note"
|
|
104
|
+
]
|
|
105
|
+
}
|
|
106
|
+
},
|
|
107
|
+
{
|
|
108
|
+
"id": "verify",
|
|
109
|
+
"name": "Verify each claim",
|
|
110
|
+
"description": "check claims against sources, rate each",
|
|
111
|
+
"prompt": "Step 1: For EACH numbered claim, find authoritative evidence (use web search/fetch if available, prioritizing primary and official sources; otherwise reason from well-established knowledge and clearly mark anything you cannot externally confirm). Step 2: Rate each claim: TRUE (supported), FALSE (contradicted), MISLEADING (technically true but framed to deceive, or missing critical context), or UNVERIFIABLE (no reliable source found). Step 3: For every claim capture: the rating, the supporting source(s) with title/URL, a one-line evidence note, and — for false/misleading — the CORRECT statement. Step 4: Be adversarial: check whether a statistic is current, whether a quote is accurate and in context, and whether an attribution is right. Step 5: Do NOT mark a claim true just because it is plausible — require evidence, and prefer 'unverifiable' to a guess. Good looks like: each claim with a verdict and a citable source. Stage the verified claim table to `write_task_note`. Do not write the report yet.\n\nThe deliverable `{{workPath}}/verify.md` lands in the project's artifacts drawer — write it with `write_artifact` and read it back with `read_artifact`; the shipped workspace stays untouched.",
|
|
112
|
+
"suggestedRole": "researcher",
|
|
113
|
+
"advanceWhen": {
|
|
114
|
+
"file": "{{workPath}}/verify.md",
|
|
115
|
+
"minBytes": 1,
|
|
116
|
+
"sniff": "nonempty",
|
|
117
|
+
"artifact": true
|
|
118
|
+
},
|
|
119
|
+
"gate": {
|
|
120
|
+
"at": "completion",
|
|
121
|
+
"checks": [
|
|
122
|
+
{
|
|
123
|
+
"kind": "minBytes",
|
|
124
|
+
"file": "{{workPath}}/verify.md",
|
|
125
|
+
"bytes": 120,
|
|
126
|
+
"artifact": true
|
|
127
|
+
},
|
|
128
|
+
{
|
|
129
|
+
"kind": "sniff",
|
|
130
|
+
"file": "{{workPath}}/verify.md",
|
|
131
|
+
"sniff": "nonempty",
|
|
132
|
+
"artifact": true
|
|
133
|
+
}
|
|
134
|
+
],
|
|
135
|
+
"onReject": "verify",
|
|
136
|
+
"maxAttempts": 3
|
|
137
|
+
},
|
|
138
|
+
"next": "report",
|
|
139
|
+
"toolPolicy": {
|
|
140
|
+
"disallowBuiltinToolsets": [
|
|
141
|
+
"ai-apps",
|
|
142
|
+
"archives",
|
|
143
|
+
"audio",
|
|
144
|
+
"browser-automation",
|
|
145
|
+
"code-execution",
|
|
146
|
+
"craftbooks",
|
|
147
|
+
"data-tables",
|
|
148
|
+
"entity-intel",
|
|
149
|
+
"git",
|
|
150
|
+
"image-intel",
|
|
151
|
+
"images",
|
|
152
|
+
"role-delegation",
|
|
153
|
+
"role-delegation-escalation",
|
|
154
|
+
"security-intel",
|
|
155
|
+
"team-management",
|
|
156
|
+
"videos",
|
|
157
|
+
"workspace-fs-write"
|
|
158
|
+
],
|
|
159
|
+
"outputMedium": "artifact",
|
|
160
|
+
"additionalOutputMedia": [
|
|
161
|
+
"task-note"
|
|
162
|
+
]
|
|
163
|
+
},
|
|
164
|
+
"consumes": [
|
|
165
|
+
{
|
|
166
|
+
"file": "{{workPath}}/scope.md",
|
|
167
|
+
"artifact": true
|
|
168
|
+
}
|
|
169
|
+
]
|
|
170
|
+
},
|
|
171
|
+
{
|
|
172
|
+
"id": "report",
|
|
173
|
+
"name": "Write the accuracy report",
|
|
174
|
+
"description": "claim-by-claim verdicts + corrections",
|
|
175
|
+
"prompt": "Step 1: Open `{{workPath}}/accuracy-review.md` and write an overall trust rating (e.g. Accurate / Mostly accurate with fixes / Significant inaccuracies) with the count of false/misleading/unverifiable claims. Step 2: Add an '## Issues' section listing every FALSE, MISLEADING, and UNVERIFIABLE claim first: the original claim, the verdict, the correct statement, and the cited source. Step 3: Add a '## Verified claims' section confirming the claims that checked out (with sources), so the review's rigor is visible. Step 4: Add a '## Sources' list of everything consulted. Step 5: Note any claim that needs a subject-matter expert. Every issue must include a correction and a source citation; never assert a correction without evidence. On a loop-back, fix only the named gaps. `write_task_note` the report path and the counts by verdict.\n\nThe deliverable `{{workPath}}/accuracy-review.md` lands in the project's artifacts drawer — write it with `write_artifact` and read it back with `read_artifact`; the shipped workspace stays untouched.",
|
|
176
|
+
"suggestedRole": "reviewer",
|
|
177
|
+
"advanceWhen": {
|
|
178
|
+
"file": "{{workPath}}/accuracy-review.md",
|
|
179
|
+
"minBytes": 1,
|
|
180
|
+
"sniff": "nonempty",
|
|
181
|
+
"artifact": true
|
|
182
|
+
},
|
|
183
|
+
"gate": {
|
|
184
|
+
"at": "completion",
|
|
185
|
+
"checks": [
|
|
186
|
+
{
|
|
187
|
+
"kind": "minBytes",
|
|
188
|
+
"file": "{{workPath}}/accuracy-review.md",
|
|
189
|
+
"bytes": 1500,
|
|
190
|
+
"artifact": true
|
|
191
|
+
},
|
|
192
|
+
{
|
|
193
|
+
"kind": "contains",
|
|
194
|
+
"file": "{{workPath}}/accuracy-review.md",
|
|
195
|
+
"pattern": "(?:^|\\n)#{1,3}\\s+\\S",
|
|
196
|
+
"flags": "i",
|
|
197
|
+
"label": "at least one markdown heading",
|
|
198
|
+
"artifact": true
|
|
199
|
+
}
|
|
200
|
+
],
|
|
201
|
+
"onReject": "report",
|
|
202
|
+
"maxAttempts": 4
|
|
203
|
+
},
|
|
204
|
+
"next": "evaluate",
|
|
205
|
+
"toolPolicy": {
|
|
206
|
+
"disallowBuiltinToolsets": [
|
|
207
|
+
"ai-apps",
|
|
208
|
+
"archives",
|
|
209
|
+
"audio",
|
|
210
|
+
"browser-automation",
|
|
211
|
+
"code-execution",
|
|
212
|
+
"craftbooks",
|
|
213
|
+
"data-tables",
|
|
214
|
+
"entity-intel",
|
|
215
|
+
"git",
|
|
216
|
+
"image-intel",
|
|
217
|
+
"images",
|
|
218
|
+
"role-delegation",
|
|
219
|
+
"role-delegation-escalation",
|
|
220
|
+
"security-intel",
|
|
221
|
+
"team-management",
|
|
222
|
+
"videos",
|
|
223
|
+
"web",
|
|
224
|
+
"workspace-fs-write"
|
|
225
|
+
],
|
|
226
|
+
"outputMedium": "artifact",
|
|
227
|
+
"additionalOutputMedia": [
|
|
228
|
+
"task-note"
|
|
229
|
+
]
|
|
230
|
+
},
|
|
231
|
+
"consumes": [
|
|
232
|
+
{
|
|
233
|
+
"file": "{{workPath}}/verify.md",
|
|
234
|
+
"artifact": true
|
|
235
|
+
}
|
|
236
|
+
]
|
|
237
|
+
},
|
|
238
|
+
{
|
|
239
|
+
"id": "evaluate",
|
|
240
|
+
"name": "Evaluate",
|
|
241
|
+
"description": "Grade the deliverable against every acceptance criterion. All pass → finish; any fail → loop back and fix the gap.",
|
|
242
|
+
"prompt": "Open `{{workPath}}/accuracy-review.md` and grade it against the scope standard. Check EACH criterion and write PASS/FAIL with a one-line reason: (1) every checkable claim from the content was extracted and addressed; (2) each claim has a verdict (true/false/misleading/unverifiable); (3) each false/misleading claim has a correction AND a cited source; (4) verified claims cite their sources; (5) unverifiable claims are marked as such rather than assumed true; (6) a sources list is present; (7) the overall trust rating matches the findings. If any criterion fails, name the exact gap and loop back.\n\nThe deliverable lives in the project's artifacts drawer — open `{{workPath}}/accuracy-review.md` with `read_artifact`, not `read_file`.\n\nThen route — this is the whole point of the loop:\n\n- **Every criterion PASSES →** call `advance_task_step({ ref, stepId: \"evaluate\", next: \"finish\" })`.\n- **Any criterion FAILS →** write the specific gaps to notes, then call `advance_task_step({ ref, stepId: \"evaluate\", next: \"report\" })` to loop back. The builder fixes exactly those gaps.\n\nNever route to `finish` while any criterion is unmet. The build phase's completion gate already blocked a grossly-incomplete deliverable; your job is the judgment an automated check cannot make (does it actually work, read well, look right). After ~3 unproductive loops, stop and report DONE_WITH_CONCERNS so the user can step in.",
|
|
243
|
+
"suggestedRole": "reviewer",
|
|
244
|
+
"consumes": [
|
|
245
|
+
{
|
|
246
|
+
"file": "{{workPath}}/accuracy-review.md",
|
|
247
|
+
"artifact": true
|
|
248
|
+
}
|
|
249
|
+
],
|
|
250
|
+
"next": "report",
|
|
251
|
+
"toolPolicy": {
|
|
252
|
+
"disallowBuiltinToolsets": [
|
|
253
|
+
"ai-apps",
|
|
254
|
+
"archives",
|
|
255
|
+
"audio",
|
|
256
|
+
"browser-automation",
|
|
257
|
+
"code-execution",
|
|
258
|
+
"craftbooks",
|
|
259
|
+
"data-tables",
|
|
260
|
+
"entity-intel",
|
|
261
|
+
"git",
|
|
262
|
+
"image-intel",
|
|
263
|
+
"images",
|
|
264
|
+
"role-delegation",
|
|
265
|
+
"role-delegation-escalation",
|
|
266
|
+
"security-intel",
|
|
267
|
+
"team-management",
|
|
268
|
+
"videos",
|
|
269
|
+
"web",
|
|
270
|
+
"workspace-fs-write"
|
|
271
|
+
],
|
|
272
|
+
"outputMedium": "task-note"
|
|
273
|
+
}
|
|
274
|
+
},
|
|
275
|
+
{
|
|
276
|
+
"id": "finish",
|
|
277
|
+
"name": "Finish",
|
|
278
|
+
"description": "All acceptance criteria met. Stamp a short summary and report DONE.",
|
|
279
|
+
"prompt": "Every acceptance criterion passed. Write a one-paragraph DONE summary to task notes via `write_task_note`: what was built, the deliverable path(s), and a one-line confirmation that each criterion is met. Then report DONE.",
|
|
280
|
+
"suggestedRole": "developer",
|
|
281
|
+
"terminal": true,
|
|
282
|
+
"toolPolicy": {
|
|
283
|
+
"disallowBuiltinToolsets": [
|
|
284
|
+
"ai-apps",
|
|
285
|
+
"archives",
|
|
286
|
+
"artifacts",
|
|
287
|
+
"audio",
|
|
288
|
+
"browser-automation",
|
|
289
|
+
"code-execution",
|
|
290
|
+
"craftbooks",
|
|
291
|
+
"data-tables",
|
|
292
|
+
"entity-intel",
|
|
293
|
+
"git",
|
|
294
|
+
"image-intel",
|
|
295
|
+
"images",
|
|
296
|
+
"role-delegation",
|
|
297
|
+
"role-delegation-escalation",
|
|
298
|
+
"security-intel",
|
|
299
|
+
"team-management",
|
|
300
|
+
"videos",
|
|
301
|
+
"web",
|
|
302
|
+
"workspace-fs-write"
|
|
303
|
+
],
|
|
304
|
+
"outputMedium": "task-note"
|
|
305
|
+
}
|
|
306
|
+
}
|
|
307
|
+
],
|
|
308
|
+
"version": "1.1.0",
|
|
309
|
+
"releasedAt": "2026-09-24T12:00:00Z",
|
|
310
|
+
"minGezelVersion": "1.26267"
|
|
311
|
+
}
|