fdeops 3.30.0 → 4.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +10 -2
- package/bin/fde.js +35 -17
- package/bin/lib/context.js +6 -0
- package/bin/lib/masking.js +26 -6
- package/bin/lib/memory.js +3 -3
- package/bin/lib/setup.js +115 -33
- package/mcp/fdeops-ingest/package.json +1 -1
- package/mcp/fdeops-ingest/server.js +7 -1
- package/package.json +1 -1
- package/plugin.json +1 -1
- package/skills/fde/SKILL.md +10 -8
- package/skills/fde/references/ai.md +20 -20
- package/skills/fde/references/eval-pack.md +2 -2
- package/skills/fde/references/plan.md +2 -2
- package/skills/fde/references/review.md +7 -7
- package/skills/fde/references/rollback.md +2 -2
- package/skills/fde/references/ship.md +28 -37
package/README.md
CHANGED
|
@@ -40,6 +40,16 @@ One chat. Name the client:
|
|
|
40
40
|
|
|
41
41
|
That creates `~/fde-engagements/client01/.fde/` on your laptop. Paste kickoff notes in the same thread. `@fde` picks what to check. You still decide. After a meeting you review what changed, new asks, open questions, and next actions. Correct the proposal, then confirm the update.
|
|
42
42
|
|
|
43
|
+
### Make it fit your work
|
|
44
|
+
|
|
45
|
+
Setup asks three short questions: how you work, what would help first, and what to mask before sharing context with your agent. Review your choices before saving; change them anytime.
|
|
46
|
+
|
|
47
|
+
```bash
|
|
48
|
+
npx fdeops@latest setup
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
[See the choices and custom masking options](docs/USAGE.md#make-fdeops-fit-your-day).
|
|
52
|
+
|
|
43
53
|
Open the engagement fieldbook:
|
|
44
54
|
|
|
45
55
|
```bash
|
|
@@ -56,8 +66,6 @@ Read-only HTML of the record - promised, measured, accepted, and evidence. Regen
|
|
|
56
66
|
/plugin install fdeops@fdeops
|
|
57
67
|
```
|
|
58
68
|
|
|
59
|
-
Make it fit your day: `fde setup` asks three choices for your daily overview, context size, and report masking. [First-use setup](docs/USAGE.md#make-fdeops-fit-your-day).
|
|
60
|
-
|
|
61
69
|
The plugin adds session hooks and the slash commands below. Skill-only installation does not add hooks. See the [installation guide](docs/install.md) for setup details.
|
|
62
70
|
|
|
63
71
|
</details>
|
package/bin/fde.js
CHANGED
|
@@ -98,9 +98,9 @@ function gitIsAncestor(eng, olderHash, newerHash) {
|
|
|
98
98
|
} catch (_) { return false }
|
|
99
99
|
}
|
|
100
100
|
|
|
101
|
-
function gitLogHash(eng, args) {
|
|
101
|
+
function gitLogHash(eng, args, format = '%H') {
|
|
102
102
|
try {
|
|
103
|
-
return execFileSync('git', ['log', '-1',
|
|
103
|
+
return execFileSync('git', ['log', '-1', `--format=${format}`, ...args], {
|
|
104
104
|
cwd: eng, encoding: 'utf8', stdio: ['ignore', 'pipe', 'ignore'], timeout: 15000,
|
|
105
105
|
}).trim()
|
|
106
106
|
} catch (_) { return '' }
|
|
@@ -821,7 +821,7 @@ function phaseLabel(phase) {
|
|
|
821
821
|
// colon) collapses to '' here, which callers treat as "nothing to show".
|
|
822
822
|
function firstLine(md, maxLen) {
|
|
823
823
|
for (const raw of md.split('\n')) {
|
|
824
|
-
const l = raw.trim()
|
|
824
|
+
const l = maskDisplay(raw).trim()
|
|
825
825
|
if (!l || /^#{1,6}\s/.test(l)) continue
|
|
826
826
|
const clean = maskDisplay(l.replace(/^\*\*[^*]+:\*\*\s*/, '').replace(/\*\*/g, '').replace(/^["']|["']$/g, '').trim())
|
|
827
827
|
if (!clean) continue
|
|
@@ -1041,7 +1041,7 @@ function parseSignalHistoryEntries(eng) {
|
|
|
1041
1041
|
}
|
|
1042
1042
|
|
|
1043
1043
|
function displayNameFromSignalText(text) {
|
|
1044
|
-
if (preferences.privacy === 'reports' && masking.mask(text) !== text) return 'Contact (identifier masked)'
|
|
1044
|
+
if ((preferences.privacy === 'reports' || !['dashboard', 'vault'].includes(process.argv[2])) && masking.mask(text) !== text) return 'Contact (identifier masked)'
|
|
1045
1045
|
const person = personFromSignalText(text)
|
|
1046
1046
|
if (person) return person
|
|
1047
1047
|
const t = String(text).trim()
|
|
@@ -1166,7 +1166,7 @@ function extractRisks(eng) {
|
|
|
1166
1166
|
// sides look like real numbers/percentages, deduped by value pair, capped at
|
|
1167
1167
|
// 4; nothing reliable found -> the widget stays empty, never an invented number.
|
|
1168
1168
|
function extractStats(eng) {
|
|
1169
|
-
const text = readClean(eng, 'delivery.md') + '\n' + readClean(eng, 'decisions.md')
|
|
1169
|
+
const text = maskDisplay(readClean(eng, 'delivery.md') + '\n' + readClean(eng, 'decisions.md'))
|
|
1170
1170
|
const patterns = [
|
|
1171
1171
|
/(\d+(?:\.\d+)?%)[^\n%]{0,40}?(?:→|->)[^\n%]{0,20}?(\d+(?:\.\d+)?%)/g, // "X% ... -> Y%"
|
|
1172
1172
|
/(\d+(?:\.\d+)?%)\s+to\s+(\d+(?:\.\d+)?%)/gi, // "X% to Y%"
|
|
@@ -1458,6 +1458,7 @@ function cmdResume(args) {
|
|
|
1458
1458
|
const risks = readClean(eng, 'risks.md')
|
|
1459
1459
|
process.stdout.write(maskedSections([
|
|
1460
1460
|
policy ? `CLIENT POLICY - trust-profile.md\n${policy}` : '',
|
|
1461
|
+
preferences.work ? `WORKING PREFERENCES: ${preferences.work}. Starting help: ${preferences.start}.\n${setup.nextStep(preferences)}\nThis is a starting preference, not a client fact; current instructions and engagement state take precedence.` : '',
|
|
1461
1462
|
`${intro}\n\nENGAGEMENT: ${eng}`,
|
|
1462
1463
|
success ? `CURRENT GOALS & ACCEPTANCE - success.md\n${success}` : '',
|
|
1463
1464
|
risks ? `OPEN RISKS - risks.md\n${extractRisks(eng).map(r => r.text).join('\n') || '(none recorded)'}` : '',
|
|
@@ -1575,7 +1576,7 @@ function cmdLog(args) {
|
|
|
1575
1576
|
if (hit && force) console.error(`warning: logging possible ${hit} (--force)`)
|
|
1576
1577
|
if (retire) {
|
|
1577
1578
|
const n = retireOpenRisks(eng, text)
|
|
1578
|
-
if (!n) { console.error(`no open risk matched ${JSON.stringify(text)}`); process.exit(1) }
|
|
1579
|
+
if (!n) { console.error(`no open risk matched ${JSON.stringify(masking.mask(text))}`); process.exit(1) }
|
|
1579
1580
|
const hash = commitMemory(eng, 'retire risk', { files: ['risks.md'] })
|
|
1580
1581
|
console.log(`retired ${n} risk(s) → risks.md${hash ? ` @${hash}` : ''}`)
|
|
1581
1582
|
return
|
|
@@ -1916,6 +1917,19 @@ function stripApprovedStamp(text) {
|
|
|
1916
1917
|
return String(text || '').replace(/\s*\[approved:\s*[^\]]+\]/i, '').trim()
|
|
1917
1918
|
}
|
|
1918
1919
|
|
|
1920
|
+
function proposedSigner(body) {
|
|
1921
|
+
// Unknown or tentative ownership is a note, never a fabricated sign-off.
|
|
1922
|
+
const sources = body.match(/\[source:[^\]]*\]/gi) || []
|
|
1923
|
+
const prose = body.replace(/\[source:[^\]]*\]/gi, '').trim()
|
|
1924
|
+
if (/\?|\b(?:maybe|might|whether|could|should|would|if|unless)\b/i.test(prose)) return ''
|
|
1925
|
+
// Check uncertainty before removing a suffix, then validate the actual
|
|
1926
|
+
// identity so "customer sponsor signs off" cannot become a named person.
|
|
1927
|
+
if (!require('./lib/value-ledger').acceptanceName(prose)) return ''
|
|
1928
|
+
const identity = prose.replace(/\s+signs?(?:\s+off)?\b.*$/i, '').trim()
|
|
1929
|
+
if (!require('./lib/value-ledger').acceptanceName(identity)) return ''
|
|
1930
|
+
return [identity, ...sources].join(' ')
|
|
1931
|
+
}
|
|
1932
|
+
|
|
1919
1933
|
// One screen a human can confirm in two minutes. The file-by-file routing
|
|
1920
1934
|
// still prints after this - agents edit prefixes; people read this.
|
|
1921
1935
|
function printDebriefReview(text, eng) {
|
|
@@ -1933,7 +1947,7 @@ function printDebriefReview(text, eng) {
|
|
|
1933
1947
|
if (type === 'decision') {
|
|
1934
1948
|
const who = approvedStamp(body)
|
|
1935
1949
|
const core = previewLine(stripApprovedStamp(body), 90)
|
|
1936
|
-
buckets.decided.push(who ? `${core} (approved ${who})` : `${core} (unconfirmed)`)
|
|
1950
|
+
buckets.decided.push(!hasSource(body) ? `${core} (CLAIM - source missing; unconfirmed)` : who ? `${core} (approved ${who}; source recorded)` : `${core} (unconfirmed; source recorded)`)
|
|
1937
1951
|
} else if (type === 'ask') {
|
|
1938
1952
|
buckets.asked.push(previewLine(body, 100))
|
|
1939
1953
|
} else if (type === 'scope') {
|
|
@@ -1945,7 +1959,8 @@ function printDebriefReview(text, eng) {
|
|
|
1945
1959
|
} else if (type === 'next') {
|
|
1946
1960
|
buckets.next.push(previewLine(body, 100))
|
|
1947
1961
|
} else if (type === 'signer') {
|
|
1948
|
-
buckets.signer.push(previewLine(body, 80))
|
|
1962
|
+
if (proposedSigner(body)) buckets.signer.push(previewLine(body, 80))
|
|
1963
|
+
else buckets.open.push(`signer authority missing or uncertain: ${previewLine(body, 80)}`)
|
|
1949
1964
|
}
|
|
1950
1965
|
}
|
|
1951
1966
|
console.log('REVIEW (one screen - confirm once, then apply)\n')
|
|
@@ -2005,7 +2020,7 @@ function changeReviewIssues(eng) {
|
|
|
2005
2020
|
const del = latestDeliveryEntry(readClean(eng, 'delivery.md'))
|
|
2006
2021
|
const dec = latestDatedDecision(readClean(eng, 'decisions.md'))
|
|
2007
2022
|
const delHash = del.line
|
|
2008
|
-
?
|
|
2023
|
+
? gitLogHash(eng, ['-S', del.line, '--', 'delivery.md'])
|
|
2009
2024
|
: ''
|
|
2010
2025
|
if (del.date && dec.date && dec.date > del.date && delHash) {
|
|
2011
2026
|
issues.push(
|
|
@@ -2168,7 +2183,8 @@ function routeDebriefInput(eng, input, { dry, force, sealed = [], allowReplay =
|
|
|
2168
2183
|
continue
|
|
2169
2184
|
}
|
|
2170
2185
|
if (type === 'signer') {
|
|
2171
|
-
const who =
|
|
2186
|
+
const who = proposedSigner(body)
|
|
2187
|
+
if (!who) { ctxLines.push(line); continue }
|
|
2172
2188
|
if (dry) {
|
|
2173
2189
|
console.log(`→ success.md **Stakeholder who signs off:** ${previewLine(who)}`)
|
|
2174
2190
|
console.log(`→ stakeholders.md ${previewLine(`- [${date}] ${who} signs off`)}`)
|
|
@@ -2261,6 +2277,7 @@ function runDebrief(args, eng) {
|
|
|
2261
2277
|
const applyIdx = args.indexOf('--apply')
|
|
2262
2278
|
const apply = applyIdx !== -1
|
|
2263
2279
|
if (apply) args.splice(applyIdx, 1)
|
|
2280
|
+
if (smart && apply) throw new Error('run --smart first, review the proposal, then confirm with --apply in a separate command')
|
|
2264
2281
|
let force = false
|
|
2265
2282
|
const forceIdx = args.indexOf('--force')
|
|
2266
2283
|
if (forceIdx !== -1) { force = true; args.splice(forceIdx, 1) }
|
|
@@ -2819,7 +2836,7 @@ function silentCommitIssues(eng) {
|
|
|
2819
2836
|
// Pickaxe on the entry text, not the file: a later status edit to
|
|
2820
2837
|
// delivery.md must not become the cutoff and hide commits before it.
|
|
2821
2838
|
const raw = entry.line
|
|
2822
|
-
?
|
|
2839
|
+
? gitLogHash(eng, ['-S', entry.line, '--', 'delivery.md'], '%cI')
|
|
2823
2840
|
: ''
|
|
2824
2841
|
// git --since is inclusive at second grain; a commit in the same second as
|
|
2825
2842
|
// the receipt is the receipt's own work, not a silent one.
|
|
@@ -3355,10 +3372,10 @@ function cmdRedact(args) {
|
|
|
3355
3372
|
})
|
|
3356
3373
|
}
|
|
3357
3374
|
if (!hits.length) {
|
|
3358
|
-
console.log(`redact: no lines contain ${JSON.stringify(term)}`)
|
|
3375
|
+
console.log(`redact: no lines contain ${JSON.stringify(masking.mask(term))}`)
|
|
3359
3376
|
return
|
|
3360
3377
|
}
|
|
3361
|
-
console.log(`REDACT - ${hits.length} matching line(s) for ${JSON.stringify(term)}`)
|
|
3378
|
+
console.log(`REDACT - ${hits.length} matching line(s) for ${JSON.stringify(masking.mask(term))}`)
|
|
3362
3379
|
hits.slice(0, 20).forEach(h => {
|
|
3363
3380
|
const preview = h.line.length > 100 ? masking.mask(h.line).slice(0, 97) + '…' : h.line
|
|
3364
3381
|
console.log(` ${h.file}:${h.lineNo} ${preview}`)
|
|
@@ -3758,13 +3775,13 @@ function cmdDashboard(args) {
|
|
|
3758
3775
|
].map(([f, title]) => [title, readClean(e.dir, f)])
|
|
3759
3776
|
.filter(([, md]) => render.hasRealContent(md))
|
|
3760
3777
|
.map(([title, md]) => ({ title, html: render.mdBlockHtml(maskDisplay(md), parseMdTable) }))
|
|
3761
|
-
e.searchBlob = render.escapeHtml([
|
|
3778
|
+
e.searchBlob = render.escapeHtml(maskDisplay([
|
|
3762
3779
|
e.name, e.next, e.lastSession, e.reality, e.brief,
|
|
3763
3780
|
...e.log.map(g => g.text), ...e.risks.map(r => r.text),
|
|
3764
3781
|
...e.stakeholders.map(p => `${p.name} ${p.role} ${p.note}`),
|
|
3765
3782
|
...e.moreSections.map(s => s.title),
|
|
3766
3783
|
...e.valueRows.map(r => `${r.slice} ${r.promised} ${r.measured} ${r.accepted} ${r.evidence}`),
|
|
3767
|
-
].join(' ').toLowerCase())
|
|
3784
|
+
].join(' ').toLowerCase()))
|
|
3768
3785
|
})
|
|
3769
3786
|
|
|
3770
3787
|
const html = render.buildFieldbookHtml({ engagements: maskReport(engagements), today, generatedAt: new Date().toISOString() })
|
|
@@ -4170,7 +4187,8 @@ let outputBudget
|
|
|
4170
4187
|
if (['resume', 'recall', 'handoff', 'defend'].includes(cmd) && !rawArgs.some(a => ['--full', '--init', '--bind', '--out'].includes(a))) {
|
|
4171
4188
|
try { outputBudget = context.budgetArgs(rawArgs, (setupStore.read() || setup.DEFAULTS).context === 'compact' ? 4096 : 16384).maxBytes } catch (_) {}
|
|
4172
4189
|
}
|
|
4173
|
-
require('./lib/masking').protectOutput(
|
|
4190
|
+
require('./lib/masking').protectOutput(cmd === 'setup' && rawArgs.length === 1 && rawArgs[0] === '--show'
|
|
4191
|
+
? require('./lib/masking').createMasking(ENGAGEMENTS_ROOT, { custom: false }) : masking, { maxBytes: outputBudget })
|
|
4174
4192
|
let args
|
|
4175
4193
|
try {
|
|
4176
4194
|
args = rawArgs.map(arg => masking.restore(arg))
|
|
@@ -4192,7 +4210,7 @@ switch (cmd) {
|
|
|
4192
4210
|
case 'setup': finishAsync(setup.command(setupStore, args)); break
|
|
4193
4211
|
case 'privacy':
|
|
4194
4212
|
if (args.length) { console.error('usage: fde privacy'); process.exitCode = 2; break }
|
|
4195
|
-
console.log(`FDEOps ${require('../package.json').version} - identifier masking enabled by default.\nCLI responses, smart proposals, handoff packets and ingest MCP results use local aliases.\nPatterns: common emails, international/US phones, SSN-shaped identifiers and supported credentials.\nNames and arbitrary sensitive prose are not detected; mark them <private
|
|
4213
|
+
console.log(`FDEOps ${require('../package.json').version} - identifier masking enabled by default.\nCLI responses, smart proposals, handoff packets and ingest MCP results use local aliases.\nPatterns: common emails, international/US phones, SSN-shaped identifiers and supported credentials.\nNames and arbitrary sensitive prose are not automatically detected; mark them <private> or supply local custom terms with fde setup.\nRaw files, pasted chat and upstream MCP content bypass this protection. Local reports retain identifiers by default; fde setup can also mask newly generated report content.`)
|
|
4196
4214
|
break
|
|
4197
4215
|
case 'demo': cmdDemo(args); break
|
|
4198
4216
|
case 'scan': cmdScan(); break
|
package/bin/lib/context.js
CHANGED
|
@@ -37,6 +37,12 @@ function boundedSections(sections, maxBytes = DEFAULT_BYTES) {
|
|
|
37
37
|
const available = maxBytes - Buffer.byteLength(out + footer) - 2
|
|
38
38
|
const share = Math.max(0, Math.floor(available / (filled.length - i)))
|
|
39
39
|
const text = filled[i]
|
|
40
|
+
// A truncation marker also consumes the budget. If the fair share cannot
|
|
41
|
+
// fit one, stop with one marker instead of overflowing for every section.
|
|
42
|
+
if (Buffer.byteLength(text) > share && share < Buffer.byteLength(OMITTED)) {
|
|
43
|
+
out += (out ? '\n\n' : '') + clipUtf8(text, Math.max(0, available - Buffer.byteLength(OMITTED))) + clipUtf8(OMITTED, Math.max(0, available))
|
|
44
|
+
break
|
|
45
|
+
}
|
|
40
46
|
const next = Buffer.byteLength(text) <= share ? text : clipUtf8(text, Math.max(0, share - Buffer.byteLength(OMITTED))) + OMITTED
|
|
41
47
|
out += (out ? '\n\n' : '') + next
|
|
42
48
|
}
|
package/bin/lib/masking.js
CHANGED
|
@@ -6,7 +6,7 @@ const fs = require('node:fs')
|
|
|
6
6
|
const path = require('node:path')
|
|
7
7
|
const crypto = require('node:crypto')
|
|
8
8
|
const { StringDecoder } = require('node:string_decoder')
|
|
9
|
-
const ALIAS = /\[\[(email|phone|identifier|credential):[a-f0-9]{16}\]\]/g
|
|
9
|
+
const ALIAS = /\[\[(email|phone|identifier|credential|term):[a-f0-9]{16}\]\]/g
|
|
10
10
|
const PATTERNS = [
|
|
11
11
|
['credential', /-----BEGIN (?:RSA |EC |OPENSSH )?PRIVATE KEY-----[\s\S]*?(?:-----END (?:RSA |EC |OPENSSH )?PRIVATE KEY-----|$)/g],
|
|
12
12
|
['credential', /\b(?:AKIA[0-9A-Z]{16}|ghp_[A-Za-z0-9]{20,}|github_pat_[A-Za-z0-9_]{20,}|sk-[A-Za-z0-9_-]{20,}|xox[baprs]-[A-Za-z0-9-]{10,})\b/g],
|
|
@@ -25,7 +25,27 @@ function replacements(text, replace) {
|
|
|
25
25
|
for (const [kind, pattern] of PATTERNS) out = out.replace(pattern, value => replace(kind, value))
|
|
26
26
|
return out
|
|
27
27
|
}
|
|
28
|
-
function createMasking(root) {
|
|
28
|
+
function createMasking(root, { custom = true } = {}) {
|
|
29
|
+
const settings = require('./setup').createSetup(root)
|
|
30
|
+
function replaceAll(text, replace) {
|
|
31
|
+
let terms = []
|
|
32
|
+
if (custom) {
|
|
33
|
+
const profile = settings.read()
|
|
34
|
+
if (profile && profile.masking === 'custom') terms = profile.terms
|
|
35
|
+
}
|
|
36
|
+
const escape = value => value.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')
|
|
37
|
+
const pattern = terms.length ? new RegExp(terms.slice().sort((a, b) => b.length - a.length).map(term =>
|
|
38
|
+
'(?<![\\p{L}\\p{N}_])' + escape(term) + '(?![\\p{L}\\p{N}_])').join('|'), 'giu') : null
|
|
39
|
+
// Existing aliases are protocol tokens, never input for further masking.
|
|
40
|
+
const pieces = String(text).split(/(\[\[(?:email|phone|identifier|credential|term):[a-f0-9]{16}\]\])/g)
|
|
41
|
+
return pieces.map((piece, i) => {
|
|
42
|
+
if (i % 2) return piece
|
|
43
|
+
const builtIn = replacements(piece, replace)
|
|
44
|
+
if (!pattern) return builtIn
|
|
45
|
+
return builtIn.split(/(\[\[(?:email|phone|identifier|credential|term):[a-f0-9]{16}\]\])/g)
|
|
46
|
+
.map((part, j) => j % 2 ? part : part.replace(pattern, value => replace('term', value))).join('')
|
|
47
|
+
}).join('')
|
|
48
|
+
}
|
|
29
49
|
const directory = path.join(root, '.privacy'), file = path.join(directory, 'identifiers.json')
|
|
30
50
|
function directoryReady(create) {
|
|
31
51
|
if (create) fs.mkdirSync(root, { recursive: true })
|
|
@@ -43,7 +63,7 @@ function createMasking(root) {
|
|
|
43
63
|
if (data.version !== 1 || !Array.isArray(data.entries) || data.entries.length > 50000) throw new Error(FAIL)
|
|
44
64
|
const seen = new Set()
|
|
45
65
|
for (const entry of data.entries) {
|
|
46
|
-
if (!entry || typeof entry.value !== 'string' || !/^\[\[(email|phone|identifier|credential):[a-f0-9]{16}\]\]$/.test(entry.alias) || seen.has(entry.alias)) throw new Error(FAIL)
|
|
66
|
+
if (!entry || typeof entry.value !== 'string' || !/^\[\[(email|phone|identifier|credential|term):[a-f0-9]{16}\]\]$/.test(entry.alias) || seen.has(entry.alias)) throw new Error(FAIL)
|
|
47
67
|
seen.add(entry.alias)
|
|
48
68
|
}
|
|
49
69
|
return data
|
|
@@ -76,11 +96,11 @@ function createMasking(root) {
|
|
|
76
96
|
function mask(text) {
|
|
77
97
|
text = String(text)
|
|
78
98
|
let detected = false
|
|
79
|
-
|
|
99
|
+
replaceAll(text, (_, value) => { detected = true; return value })
|
|
80
100
|
if (!detected) return text
|
|
81
101
|
return transact(true, entries => {
|
|
82
102
|
const values = new Map(entries.map(e => [e.value, e.alias]))
|
|
83
|
-
return
|
|
103
|
+
return replaceAll(text, (kind, value) => {
|
|
84
104
|
if (!values.has(value)) {
|
|
85
105
|
const alias = `[[${kind}:${crypto.randomBytes(8).toString('hex')}]]`
|
|
86
106
|
entries.push({ alias, value }); values.set(value, alias)
|
|
@@ -91,7 +111,7 @@ function createMasking(root) {
|
|
|
91
111
|
}
|
|
92
112
|
function restore(text) {
|
|
93
113
|
text = String(text)
|
|
94
|
-
if (/\[\[(email|phone|identifier|credential):/.test(text.replace(ALIAS, ''))) throw new Error(FAIL)
|
|
114
|
+
if (/\[\[(email|phone|identifier|credential|term):/.test(text.replace(ALIAS, ''))) throw new Error(FAIL)
|
|
95
115
|
if (!text.match(ALIAS)) return text
|
|
96
116
|
return transact(false, entries => {
|
|
97
117
|
const aliases = new Map(entries.map(e => [e.alias, e.value]))
|
package/bin/lib/memory.js
CHANGED
|
@@ -53,7 +53,7 @@ function createMemoryApi(deps) {
|
|
|
53
53
|
execFileSync('git', ['add', '--', f], { cwd: eng, stdio: 'ignore', timeout: 10000 })
|
|
54
54
|
}
|
|
55
55
|
} else {
|
|
56
|
-
execFileSync('git', ['add', '-A', '--', '.', ':(exclude).privacy'], { cwd: eng, stdio: 'ignore', timeout: 10000 })
|
|
56
|
+
execFileSync('git', ['add', '-A', '--', '.', ':(exclude).privacy', ':(exclude).preferences.json'], { cwd: eng, stdio: 'ignore', timeout: 10000 })
|
|
57
57
|
}
|
|
58
58
|
const porcelain = execFileSync('git', ['status', '--porcelain'], {
|
|
59
59
|
cwd: eng, encoding: 'utf8', timeout: 10000, stdio: ['ignore', 'pipe', 'ignore'],
|
|
@@ -73,7 +73,7 @@ function createMemoryApi(deps) {
|
|
|
73
73
|
}
|
|
74
74
|
}
|
|
75
75
|
// Even a previously staged alias dictionary must never enter a CLI commit.
|
|
76
|
-
const privateStaged = execFileSync('git', ['diff', '--cached', '--name-only'], { cwd: eng, encoding: 'utf8', timeout: 10000 }).split('\n').some(f => f.split('/').
|
|
76
|
+
const privateStaged = execFileSync('git', ['diff', '--cached', '--name-only'], { cwd: eng, encoding: 'utf8', timeout: 10000 }).split('\n').some(f => f.split('/').some(part => ['.privacy', '.preferences.json'].includes(part)))
|
|
77
77
|
if (privateStaged) throw new Error('private alias state must not be staged in engagement history')
|
|
78
78
|
const still = execFileSync('git', ['diff', '--cached', '--name-only'], {
|
|
79
79
|
cwd: eng, encoding: 'utf8', timeout: 10000, stdio: ['ignore', 'pipe', 'ignore'],
|
|
@@ -134,7 +134,7 @@ function createMemoryApi(deps) {
|
|
|
134
134
|
execFileSync('git', ['init'], { cwd: eng, stdio: 'ignore', timeout: 10000 })
|
|
135
135
|
atomicWriteFile(
|
|
136
136
|
path.join(eng, '.gitignore'),
|
|
137
|
-
['*.lock', '*.tmp', '.last-write', '.debrief-propose', '.debrief-private', '.debrief-seal', '.privacy/', ''].join('\n')
|
|
137
|
+
['*.lock', '*.tmp', '.last-write', '.debrief-propose', '.debrief-private', '.debrief-seal', '.privacy/', '.preferences.json', ''].join('\n')
|
|
138
138
|
)
|
|
139
139
|
const owner = writeOwnerIfMissing(eng)
|
|
140
140
|
configureMemoryGitIdentity(eng, owner)
|
package/bin/lib/setup.js
CHANGED
|
@@ -5,7 +5,7 @@ const crypto = require('node:crypto')
|
|
|
5
5
|
const paths = require('./install-paths')
|
|
6
6
|
|
|
7
7
|
const DEFAULTS = { view: 'current', context: 'standard', privacy: 'agent' }
|
|
8
|
-
const
|
|
8
|
+
const SETTINGS = [
|
|
9
9
|
{ key: 'view', title: 'What should your daily overview show?', options: [
|
|
10
10
|
['current', 'The client I am working on'], ['portfolio', 'All my clients'],
|
|
11
11
|
] },
|
|
@@ -17,14 +17,60 @@ const QUESTIONS = [
|
|
|
17
17
|
['reports', 'Agent context and newly generated Fieldbook/vault reports'],
|
|
18
18
|
] },
|
|
19
19
|
]
|
|
20
|
-
const
|
|
20
|
+
const QUESTIONS = [
|
|
21
|
+
{ key: 'work', title: 'How do you work?', options: [
|
|
22
|
+
['single', 'One client'], ['multiple', 'Several clients'], ['team', 'Leading a delivery team'],
|
|
23
|
+
] },
|
|
24
|
+
{ key: 'start', title: 'What would help you first?', options: [
|
|
25
|
+
['new', 'Starting an engagement'], ['daily', 'Continuing daily work'], ['takeover', 'Taking over existing work'],
|
|
26
|
+
] },
|
|
27
|
+
{ key: 'masking', title: 'What should FDEOps hide before sharing context with your agent?', options: [
|
|
28
|
+
['standard', 'Common identifiers and secrets'], ['custom', 'Those, plus names and terms I specify'],
|
|
29
|
+
] },
|
|
30
|
+
]
|
|
31
|
+
const PROFILE_DEFAULTS = { work: 'single', start: 'daily', masking: 'standard', terms: [] }
|
|
32
|
+
const LIMITS = 'Both choices hide marked private content. Masking reduces exposure; it does not guarantee anonymity or permission to share client data. Raw file tools, pasted chat and other tools can bypass it. AI-provider settings are unchanged.'
|
|
33
|
+
function validateTerms(terms) {
|
|
34
|
+
if (!Array.isArray(terms) || terms.length > 100 || terms.some(t => typeof t !== 'string' || t.length < 2 || t.length > 128 || t !== t.trim() || /[\r\n\x00-\x1f]|\[\[|\]\]/.test(t))) {
|
|
35
|
+
throw new Error('Use up to 100 names or terms, one per line, each 2-128 characters; control characters and alias markers are not allowed.')
|
|
36
|
+
}
|
|
37
|
+
return [...new Set(terms)]
|
|
38
|
+
}
|
|
21
39
|
function validate(value) {
|
|
40
|
+
const keys = value && Object.keys(value)
|
|
41
|
+
const personal = value && Object.hasOwn(value, 'work')
|
|
22
42
|
if (!value || typeof value !== 'object' || Array.isArray(value) ||
|
|
23
|
-
|
|
43
|
+
keys.length !== (personal ? 7 : 3) ||
|
|
44
|
+
(personal ? SETTINGS.concat(QUESTIONS) : SETTINGS).some(q => !q.options.some(([v]) => value[q.key] === v))) {
|
|
24
45
|
throw new Error('Invalid setup choices; run fde setup to see the supported options.')
|
|
25
46
|
}
|
|
47
|
+
if (personal) {
|
|
48
|
+
validateTerms(value.terms)
|
|
49
|
+
if (value.masking === 'custom' && !value.terms.length) throw new Error('Custom masking needs at least one name or term.')
|
|
50
|
+
}
|
|
26
51
|
return value
|
|
27
52
|
}
|
|
53
|
+
function readTerms(file) {
|
|
54
|
+
let fd
|
|
55
|
+
try {
|
|
56
|
+
paths.checkPath(file)
|
|
57
|
+
fd = fs.openSync(file, fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW | fs.constants.O_NONBLOCK)
|
|
58
|
+
const st = fs.fstatSync(fd)
|
|
59
|
+
if (!st.isFile() || st.nlink !== 1 || st.size > 65536) throw new Error('unsafe')
|
|
60
|
+
const terms = validateTerms(fs.readFileSync(fd, 'utf8').split(/\r?\n/).map(t => t.trim()).filter(Boolean))
|
|
61
|
+
if (!terms.length) throw new Error('empty')
|
|
62
|
+
return terms
|
|
63
|
+
} catch (_) { throw new Error('Cannot load custom terms. Use an ordinary local UTF-8 file with 1-100 terms, one per line, each 2-128 characters.') }
|
|
64
|
+
finally { if (fd !== undefined) fs.closeSync(fd) }
|
|
65
|
+
}
|
|
66
|
+
function nextStep(value) {
|
|
67
|
+
const start = {
|
|
68
|
+
new: 'Clarify the client problem, one measurable outcome, and who will accept it. Start with land.',
|
|
69
|
+
daily: 'Resume the current client and choose one action from the existing blockers. Start with triage.',
|
|
70
|
+
takeover: 'Check the existing evidence, open risks and previous commitments before changing anything. Start with audit.',
|
|
71
|
+
}
|
|
72
|
+
return (start[value.start] || '') + (value.work === 'team' ? ' Make responsibility and handoff clear; do not assume shared access or synchronization.' : '')
|
|
73
|
+
}
|
|
28
74
|
function createSetup(root) {
|
|
29
75
|
const file = path.join(root, '.preferences.json')
|
|
30
76
|
function read() {
|
|
@@ -33,7 +79,7 @@ function createSetup(root) {
|
|
|
33
79
|
paths.checkPath(file)
|
|
34
80
|
fd = fs.openSync(file, fs.constants.O_RDONLY | fs.constants.O_NOFOLLOW | fs.constants.O_NONBLOCK)
|
|
35
81
|
const st = fs.fstatSync(fd)
|
|
36
|
-
if (!st.isFile() || st.nlink !== 1 || st.size >
|
|
82
|
+
if (!st.isFile() || st.nlink !== 1 || st.size > 65536 || (process.platform !== 'win32' && (st.mode & 0o077))) throw new Error('unsafe')
|
|
37
83
|
return validate(JSON.parse(fs.readFileSync(fd, 'utf8')))
|
|
38
84
|
} catch (e) {
|
|
39
85
|
if (e.code === 'ENOENT') return null
|
|
@@ -53,57 +99,93 @@ function createSetup(root) {
|
|
|
53
99
|
}
|
|
54
100
|
return { read, save }
|
|
55
101
|
}
|
|
56
|
-
function describe(value) {
|
|
57
|
-
return
|
|
102
|
+
function describe(value, questions = QUESTIONS) {
|
|
103
|
+
return questions.map(q => `${q.title} ${q.options.find(([v]) => v === value[q.key])[1]}`).join('\n')
|
|
104
|
+
}
|
|
105
|
+
function saveAnswers(store, choices, questions, termsFile) {
|
|
106
|
+
const saved = store.read(), current = saved || DEFAULTS
|
|
107
|
+
if (questions.some(q => !q.options.some(([v]) => choices[q.key] === v)) || Object.keys(choices).length !== questions.length) throw new Error('Supply one valid answer for each question.')
|
|
108
|
+
let result
|
|
109
|
+
if (questions === SETTINGS) result = { ...current, ...choices }
|
|
110
|
+
else {
|
|
111
|
+
if (termsFile && choices.masking !== 'custom') throw new Error('A terms file requires custom masking.')
|
|
112
|
+
const terms = termsFile ? readTerms(termsFile) : current.terms || []
|
|
113
|
+
result = { ...current, ...choices, terms, view: !saved || (saved.work && saved.work !== choices.work) ? (choices.work === 'multiple' ? 'portfolio' : 'current') : saved.view }
|
|
114
|
+
}
|
|
115
|
+
store.save(result)
|
|
116
|
+
return result
|
|
58
117
|
}
|
|
59
118
|
function command(store, args) {
|
|
60
119
|
if (args.length === 1 && args[0] === '--show') {
|
|
61
|
-
const saved = store.read()
|
|
62
|
-
console.log(JSON.stringify({ configured: !!saved, ...(saved
|
|
120
|
+
const saved = store.read(), { terms, ...safe } = saved || DEFAULTS
|
|
121
|
+
console.log(JSON.stringify({ configured: !!saved, ...safe, ...(terms ? { termCount: terms.length, nextStep: nextStep(saved) } : {}) }))
|
|
63
122
|
return
|
|
64
123
|
}
|
|
65
|
-
if (args.length) {
|
|
66
|
-
const choices = {}
|
|
67
|
-
|
|
68
|
-
|
|
124
|
+
if (args.length && !(args.length === 1 && args[0] === '--settings')) {
|
|
125
|
+
const choices = {}, rest = args.filter(a => a !== '--save')
|
|
126
|
+
let termsFile
|
|
127
|
+
if (args.filter(a => a === '--save').length !== 1) throw new Error('Use fde setup, or supply all three answers with --save.')
|
|
69
128
|
for (let i = 0; i < rest.length; i += 2) {
|
|
70
129
|
const key = rest[i].replace(/^--/, '')
|
|
71
|
-
if (
|
|
130
|
+
if (key === 'terms-file' && !termsFile && rest[i + 1]) { termsFile = rest[i + 1]; continue }
|
|
131
|
+
if (!rest[i].startsWith('--') || !SETTINGS.concat(QUESTIONS).some(q => q.key === key) || Object.hasOwn(choices, key)) throw new Error('Unknown or repeated setup option.')
|
|
72
132
|
choices[key] = rest[i + 1]
|
|
73
133
|
}
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
134
|
+
const questions = Object.hasOwn(choices, 'work') ? QUESTIONS : SETTINGS
|
|
135
|
+
if (termsFile && questions === SETTINGS) throw new Error('Use personal setup to change custom terms.')
|
|
136
|
+
const result = saveAnswers(store, choices, questions, termsFile)
|
|
137
|
+
console.log('Setup saved locally. Change it anytime with fde setup.\n' + describe(result, questions) + '\n' + nextStep(result) + '\n' + LIMITS)
|
|
77
138
|
return
|
|
78
139
|
}
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
140
|
+
const questions = args[0] === '--settings' ? SETTINGS : QUESTIONS
|
|
141
|
+
if (process.stdin.isTTY && process.stdout.isTTY) return interactive(store, questions)
|
|
142
|
+
console.log('SETUP: ask these three questions together. No settings have been changed.')
|
|
143
|
+
for (const [i, q] of questions.entries()) console.log(`${i + 1}. ${q.title}\n` + q.options.map(([value, label]) => ` ${value}: ${label}`).join('\n'))
|
|
144
|
+
console.log(LIMITS)
|
|
145
|
+
console.log(questions === QUESTIONS
|
|
146
|
+
? 'After answers: fde setup --work single|multiple|team --start new|daily|takeover --masking standard|custom [--terms-file LOCAL_FILE] --save\nFor custom masking, enter names locally with fde setup, or provide a local terms-file path; do not ask to paste sensitive names into chat. Existing terms are kept unless a replacement file is supplied.'
|
|
147
|
+
: 'Save settings: fde setup --view current|portfolio --context standard|compact --privacy agent|reports --save')
|
|
148
|
+
console.log('fde setup --show shows choices and term count only. fde setup --settings changes display and context options.')
|
|
83
149
|
}
|
|
84
|
-
async function interactive(store) {
|
|
85
|
-
// Only static prompts and validated enum labels bypass buffered CLI output.
|
|
150
|
+
async function interactive(store, questions) {
|
|
86
151
|
const emit = text => fs.writeSync(1, text)
|
|
152
|
+
const current = { ...PROFILE_DEFAULTS, ...(store.read() || DEFAULTS) }, choices = {}
|
|
87
153
|
const rl = require('node:readline').createInterface({ input: process.stdin, terminal: false })
|
|
88
154
|
const lines = rl[Symbol.asyncIterator]()
|
|
89
|
-
const current = store.read() || DEFAULTS, choices = {}
|
|
90
155
|
try {
|
|
91
|
-
emit('Make FDEOps fit your
|
|
92
|
-
for (const q of
|
|
156
|
+
emit('Make FDEOps fit your work. Enter keeps the current choice. Ctrl-C cancels.\n' + LIMITS + '\n')
|
|
157
|
+
for (const q of questions) {
|
|
93
158
|
const selected = q.options.findIndex(([v]) => v === current[q.key])
|
|
94
159
|
while (true) {
|
|
95
160
|
emit(q.title + '\n' + q.options.map(([, label], i) => ` ${i + 1}. ${label}${i === selected ? ' (current)' : ''}`).join('\n') + '\n> ')
|
|
96
161
|
const next = await lines.next()
|
|
97
|
-
if (next.done) { emit('\
|
|
98
|
-
const value = next.value.trim(), index = value === '' ? selected : /^[
|
|
99
|
-
if (index >= 0) { choices[q.key] = q.options[index][0]; break }
|
|
100
|
-
emit('Choose
|
|
162
|
+
if (next.done) { emit('\nCancelled; nothing saved.\n'); return }
|
|
163
|
+
const value = next.value.trim(), index = value === '' ? selected : /^[123]$/.test(value) ? Number(value) - 1 : -1
|
|
164
|
+
if (index >= 0 && index < q.options.length) { choices[q.key] = q.options[index][0]; break }
|
|
165
|
+
emit('Choose a listed number, or press Enter.\n')
|
|
101
166
|
}
|
|
102
167
|
}
|
|
103
|
-
|
|
168
|
+
let terms = current.terms
|
|
169
|
+
if (choices.masking === 'custom') {
|
|
170
|
+
emit(`Names or terms to hide, one per line; finish with an empty line. Enter alone keeps ${terms.length} saved terms. This input stays local.\n`)
|
|
171
|
+
const entered = []
|
|
172
|
+
while (true) {
|
|
173
|
+
const next = await lines.next()
|
|
174
|
+
if (next.done) { emit('\nCancelled; nothing saved.\n'); return }
|
|
175
|
+
if (!next.value.trim()) break
|
|
176
|
+
entered.push(next.value.trim())
|
|
177
|
+
validateTerms(entered)
|
|
178
|
+
}
|
|
179
|
+
terms = entered.length ? validateTerms(entered) : terms
|
|
180
|
+
if (!terms.length) throw new Error('Custom masking needs at least one term; nothing saved.')
|
|
181
|
+
}
|
|
182
|
+
const saved = store.read()
|
|
183
|
+
const result = questions === SETTINGS ? { ...(saved || DEFAULTS), ...choices }
|
|
184
|
+
: { ...(saved || DEFAULTS), ...choices, terms, view: !saved || (saved.work && saved.work !== choices.work) ? (choices.work === 'multiple' ? 'portfolio' : 'current') : saved.view }
|
|
185
|
+
emit('\n' + describe(result, questions) + (choices.masking === 'custom' ? `\n${terms.length} custom terms ready to save locally.` : '') + '\nSave these choices? [y/N] ')
|
|
104
186
|
const next = await lines.next()
|
|
105
|
-
if (!next.done && /^y(es)?$/i.test(next.value.trim())) { store.save(
|
|
106
|
-
else emit('\
|
|
187
|
+
if (!next.done && /^y(es)?$/i.test(next.value.trim())) { store.save(result); emit('\nSetup saved. Change it anytime with fde setup.\n' + nextStep(result) + '\n') }
|
|
188
|
+
else emit('\nCancelled; nothing saved.\n')
|
|
107
189
|
} finally { rl.close() }
|
|
108
190
|
}
|
|
109
|
-
module.exports = { createSetup, command, DEFAULTS }
|
|
191
|
+
module.exports = { createSetup, command, DEFAULTS, nextStep }
|
|
@@ -176,7 +176,13 @@ function cliPayload(out) {
|
|
|
176
176
|
|
|
177
177
|
function toolResult(payload) {
|
|
178
178
|
let text
|
|
179
|
-
try {
|
|
179
|
+
try {
|
|
180
|
+
const clean = value => typeof value === 'string' ? masking.mask(value)
|
|
181
|
+
: Array.isArray(value) ? value.map(clean)
|
|
182
|
+
: value && typeof value === 'object' ? Object.fromEntries(Object.entries(value).map(([key, item]) => [key, clean(item)])) : value
|
|
183
|
+
const safe = clean(payload)
|
|
184
|
+
text = typeof safe === 'string' ? safe : JSON.stringify(safe, null, 2)
|
|
185
|
+
}
|
|
180
186
|
catch (_) { return { isError: true, content: [{ type: 'text', text: 'privacy masking unavailable; no unmasked tool output returned' }] } }
|
|
181
187
|
return { content: [{ type: 'text', text }] }
|
|
182
188
|
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "fdeops",
|
|
3
|
-
"version": "
|
|
3
|
+
"version": "4.0.0",
|
|
4
4
|
"description": "Client delivery tools for Forward Deployed Engineers. One @fde skill, local Markdown engagement records, and an offline dashboard for decisions, evidence, approvals, and next actions.",
|
|
5
5
|
"bin": {
|
|
6
6
|
"fdeops": "bin/install.js",
|
package/plugin.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"$schema": "https://agent-plugins.org/schemas/1.0.0/plugin.schema.json",
|
|
3
3
|
"name": "fdeops",
|
|
4
|
-
"version": "
|
|
4
|
+
"version": "4.0.0",
|
|
5
5
|
"description": "Forward deployed engineering skills for AI coding agents. One @fde skill for the client work around the code. You confirm; then it lands in .fde/ on your laptop.",
|
|
6
6
|
"author": {
|
|
7
7
|
"name": "Subash Natarajan",
|
package/skills/fde/SKILL.md
CHANGED
|
@@ -36,13 +36,13 @@ On someone else's site the work is not "write code, remember later." Every chang
|
|
|
36
36
|
|
|
37
37
|
1. **Name it** in `decisions.md` (plan), or timebox the riskiest assumption and record what the POC proves.
|
|
38
38
|
2. **Characterise their code** before you change it. Brownfield: their tests, their runner. Greenfield: the empty tree, first path they can click.
|
|
39
|
-
3. **
|
|
39
|
+
3. **Verify, then prove delivery.** Use their checks and the agreed representative environment; at the delivery checkpoint the signer in `success.md` can replay and reject the acceptance check. See `ship` for evidence requirements.
|
|
40
40
|
4. **If a model judges:** `evals.md` Verdict SHIP before that change is done (eval-pack).
|
|
41
|
-
5. **Log delivery.** Outcome is promised → measured → accepted, not a green CI. Then go live with a
|
|
41
|
+
5. **Log delivery.** Outcome is promised → measured → accepted, not a green CI. Then go live with a tested recovery path (`ship`).
|
|
42
42
|
|
|
43
|
-
A
|
|
43
|
+
Scale the loop to the change. A routine, reversible fix within confirmed scope reuses the existing outcome, signer, acceptance criteria, and engineering plan; batch its verification into a concise delivery receipt. It does not need a new sponsor decision or staging ceremony per edit. New outcomes, changed acceptance or authority, and production release decisions still need the relevant confirmation and evidence. This does not bypass confirmation for judgment written into the engagement record.
|
|
44
44
|
|
|
45
|
-
**
|
|
45
|
+
**Status is explicit.** Record what is implemented, verified, deployed, and accepted separately. A routine fix may be implementation-complete before release or customer acceptance; state what remains and attach the current verification receipt. A coding pack may write the function; `@fde` owns the engagement evidence.
|
|
46
46
|
|
|
47
47
|
## Working with an engineering pack
|
|
48
48
|
|
|
@@ -58,7 +58,9 @@ Fallbacks: `node ~/.claude/fdeops/fde.js …`, then `npx --yes fdeops …`. Skil
|
|
|
58
58
|
|
|
59
59
|
## First-use preferences
|
|
60
60
|
|
|
61
|
-
Run `fde setup --show` before client reads. If unavailable, update the CLI before offering setup
|
|
61
|
+
Run `fde setup --show` before client reads. If unavailable, update the CLI before offering setup. If `configured` is false, bind the named client, then run `fde setup` and present its three questions together: how they work, what would help first, and what to mask. Save their explicit answers; never infer permission to share data. For custom masking, the optional fourth question asks them to enter terms **locally** with `fde setup`, or give a local terms-file path. Do not ask them to paste sensitive names into chat or open that file with model-facing file tools. Pass the path directly to `--terms-file`; inspect only the returned count, never `.preferences.json` or the alias dictionary. If they skip, keep existing defaults.
|
|
62
|
+
|
|
63
|
+
Use `work` to tailor the help: single = focus on the bound client; multiple = portfolio overview with one bound client per write; team = clarify responsibility and handoff, without implying shared storage. `start` chooses the initial route when no more specific request or record determines it: new → land, daily → triage, takeover → audit. Current client evidence and the user's request always take precedence; never restart an existing engagement because of this preference. `masking` selects standard patterns or those plus custom terms. Older technical settings remain valid; offer personal setup when requested rather than resetting them. Do not ask again per client. `fde setup --settings` keeps display, context size and report masking editable. Choices do not configure models or approve client data use.
|
|
62
64
|
|
|
63
65
|
## Entry (every session)
|
|
64
66
|
|
|
@@ -207,8 +209,8 @@ Ready to build with no `terrain.md` / plan: discover or plan first. Takeover wit
|
|
|
207
209
|
|
|
208
210
|
- Never ask the FDE to pick a phase. That's your job.
|
|
209
211
|
- Same six stages at any scale. Overlays carry the industry. Greenfield and brownfield change the first move inside ship, not the map.
|
|
210
|
-
- Ground loop on a bound client: name → characterise →
|
|
211
|
-
-
|
|
212
|
+
- Ground loop on a bound client: name → characterise → verify in the agreed environment → authorize release → log. A coding pack may write the function. `@fde` still owns done. When they disagree, their repo and the signer win.
|
|
213
|
+
- Customer delivery needs a replayable acceptance check in the agreed environment; reuse existing criteria for routine fixes. Missing evidence means unproven, not an observed test failure. Never equate implementation-complete with deployed or customer-accepted.
|
|
212
214
|
- Read `context.md` before speaking. One sharp question - never a barrage.
|
|
213
215
|
- Never invent people, meetings, or numbers - `unknown - ask:` beats a polished lie.
|
|
214
216
|
- Every phase ends with its artifact written. No artifact, no "done."
|
|
@@ -218,4 +220,4 @@ Ready to build with no `terrain.md` / plan: discover or plan first. Takeover wit
|
|
|
218
220
|
|
|
219
221
|
## Identifier masking
|
|
220
222
|
|
|
221
|
-
Before reading engagement content in a session, run `fde privacy` to verify runtime support. If the command is unavailable, stop and update the CLI; a new skill alone does not upgrade an older executable. Use CLI context and previews for model input. They mask common email, phone, SSN-shaped, and credential patterns by default; aliases remain consistent within the local engagements root. Preserve complete alias tokens when drafting updates; the CLI resolves them locally. Never read the private `.privacy/` dictionary, sealed sidecars, raw sensitive notes, or local dashboard/vault files to recover an identity. Names, company names, addresses, and unrecognized formats are not automatically detected: keep sensitive prose in `<private>` blocks. Direct file tools, pasted chat, and upstream source MCPs bypass this boundary.
|
|
223
|
+
Before reading engagement content in a session, run `fde privacy` to verify runtime support. If the command is unavailable, stop and update the CLI; a new skill alone does not upgrade an older executable. Use CLI context and previews for model input. They mask common email, phone, SSN-shaped, and credential patterns by default; aliases remain consistent within the local engagements root. Preserve complete alias tokens when drafting updates; the CLI resolves them locally. Never read the private `.privacy/` dictionary, sealed sidecars, raw sensitive notes, or local dashboard/vault files to recover an identity. Custom masking additionally hides the literal names or terms the user supplied locally, ignoring letter case and matching whole terms. It does not infer variants or discover names. Names, company names, addresses, and unrecognized formats are otherwise not automatically detected: keep sensitive prose in `<private>` blocks. Direct file tools, pasted chat, and upstream source MCPs bypass this boundary.
|
|
@@ -20,19 +20,19 @@ Record in `trust-profile.md` under `## AI policy`.
|
|
|
20
20
|
|
|
21
21
|
## Model selection - choosing the right tool
|
|
22
22
|
|
|
23
|
-
|
|
23
|
+
Compare plausible approaches against the task’s quality, latency, privacy, and operating constraints. Optimize total cost per successful outcome, including retries, review, failures, and maintenance; call price alone can select the more expensive system.
|
|
24
24
|
|
|
25
|
-
**
|
|
26
|
-
1. **Can rules solve it?** If yes, no model needed.
|
|
27
|
-
2. **Can a small/fast model solve it?**
|
|
28
|
-
3. **
|
|
29
|
-
4. **Does it need fine-tuning?**
|
|
25
|
+
**Candidate approaches (test the plausible ones, not a mandatory ladder):**
|
|
26
|
+
1. **Can rules solve it?** If yes, no model needed. Include their implementation and maintenance cost.
|
|
27
|
+
2. **Can a small/fast model solve it?** Include one when suitable for the task and hosting constraints; measure quality and total cost.
|
|
28
|
+
3. **Would a more capable model improve the outcome?** It may reduce retries, supervision, or implementation complexity enough to justify its price. Verify current model availability and capabilities in official documentation.
|
|
29
|
+
4. **Does it need fine-tuning?** Consider it when representative data and a held-out evaluation support a persistent domain gap. Compare against prompt/retrieval changes; justify dataset coverage and training, serving, and maintenance costs rather than assuming a fixed example count suffices.
|
|
30
30
|
|
|
31
31
|
**Evaluation method (before choosing):**
|
|
32
|
-
- Build a test set
|
|
33
|
-
- Run
|
|
34
|
-
- Score:
|
|
35
|
-
-
|
|
32
|
+
- Build a representative test set with expected outcomes and critical failure cases. Size it to diversity, consequence, and uncertainty; a small pilot set cannot establish rare-failure safety.
|
|
33
|
+
- Run a bounded shortlist against the same held-out cases; record model/version and settings.
|
|
34
|
+
- Score: task success, critical failures, latency, total cost per successful outcome, and failure modes, including repeated runs when variability matters.
|
|
35
|
+
- Select the approach that meets the agreed constraints with the best measured tradeoff. Record uncertainty and what would trigger re-evaluation.
|
|
36
36
|
|
|
37
37
|
Write model selection rationale to `decisions.md`. Include: models tested, test set size, scores, cost comparison.
|
|
38
38
|
|
|
@@ -42,7 +42,7 @@ When any slice touches a model, embeddings, RAG, or an agent: create or update `
|
|
|
42
42
|
|
|
43
43
|
**Minimum pack (do not grow until the minimum exists):**
|
|
44
44
|
1. **Component + quality bar** - one sentence each; kill switch / fallback named.
|
|
45
|
-
2. **Golden cases** -
|
|
45
|
+
2. **Golden cases** - representative inputs with expected outputs and a pass rule. 5-20 can seed a pilot, not certify readiness; expand for risk and coverage. Prefer real production-shaped data (sanitized).
|
|
46
46
|
3. **Failure modes** - at least the silent ones: hallucination/ungrounded, retrieval miss (if RAG), drift, cost runaway.
|
|
47
47
|
4. **Pass/fail** - dated run; Verdict **SHIP** or **NO-SHIP**; critical fails must be 0.
|
|
48
48
|
5. **HITL gate** - which decisions need human review before action (align with `trust-profile.md`). Empty when policy requires review → NO-SHIP.
|
|
@@ -59,9 +59,9 @@ When the AI needs to answer questions about the client's data:
|
|
|
59
59
|
3. **Generate** - chunks + query → LLM → answer with citations
|
|
60
60
|
|
|
61
61
|
**Common failure modes:**
|
|
62
|
-
- **Chunk size wrong.** Too small = lost context. Too large = noise drowns signal.
|
|
62
|
+
- **Chunk size wrong.** Too small = lost context. Too large = noise drowns signal. Choose boundaries from document structure and answer needs; tune size, overlap, and top-K against retrieval and answer-quality evals, latency, and context limits.
|
|
63
63
|
- **No citation/grounding.** If the model can't point to where it found the answer, you can't verify it. Always require source attribution.
|
|
64
|
-
- **Stale index.** Documents update, embeddings don't. Define
|
|
64
|
+
- **Stale index.** Documents update, embeddings don't. Define refresh and deletion handling from source update patterns and acceptable staleness; test them.
|
|
65
65
|
- **Retrieval miss.** The right document exists but wasn't retrieved. Test with known-answer queries where the answer IS in the corpus - if retrieval misses these, the embedding model or chunking strategy needs work.
|
|
66
66
|
|
|
67
67
|
## Agent and agentic systems
|
|
@@ -71,23 +71,23 @@ When the AI takes actions (not just generates text):
|
|
|
71
71
|
**Safety principles:**
|
|
72
72
|
- **Least privilege.** An agent gets the minimum permissions needed. Never give an agent admin access "for convenience."
|
|
73
73
|
- **Confirmation gates.** Any destructive or irreversible action requires human confirmation. Delete, send, transfer, publish = confirm before execute.
|
|
74
|
-
- **
|
|
74
|
+
- **Observable execution.** Record tool/action summaries, versions, timing, cost, outcomes, validation results, and concise decision rationale. Do not request or store hidden chain-of-thought. Minimize and redact logged inputs/outputs; apply the client’s access, retention, and data policies. Never log raw `<private>` content or secrets.
|
|
75
75
|
- **Deterministic fallbacks.** When the agent fails or is uncertain, it falls back to a known-safe behavior (queue for human review, return a safe default, do nothing). "The agent got confused and did something unexpected" is never acceptable in production.
|
|
76
|
-
- **Cost caps.** Agents in loops can burn through API budgets.
|
|
76
|
+
- **Cost caps.** Agents in loops can burn through API budgets. Set request and aggregate budgets with bounded retries and stopping conditions. Choose alert thresholds early enough for the owner to act.
|
|
77
77
|
|
|
78
78
|
## AI governance - responsible deployment
|
|
79
79
|
|
|
80
80
|
**Before production:**
|
|
81
|
-
- **Bias testing.** Run the model on demographic-varied inputs.
|
|
81
|
+
- **Bias testing.** Run the model on demographic-varied inputs. Define relevant groups, harms, and acceptable disparity with the responsible owner; investigate material differences and block unresolved critical harm. Aggregate accuracy alone is insufficient.
|
|
82
82
|
- **Explainability.** Can you explain to a non-technical stakeholder why the model made a specific decision? If not, it's a black box - some jurisdictions and industries prohibit this.
|
|
83
83
|
- **Model card.** Document: what the model does, what data it was trained/tuned on, known limitations, failure modes, who owns it. One page. Required before production.
|
|
84
84
|
- **Kill switch.** Every AI component must be disable-able without taking down the feature it powers. The fallback path (rule-based, human-routed, or gracefully degraded) must work when the AI is off.
|
|
85
85
|
|
|
86
86
|
**In production:**
|
|
87
|
-
- **Drift monitoring.** Compare production outputs
|
|
87
|
+
- **Drift monitoring.** Compare production outputs against baseline quality on a cadence matched to traffic, drift risk, and impact. Quality can deteriorate gradually or fail abruptly after model, data, tool, or policy changes; monitor both patterns.
|
|
88
88
|
- **Feedback collection.** Thumbs up/down, corrections, escalations. This is your retraining signal AND your quality metric.
|
|
89
89
|
- **Cost monitoring.** Track: tokens consumed, calls made, cost per user, cost per feature. AI costs surprise everyone at scale.
|
|
90
|
-
- **Incident response.** When the AI produces harmful/wrong output: disable (kill switch), investigate (
|
|
90
|
+
- **Incident response.** When the AI produces harmful/wrong output: disable (kill switch), investigate (sanitized execution traces and observed outcomes), fix (prompt/model/data), restore. Define this BEFORE it happens.
|
|
91
91
|
|
|
92
92
|
## Writes
|
|
93
93
|
|
|
@@ -96,10 +96,10 @@ When the AI takes actions (not just generates text):
|
|
|
96
96
|
## Principles
|
|
97
97
|
|
|
98
98
|
- AI degrades silently. Monitor outputs, not just uptime.
|
|
99
|
-
-
|
|
99
|
+
- Choose by measured quality and total cost per successful outcome, within policy and latency constraints.
|
|
100
100
|
- No golden set, no AI ship (`evals.md` Verdict SHIP).
|
|
101
101
|
- Every AI component needs a kill switch and a fallback path.
|
|
102
|
-
-
|
|
102
|
+
- Debug from privacy-safe observable execution and concise rationale, never hidden chain-of-thought.
|
|
103
103
|
- Drift is inevitable. Define the detection method before shipping.
|
|
104
104
|
- Cost at scale ≠ cost at pilot. Model the 10× number before committing.
|
|
105
105
|
- Bias testing is a pre-production gate, not a post-launch audit.
|
|
@@ -14,13 +14,13 @@ Intelligence without evidence is token-maxing with a nicer name. An FDE earns tr
|
|
|
14
14
|
|
|
15
15
|
**1. Scope the judgement surface.** One sentence: which step uses model judgement, and what must never be autonomous.
|
|
16
16
|
|
|
17
|
-
**2. Build a golden set
|
|
17
|
+
**2. Build a golden set sized to risk and coverage.** A pilot may start with 5-20 cases; neither that range nor 50-100 proves broad-scale safety. Include representative segments, boundary cases, and known critical failures. Expand based on observed failure modes and uncertainty; hold out cases from tuning and repeat runs when variability matters. For each case:
|
|
18
18
|
- input (sanitized - no `<private>` raw values)
|
|
19
19
|
- expected outcome or expert-approved acceptance note
|
|
20
20
|
- pass rule (exact / contains / short rubric)
|
|
21
21
|
- source: real historical example / expert label / staged fixture
|
|
22
22
|
|
|
23
|
-
**3. Score pass/fail, not vibes.**
|
|
23
|
+
**3. Score pass/fail, not vibes.** Define the quality threshold and critical failure rules before running the suite. Run it; record count pass / fail, segment coverage, and limitations. Failures get a failure-mode tag (missing data, wrong record, format drift, hallucination, retrieval miss, unsafe action, other).
|
|
24
24
|
|
|
25
25
|
**4. Human-in-the-loop gate.** Name which outcomes require human approve before side effects. Judgement that has a side effect (write, send, transfer, ticket, deploy, pay, page) is **NO-SHIP** without a named human on their side in the loop. Do not write "none - allowed under policy" to bless lights-out write-access. Staging may run a supervised loop with a kill switch, a cost cap, and a golden set from **their** failures. Production stays gated until they have a written policy, a named owner, and dated eval receipts on real traffic.
|
|
26
26
|
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
|
|
5
5
|
**Read first:** `reality.md`, `success.md`, `terrain.md`, `stakeholders.md`. Load `business-case.md` if poc produced one. Not the full folder.
|
|
6
6
|
|
|
7
|
-
**Before
|
|
7
|
+
**Before a new delivery plan or material scope change:** run `fde doctor --ready`. Missing binary success or a named customer-side signer blocks progression: review the proposed acceptance check and authority with the FDE first. Use a test/input and observable pass/fail under **Done when:** or **Acceptance check:**. A number, role, or successful demo alone is insufficient. Do not invent missing facts to pass lint. Routine reversible fixes within confirmed scope reuse the existing signer, acceptance criteria, and engineering plan; record verification without reopening settled decisions.
|
|
8
8
|
|
|
9
9
|
## Validation gate (confirm understanding, clarify where it elevates)
|
|
10
10
|
|
|
@@ -34,7 +34,7 @@ An FDE plan is not a sprint backlog. The technical sequence is the easy part. Th
|
|
|
34
34
|
|
|
35
35
|
**3. One user action per change.** Each task delivers something visible and testable ("user submits form, sees it saved"), never a layer ("build the database layer"). See `ship`.
|
|
36
36
|
|
|
37
|
-
**4. Size to
|
|
37
|
+
**4. Size to a coherent, verifiable outcome.** Split unrelated work and tasks too complex to review or recover safely. Use bounded review sections for large cohesive changes; elapsed time and line count are signals to examine, not universal limits.
|
|
38
38
|
|
|
39
39
|
**5. AI components get explicit eval tasks.** "Output validated on 50 real production examples," "fallback tested under model unavailability," "inputs/outputs logging to <destination>" - these are pre-conditions of shipping, in the plan before build starts.
|
|
40
40
|
|
|
@@ -8,7 +8,7 @@ Engagement review ≠ product-company review: a codebase you don't own, systems
|
|
|
8
8
|
|
|
9
9
|
## Pre-flight: is this reviewable?
|
|
10
10
|
|
|
11
|
-
|
|
11
|
+
Check whether the diff has one agreed intent and can be reviewed with the available evidence. Split unrelated work; for a large cohesive change, separate generated/mechanical output from behavioral changes and review in bounded sections. Size is a warning to investigate, not a universal stop threshold.
|
|
12
12
|
|
|
13
13
|
## Stage 1 - did we build what we agreed? (you do this work)
|
|
14
14
|
|
|
@@ -43,19 +43,19 @@ Five dimensions, line-specific ("line 47 fails under concurrent writes - no lock
|
|
|
43
43
|
- **Correctness** - does what it says; edge cases; error paths traced.
|
|
44
44
|
- **Blast radius** - what breaks at 2am; downstream systems; failure mode loud (errors surface) or silent (data corrupts over time)?
|
|
45
45
|
- **Security** - input validation at boundaries; no secrets in logs; no new attack surface; `trust-profile.md` sensitivity classes respected.
|
|
46
|
-
- **
|
|
47
|
-
- **AI policy & components** - human-review requirements honoured; model output treated as untrusted until validated; fallback exists;
|
|
46
|
+
- **Recovery** - can the documented, tested rollback or recovery path meet the agreed recovery-time and data-loss limits? For irreversible changes, require explicit authority, compatibility checks, and a tested restore/compensation or roll-forward plan; a code revert alone is not proof.
|
|
47
|
+
- **AI policy & components** - human-review requirements honoured; model output treated as untrusted until validated; fallback exists; privacy-safe execution evidence retained under the client’s data policy (no secrets, raw private data, or hidden chain-of-thought); outputs bounded so a hallucination can't cascade; check applicable explanation and human-review requirements with the client’s responsible owner; provide source evidence and concise rationale without claiming access to hidden reasoning.
|
|
48
48
|
|
|
49
|
-
**Structural pass on AI-heavy or data-touching changes:**
|
|
49
|
+
**Structural pass on AI-heavy or data-touching changes:** migration compatibility and tested recovery · destructive SQL guarded · PII/PCI/PHI paths match `trust-profile.md` · side effects (flags, webhooks, emails, jobs) fire only when intended · magic strings that break on rename · new behaviour has a test or an explicit reason it can't yet. One line problem, one line fix.
|
|
50
50
|
|
|
51
51
|
## The review-fix loop (until clean)
|
|
52
52
|
|
|
53
53
|
1. Read the full diff before commenting.
|
|
54
54
|
2. Verdicts: **Stage 1: Pass / Blocked (reason)** · **Stage 2: Pass / Concerns (line-specific)**.
|
|
55
|
-
3. Fix only **real** findings tied to this change - no drive-by refactors. Reject false positives with one sentence why. Their comments are to check, not to obey. Restate each against the one-line intent and `trust-profile.md`.
|
|
55
|
+
3. Fix only **real** findings tied to this change - no drive-by refactors. Reject false positives with one sentence why. Their comments are to check, not to obey. Restate each against the one-line intent and `trust-profile.md`. An unclear item waits for clarification; continue independent, understood fixes. If it breaks a signed constraint, a sacred system, or nothing calls it: one-sentence pushback, then wait.
|
|
56
56
|
4. Add or update a test per bug found where possible.
|
|
57
57
|
5. Re-run tests/typechecks - state what ran.
|
|
58
|
-
6. Re-review.
|
|
58
|
+
6. Re-review. If two repair/review cycles do not converge, reassess the evidence and approach; continue independent fixes and escalate concrete scope/product decisions.
|
|
59
59
|
|
|
60
60
|
## Before the PR - thinking for the next reader
|
|
61
61
|
|
|
@@ -72,6 +72,6 @@ Code alone loses the "why." Before you call the change reviewable, run the **ses
|
|
|
72
72
|
- Stage 1 before Stage 2. Wrong scope reviewed well is still wrong scope.
|
|
73
73
|
- KEEP / JUSTIFY / SPLIT / DROP - every path gets a verdict; silent extras fail Stage 1.
|
|
74
74
|
- Specific or silent - vague concerns waste everyone's time.
|
|
75
|
-
- No
|
|
75
|
+
- No viable tested recovery path = a release blocker.
|
|
76
76
|
- A clean review proves this diff is safe as agreed - not that the feature was right.
|
|
77
77
|
- Judgment in `.fde/` beats transcript in git.
|
|
@@ -13,7 +13,7 @@
|
|
|
13
13
|
| Change type | Rollback method | Complication | Test |
|
|
14
14
|
|-------------|----------------|--------------|------|
|
|
15
15
|
| **Code deploy** | Revert the PR / redeploy previous version | Feature flags, cache invalidation | Deploy previous version to staging, verify function |
|
|
16
|
-
| **Database migration** |
|
|
16
|
+
| **Database migration** | Compatible rollback, restore, or roll-forward | Irreversible transforms, concurrent writes, old/new schema compatibility | Rehearse on representative staging data; verify integrity, elapsed time, and possible data loss |
|
|
17
17
|
| **Config change** | Restore previous config | Propagation delay, dependent service restarts | Flip config, verify all services pick it up |
|
|
18
18
|
| **Infrastructure** | Terraform/Pulumi rollback or manual | State drift, dependent resources | Plan the rollback, review the diff |
|
|
19
19
|
| **Data backfill** | Restore from backup or reverse script | Mixed old/new data states | Run reverse on a 100-row sample |
|
|
@@ -96,7 +96,7 @@ One statement: "Rollback tested on staging. Time: <N minutes>. Result: <pass/fai
|
|
|
96
96
|
## Principles
|
|
97
97
|
|
|
98
98
|
- A rollback plan that hasn't been tested is a wish.
|
|
99
|
-
- Time the drill
|
|
99
|
+
- Time the drill and account for differences in production scale and operating conditions; do not assume a fixed multiplier.
|
|
100
100
|
- Identify the irreversible components and name the compensating action.
|
|
101
101
|
- The drill report is evidence for the change ticket and the team's confidence.
|
|
102
102
|
- A drill that fails is a success - you found the problem before production did.
|
|
@@ -10,11 +10,11 @@ Do not ask them to pick a mode. Name where you are, then start at the matching s
|
|
|
10
10
|
- On staging, the signer in `success.md` can reject it → **go-live**
|
|
11
11
|
- Prod is the question → **go-live**. Do not start a second change.
|
|
12
12
|
|
|
13
|
-
If going live,
|
|
13
|
+
If going live, check the evidence for the recovery path: has rollback, restore, compensation, or roll-forward been exercised under representative conditions? If only planned, validate it before release. Reuse applicable drill evidence when the mechanism and relevant conditions are unchanged; record why it applies.
|
|
14
14
|
|
|
15
15
|
A bounded experiment that tests an assumption is `poc`. This skill turns a validated direction into a maintainable change on a repo they will own, then production. Inspect existing prototype code and retain suitable tested parts; replace unsafe shortcuts based on evidence. A successful demo alone does not satisfy the readiness gates below.
|
|
16
16
|
|
|
17
|
-
**Before
|
|
17
|
+
**Before a new delivery plan or material scope change:** run `fde doctor --ready`. Missing binary success or a named customer-side signer blocks progression: review the proposed acceptance check and authority with the FDE first. Use a test/input and observable pass/fail under **Done when:** or **Acceptance check:**. A number, role, or successful demo alone is insufficient. Do not invent missing facts to pass lint. Routine reversible fixes within confirmed scope reuse the existing signer and acceptance check; record verification without restarting approval. New judgment in the record still requires confirmation.
|
|
18
18
|
|
|
19
19
|
## Field (name it once, then the same loop)
|
|
20
20
|
|
|
@@ -22,18 +22,18 @@ A bounded experiment that tests an assumption is `poc`. This skill turns a valid
|
|
|
22
22
|
|--|------------|------------|
|
|
23
23
|
| What you touch | Code they already run | A new path or empty tree they will own |
|
|
24
24
|
| First move | Characterise their tests, their runner, the workaround in `terrain.md` | First path a user can click. Not the whole product. |
|
|
25
|
-
| Proof |
|
|
26
|
-
| Undo | Revert this change on its own |
|
|
25
|
+
| Proof | Agreed representative environment and replayable acceptance check | Agreed representative environment and replayable acceptance check; record what remains untested before release |
|
|
26
|
+
| Undo | Revert this change on its own | Name rollback or tested recovery; identify irreversible effects and required authority. |
|
|
27
27
|
|
|
28
28
|
Skip POC only when the killer assumption already lives in the repo (typical brownfield). If the bet is unproven, `poc` first.
|
|
29
29
|
|
|
30
|
-
**
|
|
30
|
+
**Customer delivery means:** the signer in `success.md` can replay and reject the agreed acceptance check in an environment they operate. A green check on your laptop proves only what ran there. Routine fixes can share a delivery checkpoint; distinguish implementation, verification, deployment, and acceptance.
|
|
31
31
|
|
|
32
32
|
If `terrain.md` **Data estate** lists a **Blocker** this change depends on (source or pipe): stop. That is discover, not ship. Do not build a path they cannot feed.
|
|
33
33
|
|
|
34
34
|
## Method - one change they can see
|
|
35
35
|
|
|
36
|
-
One change = one
|
|
36
|
+
One change = one coherent outcome with observable verification and a bounded recovery path. Prefer vertical slices that can be reviewed and exercised independently. A PR is how this often lands. It is not the job. The job is the change they can see.
|
|
37
37
|
|
|
38
38
|
```
|
|
39
39
|
BAD (layers):
|
|
@@ -49,7 +49,7 @@ GOOD (one user action each):
|
|
|
49
49
|
4: Admin can void a payment (auth + logic + UI) - testable
|
|
50
50
|
```
|
|
51
51
|
|
|
52
|
-
|
|
52
|
+
Prefer independently revertible changes. When data or external effects cannot be undone, name the dependency, containment, tested recovery, and authorized owner before release.
|
|
53
53
|
|
|
54
54
|
**Before you start this change:**
|
|
55
55
|
|
|
@@ -58,7 +58,7 @@ Each change is independently revertible.
|
|
|
58
58
|
- [ ] Rollback named: revert this change, or something more specific
|
|
59
59
|
- [ ] No dependency on an unmerged change (if dependent, state it and land in order)
|
|
60
60
|
- [ ] `Kill if` is written - the observation that stops this change
|
|
61
|
-
- [ ] Before-
|
|
61
|
+
- [ ] Before-state evidence identified: the relevant failing output, number, or behavior, with its source and date. For a routine fix within confirmed scope, reference applicable existing evidence and batch the delivery receipt; capture new evidence when the relevant behavior or conditions changed. Never imply an old check was rerun.
|
|
62
62
|
- [ ] Open PRs and uncommitted work in the area checked (`gh pr list`, `gh pr diff <n> --name-only`); overlap goes to `decisions.md` before you start
|
|
63
63
|
|
|
64
64
|
Your coding pack writes the function. This skill owns done. When they disagree with this repo, the repo wins.
|
|
@@ -69,34 +69,25 @@ Your coding pack writes the function. This skill owns done. When they disagree w
|
|
|
69
69
|
Read existing code in the area (search before creating)
|
|
70
70
|
→ Characterise what is already there (their tests, their runner; greenfield: the empty tree)
|
|
71
71
|
→ Implement the smallest path that works
|
|
72
|
-
→
|
|
72
|
+
→ Verify the change; demonstrate at the agreed delivery checkpoint (below)
|
|
73
73
|
→ Cleanup pass (dedupe, simplify - behaviour unchanged)
|
|
74
74
|
→ Self-review against acceptance criteria
|
|
75
75
|
→ Commit with a message the client's team can read
|
|
76
76
|
→ Update decisions.md + delivery.md
|
|
77
77
|
```
|
|
78
78
|
|
|
79
|
-
**
|
|
79
|
+
**Verify the change and prove customer delivery.** Match evidence to the reviewed revision, environment, and acceptance criteria.
|
|
80
80
|
|
|
81
|
-
-
|
|
82
|
-
-
|
|
83
|
-
-
|
|
84
|
-
- **
|
|
81
|
+
- Use **their** test commands, fixtures, and CI. Record the command, result, revision, environment, and run date in `delivery.md`. Reuse existing evidence only when the relevant code and conditions are unchanged, citing why it still applies; never claim it was rerun. Run affected checks for changed behavior and required release checks before deployment. Missing evidence means unproven, not an observed failure.
|
|
82
|
+
- At the agreed delivery checkpoint, the signer in `success.md` must be able to replay and reject the acceptance check using an interface they operate (screen, API, report, or equivalent). Routine fixes can share that checkpoint; passing tests alone does not establish customer acceptance.
|
|
83
|
+
- Prefer staging they operate. When unavailable, use an agreed, permitted representative test environment, record its owner and limitations, and resolve material release-evidence gaps before production. A local demonstration is not deployment.
|
|
84
|
+
- **Representative data.** Use permitted sanitized or synthetic fixtures that exercise relevant volumes, edge cases, and operating paths. Before go-live, record gaps such as batch timing, distribution, or production-only dependencies and their impact on the acceptance and abort checks. Resolve material gaps or explicitly narrow the release; never load sensitive production data merely to make a demo realistic.
|
|
85
85
|
- Model in the path: `eval-pack` until `evals.md` says SHIP. Do not skip because "it looked right in chat."
|
|
86
86
|
- A model drafts. A named human on their side ships. No unsupervised loop on their production. If the brief demands lights-out write-access, that is `who-decides` / `hold-scope`, not ship.
|
|
87
87
|
|
|
88
88
|
The proof is whatever this client already believes, plus one new receipt they can replay.
|
|
89
89
|
|
|
90
|
-
**Size.**
|
|
91
|
-
|
|
92
|
-
| Metric | Target | Why |
|
|
93
|
-
|--------|--------|-----|
|
|
94
|
-
| Lines changed | 100-300 | Reviewable in one sitting |
|
|
95
|
-
| Time to implement | 30-90 minutes | Testable before context decays |
|
|
96
|
-
| Files touched | 1-5 | Blast radius stays containable |
|
|
97
|
-
| Tests added | ≥1 per new behaviour | Proves this change; guards against regression |
|
|
98
|
-
|
|
99
|
-
Larger than 300 lines → split first. "It's all connected" means the design needs work, not a bigger dump.
|
|
90
|
+
**Size by reviewability and risk.** Keep one coherent intent, bounded context, and observable acceptance checks. Split unrelated behavior or work whose recovery and review cannot be understood together. Diff size and elapsed time are warning signals, not hard gates: generated changes may be large and low risk; a one-line permission change may be critical. Use the repository’s checks and add meaningful coverage for changed behavior, rather than a test-count quota.
|
|
100
91
|
|
|
101
92
|
**Show it.** Every 2-3 changes, something the customer can see: an endpoint they can hit, a UI they can click, a metric that moved, a risk that was retired. Technical progress invisible to stakeholders is trust decay. `delivery.md` gets updated after every visible change.
|
|
102
93
|
|
|
@@ -106,7 +97,7 @@ Larger than 300 lines → split first. "It's all connected" means the design nee
|
|
|
106
97
|
- If it's NOT in `decisions.md`: log it as a scope receipt (see `hold-scope.md`), don't touch it.
|
|
107
98
|
- Ugly code outside this change stays ugly. That is discipline, not laziness.
|
|
108
99
|
|
|
109
|
-
After each change:
|
|
100
|
+
After each change: required checks pass with applicable evidence, acceptance criteria evaluated, blast radius as declared, `Kill if` still false. At the agreed delivery checkpoint: what did they see, and what is their signal? Before production, complete the go-live gates below.
|
|
110
101
|
|
|
111
102
|
---
|
|
112
103
|
|
|
@@ -141,7 +132,7 @@ Score each dimension green/amber/red. This is the gate, not a suggestion:
|
|
|
141
132
|
| Dimension | Green | Amber | Red |
|
|
142
133
|
|-----------|-------|-------|-----|
|
|
143
134
|
| Tests | All pass on deploy branch | Flaky tests skipped with justification | Failures present or tests not run |
|
|
144
|
-
|
|
|
135
|
+
| Recovery | Applicable tested rollback/restore/compensation/roll-forward meets agreed recovery and data-loss limits | Documented; drill evidence needs refresh | No viable recovery, failed drill, or irreversible effects lack explicit authority |
|
|
145
136
|
| Sign-off | Stakeholder approval in `decisions.md` with date | Verbal approval, not logged | No approval sought |
|
|
146
137
|
| Runbook | Exists and someone other than you has read it | Exists but unreviewed | Missing |
|
|
147
138
|
| Monitoring | Alerts configured, owner named, dashboard live | Alerts configured, no named owner | No monitoring |
|
|
@@ -211,16 +202,16 @@ grep -rnE "(api[_-]?key|secret|password|token)\s*[:=]\s*['\"][^'\"]{8,}" \
|
|
|
211
202
|
--include="*.js" --include="*.ts" --include="*.py" --include="*.env" \
|
|
212
203
|
--include="*.yaml" --include="*.json" . | grep -vE "example|template|test" | head
|
|
213
204
|
```
|
|
214
|
-
- DB migrations
|
|
215
|
-
-
|
|
205
|
+
- DB migrations checked for compatibility, data loss, and old/new application coexistence. Prefer expand/contract for destructive changes. Irreversible steps require explicit authority and a tested restore, compensation, or roll-forward plan.
|
|
206
|
+
- Recovery documented **and tested**, with acceptable recovery time and data loss.
|
|
216
207
|
- Monitoring alerts configured, someone watching.
|
|
217
208
|
- Team knows the deploy is happening.
|
|
218
|
-
-
|
|
209
|
+
- Deploy window has staffed observation and recovery coverage appropriate to the risk; respect the client’s change calendar and business-critical periods.
|
|
219
210
|
- **Change approval (CAB) environments:** window open, ticket approved. In banking/healthcare/gov, deploying outside an approved window is a compliance finding even when the deploy succeeds. "We didn't know there was a CAB process" is not a defence - find out before the deploy date.
|
|
220
211
|
|
|
221
212
|
## Method - the deploy
|
|
222
213
|
|
|
223
|
-
**
|
|
214
|
+
**Rollout:** Choose canary, blue/green, staged cohorts, or the client’s proven release mechanism based on isolation, traffic, and failure cost. For a canary, set cohort size, exposure cap, observation duration, minimum sample, and advance/abort thresholds before starting; allow for delayed and batch effects. Watch errors, latency, and **the business metric this change affects**. Breached thresholds or critical harm → halt expansion and execute the tested recovery/containment plan; investigate after exposure is controlled. Advance only with sufficient evidence and a named operator.
|
|
224
215
|
|
|
225
216
|
**Canary receipt** (write it, or the canary did not happen): what was watched, on whose dashboard, for how long, and that the next change did not start in the window. If prod is a CAB console, vendor button, or their pipeline, write the owner and the click path - the host agent does not get to pretend it shipped.
|
|
226
217
|
|
|
@@ -275,7 +266,7 @@ Never skip a step. The sponsor always wants to skip from pilot to standard - tha
|
|
|
275
266
|
Adoption isn't a handoff-stage problem - it starts while you are still writing the change. Software that launches to silence is software that gets decommissioned.
|
|
276
267
|
|
|
277
268
|
**During the change:**
|
|
278
|
-
- **
|
|
269
|
+
- **Controlled exposure.** Use a feature flag or equivalent isolation when it reduces rollout risk. Choose cohorts and expansion criteria from traffic and impact; name the flag owner and removal point.
|
|
279
270
|
- **Feedback loops built in.** A thumbs-up/down, a "was this helpful?", a usage counter. Instrument adoption, don't assume it.
|
|
280
271
|
- **Resistance signals.** Watch for: workaround creation (they built a spreadsheet instead of using the tool), drop-off after day 3 (onboarding fails), vocal detractors (one influential skeptic can kill adoption). Address these before launch, not after.
|
|
281
272
|
|
|
@@ -290,13 +281,13 @@ Adoption isn't a handoff-stage problem - it starts while you are still writing t
|
|
|
290
281
|
|
|
291
282
|
**`decisions.md`** - each change: what was implemented, what was tested, what was deferred, `Kill if`.
|
|
292
283
|
|
|
293
|
-
**`delivery.md`** - each visible change in business language; then the deployment record: what shipped, when,
|
|
284
|
+
**`delivery.md`** - each visible change in business language; then the deployment record: what shipped, when, recovery procedure, pulse definition, **scale-readiness assessment, and adoption metrics**. Written for whoever inherits the system.
|
|
294
285
|
|
|
295
286
|
## Checkpoint
|
|
296
287
|
|
|
297
|
-
After each change:
|
|
288
|
+
After each change: required checks pass, acceptance criteria evaluated, blast radius as declared, `Kill if` still false. Batch routine fixes at the agreed delivery checkpoint; record staging and customer acceptance separately.
|
|
298
289
|
|
|
299
|
-
Before
|
|
290
|
+
Before full exposure: the chosen rollout’s advance criteria are met with sufficient observation and business-metric evidence, and the pulse is written into `delivery.md`. Also green: value bucket named, audit receipt dated, eval receipt **n/a or pass**, **intent vs diff clean** (no unresolved SPLIT/DROP). Missing any of those → not green. For enterprise-scale: scale-readiness gate passed before broad rollout.
|
|
300
291
|
|
|
301
292
|
## Worked example
|
|
302
293
|
|
|
@@ -315,10 +306,10 @@ Greenfield is the same loop with an empty tree: first path a user can click, on
|
|
|
315
306
|
## Principles
|
|
316
307
|
|
|
317
308
|
- One user action per change. Layers are untestable until assembled.
|
|
318
|
-
-
|
|
309
|
+
- Prove the agreed outcome in their environment and test recovery. Local green is not customer delivery.
|
|
319
310
|
- The ugly code outside this change stays ugly. That's discipline, not laziness.
|
|
320
|
-
- A deployment
|
|
321
|
-
-
|
|
311
|
+
- A deployment needs tested recovery within agreed time and data-loss limits; irreversible effects require explicit authority.
|
|
312
|
+
- Halt expansion on breached thresholds or critical harm; contain exposure with the tested recovery plan before investigating.
|
|
322
313
|
- Verify the business metric, not just the technical one.
|
|
323
314
|
- No value bucket, no green ship. No pulse, no done.
|
|
324
315
|
- Diff larger than the stated intent without KEEP/JUSTIFY receipts = fix-first.
|