fdeops 3.31.0 → 4.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -40,6 +40,16 @@ One chat. Name the client:
40
40
 
41
41
  That creates `~/fde-engagements/client01/.fde/` on your laptop. Paste kickoff notes in the same thread. `@fde` picks what to check. You still decide. After a meeting you review what changed, new asks, open questions, and next actions. Correct the proposal, then confirm the update.
42
42
 
43
+ ### Make it fit your work
44
+
45
+ Setup asks three short questions: how you work, what would help first, and what to mask before sharing context with your agent. Review your choices before saving; change them anytime.
46
+
47
+ ```bash
48
+ npx fdeops@latest setup
49
+ ```
50
+
51
+ [See the choices and custom masking options](docs/USAGE.md#make-fdeops-fit-your-day).
52
+
43
53
  Open the engagement fieldbook:
44
54
 
45
55
  ```bash
@@ -56,8 +66,6 @@ Read-only HTML of the record - promised, measured, accepted, and evidence. Regen
56
66
  /plugin install fdeops@fdeops
57
67
  ```
58
68
 
59
- Make it fit your day: `fde setup` asks how you work, what would help first, and what to mask before sharing context with your agent. [First-use setup](docs/USAGE.md#make-fdeops-fit-your-day).
60
-
61
69
  The plugin adds session hooks and the slash commands below. Skill-only installation does not add hooks. See the [installation guide](docs/install.md) for setup details.
62
70
 
63
71
  </details>
package/bin/fde.js CHANGED
@@ -1917,6 +1917,19 @@ function stripApprovedStamp(text) {
1917
1917
  return String(text || '').replace(/\s*\[approved:\s*[^\]]+\]/i, '').trim()
1918
1918
  }
1919
1919
 
1920
+ function proposedSigner(body) {
1921
+ // Unknown or tentative ownership is a note, never a fabricated sign-off.
1922
+ const sources = body.match(/\[source:[^\]]*\]/gi) || []
1923
+ const prose = body.replace(/\[source:[^\]]*\]/gi, '').trim()
1924
+ if (/\?|\b(?:maybe|might|whether|could|should|would|if|unless)\b/i.test(prose)) return ''
1925
+ // Check uncertainty before removing a suffix, then validate the actual
1926
+ // identity so "customer sponsor signs off" cannot become a named person.
1927
+ if (!require('./lib/value-ledger').acceptanceName(prose)) return ''
1928
+ const identity = prose.replace(/\s+signs?(?:\s+off)?\b.*$/i, '').trim()
1929
+ if (!require('./lib/value-ledger').acceptanceName(identity)) return ''
1930
+ return [identity, ...sources].join(' ')
1931
+ }
1932
+
1920
1933
  // One screen a human can confirm in two minutes. The file-by-file routing
1921
1934
  // still prints after this - agents edit prefixes; people read this.
1922
1935
  function printDebriefReview(text, eng) {
@@ -1934,7 +1947,7 @@ function printDebriefReview(text, eng) {
1934
1947
  if (type === 'decision') {
1935
1948
  const who = approvedStamp(body)
1936
1949
  const core = previewLine(stripApprovedStamp(body), 90)
1937
- buckets.decided.push(who ? `${core} (approved ${who})` : `${core} (unconfirmed)`)
1950
+ buckets.decided.push(!hasSource(body) ? `${core} (CLAIM - source missing; unconfirmed)` : who ? `${core} (approved ${who}; source recorded)` : `${core} (unconfirmed; source recorded)`)
1938
1951
  } else if (type === 'ask') {
1939
1952
  buckets.asked.push(previewLine(body, 100))
1940
1953
  } else if (type === 'scope') {
@@ -1946,7 +1959,8 @@ function printDebriefReview(text, eng) {
1946
1959
  } else if (type === 'next') {
1947
1960
  buckets.next.push(previewLine(body, 100))
1948
1961
  } else if (type === 'signer') {
1949
- buckets.signer.push(previewLine(body, 80))
1962
+ if (proposedSigner(body)) buckets.signer.push(previewLine(body, 80))
1963
+ else buckets.open.push(`signer authority missing or uncertain: ${previewLine(body, 80)}`)
1950
1964
  }
1951
1965
  }
1952
1966
  console.log('REVIEW (one screen - confirm once, then apply)\n')
@@ -2169,7 +2183,8 @@ function routeDebriefInput(eng, input, { dry, force, sealed = [], allowReplay =
2169
2183
  continue
2170
2184
  }
2171
2185
  if (type === 'signer') {
2172
- const who = body.replace(/\s+signs?(?:\s+off)?\b.*$/i, '').trim() || body.trim()
2186
+ const who = proposedSigner(body)
2187
+ if (!who) { ctxLines.push(line); continue }
2173
2188
  if (dry) {
2174
2189
  console.log(`→ success.md **Stakeholder who signs off:** ${previewLine(who)}`)
2175
2190
  console.log(`→ stakeholders.md ${previewLine(`- [${date}] ${who} signs off`)}`)
@@ -2262,6 +2277,7 @@ function runDebrief(args, eng) {
2262
2277
  const applyIdx = args.indexOf('--apply')
2263
2278
  const apply = applyIdx !== -1
2264
2279
  if (apply) args.splice(applyIdx, 1)
2280
+ if (smart && apply) throw new Error('run --smart first, review the proposal, then confirm with --apply in a separate command')
2265
2281
  let force = false
2266
2282
  const forceIdx = args.indexOf('--force')
2267
2283
  if (forceIdx !== -1) { force = true; args.splice(forceIdx, 1) }
@@ -37,6 +37,12 @@ function boundedSections(sections, maxBytes = DEFAULT_BYTES) {
37
37
  const available = maxBytes - Buffer.byteLength(out + footer) - 2
38
38
  const share = Math.max(0, Math.floor(available / (filled.length - i)))
39
39
  const text = filled[i]
40
+ // A truncation marker also consumes the budget. If the fair share cannot
41
+ // fit one, stop with one marker instead of overflowing for every section.
42
+ if (Buffer.byteLength(text) > share && share < Buffer.byteLength(OMITTED)) {
43
+ out += (out ? '\n\n' : '') + clipUtf8(text, Math.max(0, available - Buffer.byteLength(OMITTED))) + clipUtf8(OMITTED, Math.max(0, available))
44
+ break
45
+ }
40
46
  const next = Buffer.byteLength(text) <= share ? text : clipUtf8(text, Math.max(0, share - Buffer.byteLength(OMITTED))) + OMITTED
41
47
  out += (out ? '\n\n' : '') + next
42
48
  }
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "fdeops-ingest-mcp",
3
- "version": "3.31.0",
3
+ "version": "4.0.0",
4
4
  "private": true,
5
5
  "description": "Thin stdio MCP sink for FDEOps ingest (stage → propose → apply). Zero runtime dependencies.",
6
6
  "bin": {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "fdeops",
3
- "version": "3.31.0",
3
+ "version": "4.0.0",
4
4
  "description": "Client delivery tools for Forward Deployed Engineers. One @fde skill, local Markdown engagement records, and an offline dashboard for decisions, evidence, approvals, and next actions.",
5
5
  "bin": {
6
6
  "fdeops": "bin/install.js",
package/plugin.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "$schema": "https://agent-plugins.org/schemas/1.0.0/plugin.schema.json",
3
3
  "name": "fdeops",
4
- "version": "3.31.0",
4
+ "version": "4.0.0",
5
5
  "description": "Forward deployed engineering skills for AI coding agents. One @fde skill for the client work around the code. You confirm; then it lands in .fde/ on your laptop.",
6
6
  "author": {
7
7
  "name": "Subash Natarajan",
@@ -36,13 +36,13 @@ On someone else's site the work is not "write code, remember later." Every chang
36
36
 
37
37
  1. **Name it** in `decisions.md` (plan), or timebox the riskiest assumption and record what the POC proves.
38
38
  2. **Characterise their code** before you change it. Brownfield: their tests, their runner. Greenfield: the empty tree, first path they can click.
39
- 3. **Prove it on their staging.** Staging they operate, a screen the signer in `success.md` can reject.
39
+ 3. **Verify, then prove delivery.** Use their checks and the agreed representative environment; at the delivery checkpoint the signer in `success.md` can replay and reject the acceptance check. See `ship` for evidence requirements.
40
40
  4. **If a model judges:** `evals.md` Verdict SHIP before that change is done (eval-pack).
41
- 5. **Log delivery.** Outcome is promised → measured → accepted, not a green CI. Then go live with a rollback you have run (`ship`).
41
+ 5. **Log delivery.** Outcome is promised → measured → accepted, not a green CI. Then go live with a tested recovery path (`ship`).
42
42
 
43
- A throwaway file can skip the loop. Bound client work cannot.
43
+ Scale the loop to the change. A routine, reversible fix within confirmed scope reuses the existing outcome, signer, acceptance criteria, and engineering plan; batch its verification into a concise delivery receipt. It does not need a new sponsor decision or staging ceremony per edit. New outcomes, changed acceptance or authority, and production release decisions still need the relevant confirmation and evidence. This does not bypass confirmation for judgment written into the engagement record.
44
44
 
45
- **Skip is loud.** Bound + a change that will ship + no this-turn line in `delivery.md` = not done. Say that. Do not call it shipped. A coding pack may write the function; `@fde` still owns done.
45
+ **Status is explicit.** Record what is implemented, verified, deployed, and accepted separately. A routine fix may be implementation-complete before release or customer acceptance; state what remains and attach the current verification receipt. A coding pack may write the function; `@fde` owns the engagement evidence.
46
46
 
47
47
  ## Working with an engineering pack
48
48
 
@@ -209,8 +209,8 @@ Ready to build with no `terrain.md` / plan: discover or plan first. Takeover wit
209
209
 
210
210
  - Never ask the FDE to pick a phase. That's your job.
211
211
  - Same six stages at any scale. Overlays carry the industry. Greenfield and brownfield change the first move inside ship, not the map.
212
- - Ground loop on a bound client: name → characterise → prove on their staging → go live → log. A coding pack may write the function. `@fde` still owns done. When they disagree, their repo and the signer win.
213
- - Do not call a change done until the signer in `success.md` can reject it on staging they operate. No this-turn receipt in `delivery.md` is a failed test, not a note to write later.
212
+ - Ground loop on a bound client: name → characterise → verify in the agreed environment → authorize release → log. A coding pack may write the function. `@fde` still owns done. When they disagree, their repo and the signer win.
213
+ - Customer delivery needs a replayable acceptance check in the agreed environment; reuse existing criteria for routine fixes. Missing evidence means unproven, not an observed test failure. Never equate implementation-complete with deployed or customer-accepted.
214
214
  - Read `context.md` before speaking. One sharp question - never a barrage.
215
215
  - Never invent people, meetings, or numbers - `unknown - ask:` beats a polished lie.
216
216
  - Every phase ends with its artifact written. No artifact, no "done."
@@ -20,19 +20,19 @@ Record in `trust-profile.md` under `## AI policy`.
20
20
 
21
21
  ## Model selection - choosing the right tool
22
22
 
23
- Never start with the most powerful model. Start with the cheapest that meets the quality bar.
23
+ Compare plausible approaches against the task’s quality, latency, privacy, and operating constraints. Optimize total cost per successful outcome, including retries, review, failures, and maintenance; call price alone can select the more expensive system.
24
24
 
25
- **The evaluation ladder:**
26
- 1. **Can rules solve it?** If yes, no model needed. Rules are debuggable, testable, and free.
27
- 2. **Can a small/fast model solve it?** (GPT-4o-mini, Claude Haiku, local models) - try this first. Cheaper, faster, easier to self-host.
28
- 3. **Does it need a frontier model?** (GPT-4o, Claude Sonnet/Opus, Gemini Pro) - only when the quality gap is measurable and justified.
29
- 4. **Does it need fine-tuning?** Only when: you have 500+ high-quality examples, the base model fails consistently on your domain, and the cost of inference at scale justifies the training cost.
25
+ **Candidate approaches (test the plausible ones, not a mandatory ladder):**
26
+ 1. **Can rules solve it?** If yes, no model needed. Include their implementation and maintenance cost.
27
+ 2. **Can a small/fast model solve it?** Include one when suitable for the task and hosting constraints; measure quality and total cost.
28
+ 3. **Would a more capable model improve the outcome?** It may reduce retries, supervision, or implementation complexity enough to justify its price. Verify current model availability and capabilities in official documentation.
29
+ 4. **Does it need fine-tuning?** Consider it when representative data and a held-out evaluation support a persistent domain gap. Compare against prompt/retrieval changes; justify dataset coverage and training, serving, and maintenance costs rather than assuming a fixed example count suffices.
30
30
 
31
31
  **Evaluation method (before choosing):**
32
- - Build a test set: 50-100 representative inputs with expected outputs.
33
- - Run every candidate model against the test set.
34
- - Score: accuracy, latency, cost per call, failure modes.
35
- - The cheapest model that scores above the quality threshold wins.
32
+ - Build a representative test set with expected outcomes and critical failure cases. Size it to diversity, consequence, and uncertainty; a small pilot set cannot establish rare-failure safety.
33
+ - Run a bounded shortlist against the same held-out cases; record model/version and settings.
34
+ - Score: task success, critical failures, latency, total cost per successful outcome, and failure modes, including repeated runs when variability matters.
35
+ - Select the approach that meets the agreed constraints with the best measured tradeoff. Record uncertainty and what would trigger re-evaluation.
36
36
 
37
37
  Write model selection rationale to `decisions.md`. Include: models tested, test set size, scores, cost comparison.
38
38
 
@@ -42,7 +42,7 @@ When any slice touches a model, embeddings, RAG, or an agent: create or update `
42
42
 
43
43
  **Minimum pack (do not grow until the minimum exists):**
44
44
  1. **Component + quality bar** - one sentence each; kill switch / fallback named.
45
- 2. **Golden cases** - 5-20 representative inputs with expected outputs and a pass rule. Prefer real production-shaped data (sanitized).
45
+ 2. **Golden cases** - representative inputs with expected outputs and a pass rule. 5-20 can seed a pilot, not certify readiness; expand for risk and coverage. Prefer real production-shaped data (sanitized).
46
46
  3. **Failure modes** - at least the silent ones: hallucination/ungrounded, retrieval miss (if RAG), drift, cost runaway.
47
47
  4. **Pass/fail** - dated run; Verdict **SHIP** or **NO-SHIP**; critical fails must be 0.
48
48
  5. **HITL gate** - which decisions need human review before action (align with `trust-profile.md`). Empty when policy requires review → NO-SHIP.
@@ -59,9 +59,9 @@ When the AI needs to answer questions about the client's data:
59
59
  3. **Generate** - chunks + query → LLM → answer with citations
60
60
 
61
61
  **Common failure modes:**
62
- - **Chunk size wrong.** Too small = lost context. Too large = noise drowns signal. Start at 500-1000 tokens with 100-token overlap.
62
+ - **Chunk size wrong.** Too small = lost context. Too large = noise drowns signal. Choose boundaries from document structure and answer needs; tune size, overlap, and top-K against retrieval and answer-quality evals, latency, and context limits.
63
63
  - **No citation/grounding.** If the model can't point to where it found the answer, you can't verify it. Always require source attribution.
64
- - **Stale index.** Documents update, embeddings don't. Define the refresh cadence. Real-time for critical data, daily for reference docs.
64
+ - **Stale index.** Documents update, embeddings don't. Define refresh and deletion handling from source update patterns and acceptable staleness; test them.
65
65
  - **Retrieval miss.** The right document exists but wasn't retrieved. Test with known-answer queries where the answer IS in the corpus - if retrieval misses these, the embedding model or chunking strategy needs work.
66
66
 
67
67
  ## Agent and agentic systems
@@ -71,23 +71,23 @@ When the AI takes actions (not just generates text):
71
71
  **Safety principles:**
72
72
  - **Least privilege.** An agent gets the minimum permissions needed. Never give an agent admin access "for convenience."
73
73
  - **Confirmation gates.** Any destructive or irreversible action requires human confirmation. Delete, send, transfer, publish = confirm before execute.
74
- - **Observability of reasoning.** Log the agent's chain of thought, tool calls, and decisions. When it does something wrong, you need to see why.
74
+ - **Observable execution.** Record tool/action summaries, versions, timing, cost, outcomes, validation results, and concise decision rationale. Do not request or store hidden chain-of-thought. Minimize and redact logged inputs/outputs; apply the client’s access, retention, and data policies. Never log raw `<private>` content or secrets.
75
75
  - **Deterministic fallbacks.** When the agent fails or is uncertain, it falls back to a known-safe behavior (queue for human review, return a safe default, do nothing). "The agent got confused and did something unexpected" is never acceptable in production.
76
- - **Cost caps.** Agents in loops can burn through API budgets. Hard-cap per request, per user, per hour. Alert at 50% of cap.
76
+ - **Cost caps.** Agents in loops can burn through API budgets. Set request and aggregate budgets with bounded retries and stopping conditions. Choose alert thresholds early enough for the owner to act.
77
77
 
78
78
  ## AI governance - responsible deployment
79
79
 
80
80
  **Before production:**
81
- - **Bias testing.** Run the model on demographic-varied inputs. If outputs differ by protected characteristic, it doesn't ship.
81
+ - **Bias testing.** Run the model on demographic-varied inputs. Define relevant groups, harms, and acceptable disparity with the responsible owner; investigate material differences and block unresolved critical harm. Aggregate accuracy alone is insufficient.
82
82
  - **Explainability.** Can you explain to a non-technical stakeholder why the model made a specific decision? If not, it's a black box - some jurisdictions and industries prohibit this.
83
83
  - **Model card.** Document: what the model does, what data it was trained/tuned on, known limitations, failure modes, who owns it. One page. Required before production.
84
84
  - **Kill switch.** Every AI component must be disable-able without taking down the feature it powers. The fallback path (rule-based, human-routed, or gracefully degraded) must work when the AI is off.
85
85
 
86
86
  **In production:**
87
- - **Drift monitoring.** Compare production outputs weekly against the baseline quality. Models don't break - they slowly get worse as the world changes around them.
87
+ - **Drift monitoring.** Compare production outputs against baseline quality on a cadence matched to traffic, drift risk, and impact. Quality can deteriorate gradually or fail abruptly after model, data, tool, or policy changes; monitor both patterns.
88
88
  - **Feedback collection.** Thumbs up/down, corrections, escalations. This is your retraining signal AND your quality metric.
89
89
  - **Cost monitoring.** Track: tokens consumed, calls made, cost per user, cost per feature. AI costs surprise everyone at scale.
90
- - **Incident response.** When the AI produces harmful/wrong output: disable (kill switch), investigate (logged reasoning), fix (prompt/model/data), restore. Define this BEFORE it happens.
90
+ - **Incident response.** When the AI produces harmful/wrong output: disable (kill switch), investigate (sanitized execution traces and observed outcomes), fix (prompt/model/data), restore. Define this BEFORE it happens.
91
91
 
92
92
  ## Writes
93
93
 
@@ -96,10 +96,10 @@ When the AI takes actions (not just generates text):
96
96
  ## Principles
97
97
 
98
98
  - AI degrades silently. Monitor outputs, not just uptime.
99
- - Start with the cheapest model that meets the quality bar.
99
+ - Choose by measured quality and total cost per successful outcome, within policy and latency constraints.
100
100
  - No golden set, no AI ship (`evals.md` Verdict SHIP).
101
101
  - Every AI component needs a kill switch and a fallback path.
102
- - Log reasoning, not just results. Debug AI from its decisions.
102
+ - Debug from privacy-safe observable execution and concise rationale, never hidden chain-of-thought.
103
103
  - Drift is inevitable. Define the detection method before shipping.
104
104
  - Cost at scale ≠ cost at pilot. Model the 10× number before committing.
105
105
  - Bias testing is a pre-production gate, not a post-launch audit.
@@ -14,13 +14,13 @@ Intelligence without evidence is token-maxing with a nicer name. An FDE earns tr
14
14
 
15
15
  **1. Scope the judgement surface.** One sentence: which step uses model judgement, and what must never be autonomous.
16
16
 
17
- **2. Build a golden set (minimum 5-20 for a slice; prefer 50-100 before broad scale).** For each case:
17
+ **2. Build a golden set sized to risk and coverage.** A pilot may start with 5-20 cases; neither that range nor 50-100 proves broad-scale safety. Include representative segments, boundary cases, and known critical failures. Expand based on observed failure modes and uncertainty; hold out cases from tuning and repeat runs when variability matters. For each case:
18
18
  - input (sanitized - no `<private>` raw values)
19
19
  - expected outcome or expert-approved acceptance note
20
20
  - pass rule (exact / contains / short rubric)
21
21
  - source: real historical example / expert label / staged fixture
22
22
 
23
- **3. Score pass/fail, not vibes.** Run the suite. Record count pass / fail. Failures get a failure-mode tag (missing data, wrong record, format drift, hallucination, retrieval miss, unsafe action, other).
23
+ **3. Score pass/fail, not vibes.** Define the quality threshold and critical failure rules before running the suite. Run it; record count pass / fail, segment coverage, and limitations. Failures get a failure-mode tag (missing data, wrong record, format drift, hallucination, retrieval miss, unsafe action, other).
24
24
 
25
25
  **4. Human-in-the-loop gate.** Name which outcomes require human approve before side effects. Judgement that has a side effect (write, send, transfer, ticket, deploy, pay, page) is **NO-SHIP** without a named human on their side in the loop. Do not write "none - allowed under policy" to bless lights-out write-access. Staging may run a supervised loop with a kill switch, a cost cap, and a golden set from **their** failures. Production stays gated until they have a written policy, a named owner, and dated eval receipts on real traffic.
26
26
 
@@ -4,7 +4,7 @@
4
4
 
5
5
  **Read first:** `reality.md`, `success.md`, `terrain.md`, `stakeholders.md`. Load `business-case.md` if poc produced one. Not the full folder.
6
6
 
7
- **Before committing a plan or building:** run `fde doctor --ready`. Missing binary success or a named customer-side signer blocks progression: review the proposed acceptance check and authority with the FDE first. Use a test/input and observable pass/fail under **Done when:** or **Acceptance check:**. A number, role, or successful demo alone is insufficient. Do not invent missing facts to pass lint.
7
+ **Before a new delivery plan or material scope change:** run `fde doctor --ready`. Missing binary success or a named customer-side signer blocks progression: review the proposed acceptance check and authority with the FDE first. Use a test/input and observable pass/fail under **Done when:** or **Acceptance check:**. A number, role, or successful demo alone is insufficient. Do not invent missing facts to pass lint. Routine reversible fixes within confirmed scope reuse the existing signer, acceptance criteria, and engineering plan; record verification without reopening settled decisions.
8
8
 
9
9
  ## Validation gate (confirm understanding, clarify where it elevates)
10
10
 
@@ -34,7 +34,7 @@ An FDE plan is not a sprint backlog. The technical sequence is the easy part. Th
34
34
 
35
35
  **3. One user action per change.** Each task delivers something visible and testable ("user submits form, sees it saved"), never a layer ("build the database layer"). See `ship`.
36
36
 
37
- **4. Size to 30-90 minutes, PR-sized.** Longer = two tasks. Each task implementable, testable, reviewable without a thousand-line diff.
37
+ **4. Size to a coherent, verifiable outcome.** Split unrelated work and tasks too complex to review or recover safely. Use bounded review sections for large cohesive changes; elapsed time and line count are signals to examine, not universal limits.
38
38
 
39
39
  **5. AI components get explicit eval tasks.** "Output validated on 50 real production examples," "fallback tested under model unavailability," "inputs/outputs logging to <destination>" - these are pre-conditions of shipping, in the plan before build starts.
40
40
 
@@ -8,7 +8,7 @@ Engagement review ≠ product-company review: a codebase you don't own, systems
8
8
 
9
9
  ## Pre-flight: is this reviewable?
10
10
 
11
- Thousands of lines or dozens of unrelated files → **stop**, recommend the split in `decisions.md`, route to plan. Review loops fail on huge diffs. Ask: one agreed task, or did scope merge mid-build?
11
+ Check whether the diff has one agreed intent and can be reviewed with the available evidence. Split unrelated work; for a large cohesive change, separate generated/mechanical output from behavioral changes and review in bounded sections. Size is a warning to investigate, not a universal stop threshold.
12
12
 
13
13
  ## Stage 1 - did we build what we agreed? (you do this work)
14
14
 
@@ -43,19 +43,19 @@ Five dimensions, line-specific ("line 47 fails under concurrent writes - no lock
43
43
  - **Correctness** - does what it says; edge cases; error paths traced.
44
44
  - **Blast radius** - what breaks at 2am; downstream systems; failure mode loud (errors surface) or silent (data corrupts over time)?
45
45
  - **Security** - input validation at boundaries; no secrets in logs; no new attack surface; `trust-profile.md` sensitivity classes respected.
46
- - **Rollback** - revertible in under 5 minutes, documented? "We'd need a data migration to roll back" is a blocker.
47
- - **AI policy & components** - human-review requirements honoured; model output treated as untrusted until validated; fallback exists; inputs/outputs logged; outputs bounded so a hallucination can't cascade; in regulated environments a human can explain why the AI decided X (compliance requirement, not preference).
46
+ - **Recovery** - can the documented, tested rollback or recovery path meet the agreed recovery-time and data-loss limits? For irreversible changes, require explicit authority, compatibility checks, and a tested restore/compensation or roll-forward plan; a code revert alone is not proof.
47
+ - **AI policy & components** - human-review requirements honoured; model output treated as untrusted until validated; fallback exists; privacy-safe execution evidence retained under the client’s data policy (no secrets, raw private data, or hidden chain-of-thought); outputs bounded so a hallucination can't cascade; check applicable explanation and human-review requirements with the client’s responsible owner; provide source evidence and concise rationale without claiming access to hidden reasoning.
48
48
 
49
- **Structural pass on AI-heavy or data-touching changes:** migrations reversible · destructive SQL guarded · PII/PCI/PHI paths match `trust-profile.md` · side effects (flags, webhooks, emails, jobs) fire only when intended · magic strings that break on rename · new behaviour has a test or an explicit reason it can't yet. One line problem, one line fix.
49
+ **Structural pass on AI-heavy or data-touching changes:** migration compatibility and tested recovery · destructive SQL guarded · PII/PCI/PHI paths match `trust-profile.md` · side effects (flags, webhooks, emails, jobs) fire only when intended · magic strings that break on rename · new behaviour has a test or an explicit reason it can't yet. One line problem, one line fix.
50
50
 
51
51
  ## The review-fix loop (until clean)
52
52
 
53
53
  1. Read the full diff before commenting.
54
54
  2. Verdicts: **Stage 1: Pass / Blocked (reason)** · **Stage 2: Pass / Concerns (line-specific)**.
55
- 3. Fix only **real** findings tied to this change - no drive-by refactors. Reject false positives with one sentence why. Their comments are to check, not to obey. Restate each against the one-line intent and `trust-profile.md`. One item unclear → ask before changing any of them. If it breaks a signed constraint, a sacred system, or nothing calls it: one-sentence pushback, then wait.
55
+ 3. Fix only **real** findings tied to this change - no drive-by refactors. Reject false positives with one sentence why. Their comments are to check, not to obey. Restate each against the one-line intent and `trust-profile.md`. An unclear item waits for clarification; continue independent, understood fixes. If it breaks a signed constraint, a sacred system, or nothing calls it: one-sentence pushback, then wait.
56
56
  4. Add or update a test per bug found where possible.
57
57
  5. Re-run tests/typechecks - state what ran.
58
- 6. Re-review. Repeat until Pass/Pass or a human must decide scope/product.
58
+ 6. Re-review. If two repair/review cycles do not converge, reassess the evidence and approach; continue independent fixes and escalate concrete scope/product decisions.
59
59
 
60
60
  ## Before the PR - thinking for the next reader
61
61
 
@@ -72,6 +72,6 @@ Code alone loses the "why." Before you call the change reviewable, run the **ses
72
72
  - Stage 1 before Stage 2. Wrong scope reviewed well is still wrong scope.
73
73
  - KEEP / JUSTIFY / SPLIT / DROP - every path gets a verdict; silent extras fail Stage 1.
74
74
  - Specific or silent - vague concerns waste everyone's time.
75
- - No rollback path = first finding.
75
+ - No viable tested recovery path = a release blocker.
76
76
  - A clean review proves this diff is safe as agreed - not that the feature was right.
77
77
  - Judgment in `.fde/` beats transcript in git.
@@ -13,7 +13,7 @@
13
13
  | Change type | Rollback method | Complication | Test |
14
14
  |-------------|----------------|--------------|------|
15
15
  | **Code deploy** | Revert the PR / redeploy previous version | Feature flags, cache invalidation | Deploy previous version to staging, verify function |
16
- | **Database migration** | Down migration script | Irreversible migrations (column drops, data transforms) | Run down migration on staging copy |
16
+ | **Database migration** | Compatible rollback, restore, or roll-forward | Irreversible transforms, concurrent writes, old/new schema compatibility | Rehearse on representative staging data; verify integrity, elapsed time, and possible data loss |
17
17
  | **Config change** | Restore previous config | Propagation delay, dependent service restarts | Flip config, verify all services pick it up |
18
18
  | **Infrastructure** | Terraform/Pulumi rollback or manual | State drift, dependent resources | Plan the rollback, review the diff |
19
19
  | **Data backfill** | Restore from backup or reverse script | Mixed old/new data states | Run reverse on a 100-row sample |
@@ -96,7 +96,7 @@ One statement: "Rollback tested on staging. Time: <N minutes>. Result: <pass/fai
96
96
  ## Principles
97
97
 
98
98
  - A rollback plan that hasn't been tested is a wish.
99
- - Time the drill. If it takes 45 minutes on staging, it takes 90 in production at 2am.
99
+ - Time the drill and account for differences in production scale and operating conditions; do not assume a fixed multiplier.
100
100
  - Identify the irreversible components and name the compensating action.
101
101
  - The drill report is evidence for the change ticket and the team's confidence.
102
102
  - A drill that fails is a success - you found the problem before production did.
@@ -10,11 +10,11 @@ Do not ask them to pick a mode. Name where you are, then start at the matching s
10
10
  - On staging, the signer in `success.md` can reject it → **go-live**
11
11
  - Prod is the question → **go-live**. Do not start a second change.
12
12
 
13
- If going live, opening question: **has anyone actually *run* the rollback, or is it still a slide?** If only planned, that's today's work - say so plainly.
13
+ If going live, check the evidence for the recovery path: has rollback, restore, compensation, or roll-forward been exercised under representative conditions? If only planned, validate it before release. Reuse applicable drill evidence when the mechanism and relevant conditions are unchanged; record why it applies.
14
14
 
15
15
  A bounded experiment that tests an assumption is `poc`. This skill turns a validated direction into a maintainable change on a repo they will own, then production. Inspect existing prototype code and retain suitable tested parts; replace unsafe shortcuts based on evidence. A successful demo alone does not satisfy the readiness gates below.
16
16
 
17
- **Before committing a plan or building:** run `fde doctor --ready`. Missing binary success or a named customer-side signer blocks progression: review the proposed acceptance check and authority with the FDE first. Use a test/input and observable pass/fail under **Done when:** or **Acceptance check:**. A number, role, or successful demo alone is insufficient. Do not invent missing facts to pass lint.
17
+ **Before a new delivery plan or material scope change:** run `fde doctor --ready`. Missing binary success or a named customer-side signer blocks progression: review the proposed acceptance check and authority with the FDE first. Use a test/input and observable pass/fail under **Done when:** or **Acceptance check:**. A number, role, or successful demo alone is insufficient. Do not invent missing facts to pass lint. Routine reversible fixes within confirmed scope reuse the existing signer and acceptance check; record verification without restarting approval. New judgment in the record still requires confirmation.
18
18
 
19
19
  ## Field (name it once, then the same loop)
20
20
 
@@ -22,18 +22,18 @@ A bounded experiment that tests an assumption is `poc`. This skill turns a valid
22
22
  |--|------------|------------|
23
23
  | What you touch | Code they already run | A new path or empty tree they will own |
24
24
  | First move | Characterise their tests, their runner, the workaround in `terrain.md` | First path a user can click. Not the whole product. |
25
- | Proof | Their staging, a screen they already use | Their staging, or the environment they will operate. Local demo is not delivery. |
26
- | Undo | Revert this change on its own | Same. If you cannot undo it, the design is coupled. |
25
+ | Proof | Agreed representative environment and replayable acceptance check | Agreed representative environment and replayable acceptance check; record what remains untested before release |
26
+ | Undo | Revert this change on its own | Name rollback or tested recovery; identify irreversible effects and required authority. |
27
27
 
28
28
  Skip POC only when the killer assumption already lives in the repo (typical brownfield). If the bet is unproven, `poc` first.
29
29
 
30
- **Done means:** the signer in `success.md` can reject this on staging they operate. A green check on your laptop is not delivery. Do not start the next change until this one is rejectable.
30
+ **Customer delivery means:** the signer in `success.md` can replay and reject the agreed acceptance check in an environment they operate. A green check on your laptop proves only what ran there. Routine fixes can share a delivery checkpoint; distinguish implementation, verification, deployment, and acceptance.
31
31
 
32
32
  If `terrain.md` **Data estate** lists a **Blocker** this change depends on (source or pipe): stop. That is discover, not ship. Do not build a path they cannot feed.
33
33
 
34
34
  ## Method - one change they can see
35
35
 
36
- One change = one thing a user can do, with a test, that you can revert on its own. Not "all the APIs, then all the UI." Not a 2,000-line dump. A PR is how this often lands. It is not the job. The job is the change they can see.
36
+ One change = one coherent outcome with observable verification and a bounded recovery path. Prefer vertical slices that can be reviewed and exercised independently. A PR is how this often lands. It is not the job. The job is the change they can see.
37
37
 
38
38
  ```
39
39
  BAD (layers):
@@ -49,7 +49,7 @@ GOOD (one user action each):
49
49
  4: Admin can void a payment (auth + logic + UI) - testable
50
50
  ```
51
51
 
52
- Each change is independently revertible.
52
+ Prefer independently revertible changes. When data or external effects cannot be undone, name the dependency, containment, tested recovery, and authorized owner before release.
53
53
 
54
54
  **Before you start this change:**
55
55
 
@@ -58,7 +58,7 @@ Each change is independently revertible.
58
58
  - [ ] Rollback named: revert this change, or something more specific
59
59
  - [ ] No dependency on an unmerged change (if dependent, state it and land in order)
60
60
  - [ ] `Kill if` is written - the observation that stops this change
61
- - [ ] Before-receipt captured: the failing output, number, or screen as it is today, dated in `delivery.md`, before you change anything
61
+ - [ ] Before-state evidence identified: the relevant failing output, number, or behavior, with its source and date. For a routine fix within confirmed scope, reference applicable existing evidence and batch the delivery receipt; capture new evidence when the relevant behavior or conditions changed. Never imply an old check was rerun.
62
62
  - [ ] Open PRs and uncommitted work in the area checked (`gh pr list`, `gh pr diff <n> --name-only`); overlap goes to `decisions.md` before you start
63
63
 
64
64
  Your coding pack writes the function. This skill owns done. When they disagree with this repo, the repo wins.
@@ -69,34 +69,25 @@ Your coding pack writes the function. This skill owns done. When they disagree w
69
69
  Read existing code in the area (search before creating)
70
70
  → Characterise what is already there (their tests, their runner; greenfield: the empty tree)
71
71
  → Implement the smallest path that works
72
- → Prove it on their staging (below)
72
+ → Verify the change; demonstrate at the agreed delivery checkpoint (below)
73
73
  → Cleanup pass (dedupe, simplify - behaviour unchanged)
74
74
  → Self-review against acceptance criteria
75
75
  → Commit with a message the client's team can read
76
76
  → Update decisions.md + delivery.md
77
77
  ```
78
78
 
79
- **Prove it on their staging.** A green check on your laptop is not delivery.
79
+ **Verify the change and prove customer delivery.** Match evidence to the reviewed revision, environment, and acceptance criteria.
80
80
 
81
- - Run **their** test command, on **their** CI, with **their** fixtures. Write the command and the result in `delivery.md` in this turn. You do not add a runner they will not keep. If you have not run their command in this turn, you cannot write that it passed. Last session's green, "should pass," and "looks correct" are not a receipt. Missing this-turn line = not proven. Same as a failing test.
82
- - If the signer in `success.md` cannot reject this on a screen they already use, it is not proven.
83
- - Staging they operate beats a local demo. If you have no staging: `unknown - ask:` who owns an environment, then stop pretending it shipped.
84
- - **Monday-shaped data.** Staging that is empty, synthetic, or last quarter is not next Tuesday. Before go-live, write what staging is missing (volume, PII, the batch that only runs in prod, the account that only exists in the warehouse) and what that means for the kill test. If the signer cannot reject it on a screen they already operate, with data that looks like next Tuesday, it is not proven.
81
+ - Use **their** test commands, fixtures, and CI. Record the command, result, revision, environment, and run date in `delivery.md`. Reuse existing evidence only when the relevant code and conditions are unchanged, citing why it still applies; never claim it was rerun. Run affected checks for changed behavior and required release checks before deployment. Missing evidence means unproven, not an observed failure.
82
+ - At the agreed delivery checkpoint, the signer in `success.md` must be able to replay and reject the acceptance check using an interface they operate (screen, API, report, or equivalent). Routine fixes can share that checkpoint; passing tests alone does not establish customer acceptance.
83
+ - Prefer staging they operate. When unavailable, use an agreed, permitted representative test environment, record its owner and limitations, and resolve material release-evidence gaps before production. A local demonstration is not deployment.
84
+ - **Representative data.** Use permitted sanitized or synthetic fixtures that exercise relevant volumes, edge cases, and operating paths. Before go-live, record gaps such as batch timing, distribution, or production-only dependencies and their impact on the acceptance and abort checks. Resolve material gaps or explicitly narrow the release; never load sensitive production data merely to make a demo realistic.
85
85
  - Model in the path: `eval-pack` until `evals.md` says SHIP. Do not skip because "it looked right in chat."
86
86
  - A model drafts. A named human on their side ships. No unsupervised loop on their production. If the brief demands lights-out write-access, that is `who-decides` / `hold-scope`, not ship.
87
87
 
88
88
  The proof is whatever this client already believes, plus one new receipt they can replay.
89
89
 
90
- **Size.** Each change targets:
91
-
92
- | Metric | Target | Why |
93
- |--------|--------|-----|
94
- | Lines changed | 100-300 | Reviewable in one sitting |
95
- | Time to implement | 30-90 minutes | Testable before context decays |
96
- | Files touched | 1-5 | Blast radius stays containable |
97
- | Tests added | ≥1 per new behaviour | Proves this change; guards against regression |
98
-
99
- Larger than 300 lines → split first. "It's all connected" means the design needs work, not a bigger dump.
90
+ **Size by reviewability and risk.** Keep one coherent intent, bounded context, and observable acceptance checks. Split unrelated behavior or work whose recovery and review cannot be understood together. Diff size and elapsed time are warning signals, not hard gates: generated changes may be large and low risk; a one-line permission change may be critical. Use the repository’s checks and add meaningful coverage for changed behavior, rather than a test-count quota.
100
91
 
101
92
  **Show it.** Every 2-3 changes, something the customer can see: an endpoint they can hit, a UI they can click, a metric that moved, a risk that was retired. Technical progress invisible to stakeholders is trust decay. `delivery.md` gets updated after every visible change.
102
93
 
@@ -106,7 +97,7 @@ Larger than 300 lines → split first. "It's all connected" means the design nee
106
97
  - If it's NOT in `decisions.md`: log it as a scope receipt (see `hold-scope.md`), don't touch it.
107
98
  - Ugly code outside this change stays ugly. That is discipline, not laziness.
108
99
 
109
- After each change: tests pass (state the command and result), acceptance criteria met, blast radius as declared, `Kill if` still false. After every 2-3: what did they see, and what's their signal? Then, when the signer can reject it on their staging, go-live below.
100
+ After each change: required checks pass with applicable evidence, acceptance criteria evaluated, blast radius as declared, `Kill if` still false. At the agreed delivery checkpoint: what did they see, and what is their signal? Before production, complete the go-live gates below.
110
101
 
111
102
  ---
112
103
 
@@ -141,7 +132,7 @@ Score each dimension green/amber/red. This is the gate, not a suggestion:
141
132
  | Dimension | Green | Amber | Red |
142
133
  |-----------|-------|-------|-----|
143
134
  | Tests | All pass on deploy branch | Flaky tests skipped with justification | Failures present or tests not run |
144
- | Rollback | Tested end-to-end (not planned - TESTED) | Documented but untested | No rollback path defined |
135
+ | Recovery | Applicable tested rollback/restore/compensation/roll-forward meets agreed recovery and data-loss limits | Documented; drill evidence needs refresh | No viable recovery, failed drill, or irreversible effects lack explicit authority |
145
136
  | Sign-off | Stakeholder approval in `decisions.md` with date | Verbal approval, not logged | No approval sought |
146
137
  | Runbook | Exists and someone other than you has read it | Exists but unreviewed | Missing |
147
138
  | Monitoring | Alerts configured, owner named, dashboard live | Alerts configured, no named owner | No monitoring |
@@ -211,16 +202,16 @@ grep -rnE "(api[_-]?key|secret|password|token)\s*[:=]\s*['\"][^'\"]{8,}" \
211
202
  --include="*.js" --include="*.ts" --include="*.py" --include="*.env" \
212
203
  --include="*.yaml" --include="*.json" . | grep -vE "example|template|test" | head
213
204
  ```
214
- - DB migrations reversible.
215
- - Rollback documented **and tested**.
205
+ - DB migrations checked for compatibility, data loss, and old/new application coexistence. Prefer expand/contract for destructive changes. Irreversible steps require explicit authority and a tested restore, compensation, or roll-forward plan.
206
+ - Recovery documented **and tested**, with acceptable recovery time and data loss.
216
207
  - Monitoring alerts configured, someone watching.
217
208
  - Team knows the deploy is happening.
218
- - Not a Friday unless genuine emergency with someone on call.
209
+ - Deploy window has staffed observation and recovery coverage appropriate to the risk; respect the client’s change calendar and business-critical periods.
219
210
  - **Change approval (CAB) environments:** window open, ticket approved. In banking/healthcare/gov, deploying outside an approved window is a compliance finding even when the deploy succeeds. "We didn't know there was a CAB process" is not a defence - find out before the deploy date.
220
211
 
221
212
  ## Method - the deploy
222
213
 
223
- **Canary:** 1-5% of traffic, ≥10 minutes. Watch error rate, latency, and **the business metric this change affects**. Anything looks wrong → roll back immediately; investigate safely; redeploy when confident. Never investigate during the canary. Then stage up: 5% → 25% → 100%, each confirmed stable.
214
+ **Rollout:** Choose canary, blue/green, staged cohorts, or the client’s proven release mechanism based on isolation, traffic, and failure cost. For a canary, set cohort size, exposure cap, observation duration, minimum sample, and advance/abort thresholds before starting; allow for delayed and batch effects. Watch errors, latency, and **the business metric this change affects**. Breached thresholds or critical harm → halt expansion and execute the tested recovery/containment plan; investigate after exposure is controlled. Advance only with sufficient evidence and a named operator.
224
215
 
225
216
  **Canary receipt** (write it, or the canary did not happen): what was watched, on whose dashboard, for how long, and that the next change did not start in the window. If prod is a CAB console, vendor button, or their pipeline, write the owner and the click path - the host agent does not get to pretend it shipped.
226
217
 
@@ -275,7 +266,7 @@ Never skip a step. The sponsor always wants to skip from pilot to standard - tha
275
266
  Adoption isn't a handoff-stage problem - it starts while you are still writing the change. Software that launches to silence is software that gets decommissioned.
276
267
 
277
268
  **During the change:**
278
- - **Feature flags from day one.** Every new capability behind a flag. Ship to 5% of users first. Watch behavior before opening to 100%.
269
+ - **Controlled exposure.** Use a feature flag or equivalent isolation when it reduces rollout risk. Choose cohorts and expansion criteria from traffic and impact; name the flag owner and removal point.
279
270
  - **Feedback loops built in.** A thumbs-up/down, a "was this helpful?", a usage counter. Instrument adoption, don't assume it.
280
271
  - **Resistance signals.** Watch for: workaround creation (they built a spreadsheet instead of using the tool), drop-off after day 3 (onboarding fails), vocal detractors (one influential skeptic can kill adoption). Address these before launch, not after.
281
272
 
@@ -290,13 +281,13 @@ Adoption isn't a handoff-stage problem - it starts while you are still writing t
290
281
 
291
282
  **`decisions.md`** - each change: what was implemented, what was tested, what was deferred, `Kill if`.
292
283
 
293
- **`delivery.md`** - each visible change in business language; then the deployment record: what shipped, when, rollback procedure, pulse definition, **scale-readiness assessment, and adoption metrics**. Written for whoever inherits the system.
284
+ **`delivery.md`** - each visible change in business language; then the deployment record: what shipped, when, recovery procedure, pulse definition, **scale-readiness assessment, and adoption metrics**. Written for whoever inherits the system.
294
285
 
295
286
  ## Checkpoint
296
287
 
297
- After each change: tests pass, acceptance criteria met, blast radius as declared, `Kill if` still false, proven on staging they operate.
288
+ After each change: required checks pass, acceptance criteria evaluated, blast radius as declared, `Kill if` still false. Batch routine fixes at the agreed delivery checkpoint; record staging and customer acceptance separately.
298
289
 
299
- Before 100% live: canary clean, business metric verified, pulse written into `delivery.md`. Also green: value bucket named, audit receipt dated, eval receipt **n/a or pass**, **intent vs diff clean** (no unresolved SPLIT/DROP). Missing any of those → not green. For enterprise-scale: scale-readiness gate passed before broad rollout.
290
+ Before full exposure: the chosen rollout’s advance criteria are met with sufficient observation and business-metric evidence, and the pulse is written into `delivery.md`. Also green: value bucket named, audit receipt dated, eval receipt **n/a or pass**, **intent vs diff clean** (no unresolved SPLIT/DROP). Missing any of those → not green. For enterprise-scale: scale-readiness gate passed before broad rollout.
300
291
 
301
292
  ## Worked example
302
293
 
@@ -315,10 +306,10 @@ Greenfield is the same loop with an empty tree: first path a user can click, on
315
306
  ## Principles
316
307
 
317
308
  - One user action per change. Layers are untestable until assembled.
318
- - On their staging, and you can undo it. Local green is not delivery.
309
+ - Prove the agreed outcome in their environment and test recovery. Local green is not customer delivery.
319
310
  - The ugly code outside this change stays ugly. That's discipline, not laziness.
320
- - A deployment without a tested rollback is reckless.
321
- - Roll back on any canary anomaly; investigate safely.
311
+ - A deployment needs tested recovery within agreed time and data-loss limits; irreversible effects require explicit authority.
312
+ - Halt expansion on breached thresholds or critical harm; contain exposure with the tested recovery plan before investigating.
322
313
  - Verify the business metric, not just the technical one.
323
314
  - No value bucket, no green ship. No pulse, no done.
324
315
  - Diff larger than the stated intent without KEEP/JUSTIFY receipts = fix-first.