@codyswann/lisa 2.313.2 → 2.315.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (94) hide show
  1. package/dist/core/upstream-evidence-manifest.d.ts.map +1 -1
  2. package/dist/core/upstream-evidence-manifest.js +27 -5
  3. package/dist/core/upstream-evidence-manifest.js.map +1 -1
  4. package/package.json +1 -1
  5. package/plugins/lisa/.claude-plugin/plugin.json +1 -1
  6. package/plugins/lisa/.codex-plugin/plugin.json +1 -1
  7. package/plugins/lisa/.codex-plugin/skills/lisa-codify-verification/SKILL.md +23 -7
  8. package/plugins/lisa/agents/test-specialist.md +3 -1
  9. package/plugins/lisa/agents/verification-specialist.md +4 -1
  10. package/plugins/lisa/rules/eager/empirical-inquiry.md +1 -0
  11. package/plugins/lisa/rules/eager/falsifiable-checks.md +26 -0
  12. package/plugins/lisa/rules/eager/stale-state-claims.md +26 -0
  13. package/plugins/lisa/rules/eager/verification.md +1 -0
  14. package/plugins/lisa/rules/reference/falsifiable-checks.md +92 -0
  15. package/plugins/lisa/rules/reference/stale-state-claims.md +88 -0
  16. package/plugins/lisa/skills/lisa-codify-verification/SKILL.md +23 -7
  17. package/plugins/lisa-agy/agents/test-specialist.md +3 -1
  18. package/plugins/lisa-agy/agents/verification-specialist.md +4 -1
  19. package/plugins/lisa-agy/plugin.json +1 -1
  20. package/plugins/lisa-agy/skills/lisa-codify-verification/SKILL.md +23 -7
  21. package/plugins/lisa-cdk/.claude-plugin/plugin.json +1 -1
  22. package/plugins/lisa-cdk/.codex-plugin/plugin.json +1 -1
  23. package/plugins/lisa-cdk-agy/plugin.json +1 -1
  24. package/plugins/lisa-cdk-copilot/.claude-plugin/plugin.json +1 -1
  25. package/plugins/lisa-cdk-cursor/.claude-plugin/plugin.json +1 -1
  26. package/plugins/lisa-copilot/.claude-plugin/plugin.json +1 -1
  27. package/plugins/lisa-copilot/agents/test-specialist.agent.md +3 -1
  28. package/plugins/lisa-copilot/agents/verification-specialist.agent.md +4 -1
  29. package/plugins/lisa-copilot/rules/eager/empirical-inquiry.md +1 -0
  30. package/plugins/lisa-copilot/rules/eager/falsifiable-checks.md +26 -0
  31. package/plugins/lisa-copilot/rules/eager/stale-state-claims.md +26 -0
  32. package/plugins/lisa-copilot/rules/eager/verification.md +1 -0
  33. package/plugins/lisa-copilot/rules/reference/falsifiable-checks.md +92 -0
  34. package/plugins/lisa-copilot/rules/reference/stale-state-claims.md +88 -0
  35. package/plugins/lisa-copilot/skills/lisa-codify-verification/SKILL.md +23 -7
  36. package/plugins/lisa-cursor/.claude-plugin/plugin.json +1 -1
  37. package/plugins/lisa-cursor/agents/test-specialist.md +3 -1
  38. package/plugins/lisa-cursor/agents/verification-specialist.md +4 -1
  39. package/plugins/lisa-cursor/rules/empirical-inquiry.mdc +1 -0
  40. package/plugins/lisa-cursor/rules/falsifiable-checks-reference.mdc +97 -0
  41. package/plugins/lisa-cursor/rules/falsifiable-checks.mdc +31 -0
  42. package/plugins/lisa-cursor/rules/stale-state-claims-reference.mdc +93 -0
  43. package/plugins/lisa-cursor/rules/stale-state-claims.mdc +31 -0
  44. package/plugins/lisa-cursor/rules/verification.mdc +1 -0
  45. package/plugins/lisa-cursor/skills/lisa-codify-verification/SKILL.md +23 -7
  46. package/plugins/lisa-expo/.claude-plugin/plugin.json +1 -1
  47. package/plugins/lisa-expo/.codex-plugin/plugin.json +1 -1
  48. package/plugins/lisa-expo-agy/plugin.json +1 -1
  49. package/plugins/lisa-expo-copilot/.claude-plugin/plugin.json +1 -1
  50. package/plugins/lisa-expo-cursor/.claude-plugin/plugin.json +1 -1
  51. package/plugins/lisa-harper-fabric/.claude-plugin/plugin.json +1 -1
  52. package/plugins/lisa-harper-fabric/.codex-plugin/plugin.json +1 -1
  53. package/plugins/lisa-harper-fabric-agy/plugin.json +1 -1
  54. package/plugins/lisa-harper-fabric-copilot/.claude-plugin/plugin.json +1 -1
  55. package/plugins/lisa-harper-fabric-cursor/.claude-plugin/plugin.json +1 -1
  56. package/plugins/lisa-nestjs/.claude-plugin/plugin.json +1 -1
  57. package/plugins/lisa-nestjs/.codex-plugin/plugin.json +1 -1
  58. package/plugins/lisa-nestjs-agy/plugin.json +1 -1
  59. package/plugins/lisa-nestjs-copilot/.claude-plugin/plugin.json +1 -1
  60. package/plugins/lisa-nestjs-cursor/.claude-plugin/plugin.json +1 -1
  61. package/plugins/lisa-openclaw/.claude-plugin/plugin.json +1 -1
  62. package/plugins/lisa-openclaw/.codex-plugin/plugin.json +1 -1
  63. package/plugins/lisa-openclaw-agy/plugin.json +1 -1
  64. package/plugins/lisa-openclaw-copilot/.claude-plugin/plugin.json +1 -1
  65. package/plugins/lisa-openclaw-cursor/.claude-plugin/plugin.json +1 -1
  66. package/plugins/lisa-phaser/.claude-plugin/plugin.json +1 -1
  67. package/plugins/lisa-phaser/.codex-plugin/plugin.json +1 -1
  68. package/plugins/lisa-phaser-agy/plugin.json +1 -1
  69. package/plugins/lisa-phaser-copilot/.claude-plugin/plugin.json +1 -1
  70. package/plugins/lisa-phaser-cursor/.claude-plugin/plugin.json +1 -1
  71. package/plugins/lisa-rails/.claude-plugin/plugin.json +1 -1
  72. package/plugins/lisa-rails/.codex-plugin/plugin.json +1 -1
  73. package/plugins/lisa-rails-agy/plugin.json +1 -1
  74. package/plugins/lisa-rails-copilot/.claude-plugin/plugin.json +1 -1
  75. package/plugins/lisa-rails-cursor/.claude-plugin/plugin.json +1 -1
  76. package/plugins/lisa-typescript/.claude-plugin/plugin.json +1 -1
  77. package/plugins/lisa-typescript/.codex-plugin/plugin.json +1 -1
  78. package/plugins/lisa-typescript-agy/plugin.json +1 -1
  79. package/plugins/lisa-typescript-copilot/.claude-plugin/plugin.json +1 -1
  80. package/plugins/lisa-typescript-cursor/.claude-plugin/plugin.json +1 -1
  81. package/plugins/lisa-wiki/.claude-plugin/plugin.json +1 -1
  82. package/plugins/lisa-wiki/.codex-plugin/plugin.json +1 -1
  83. package/plugins/lisa-wiki-agy/plugin.json +1 -1
  84. package/plugins/lisa-wiki-copilot/.claude-plugin/plugin.json +1 -1
  85. package/plugins/lisa-wiki-cursor/.claude-plugin/plugin.json +1 -1
  86. package/plugins/src/base/agents/test-specialist.md +3 -1
  87. package/plugins/src/base/agents/verification-specialist.md +4 -1
  88. package/plugins/src/base/rules/eager/empirical-inquiry.md +1 -0
  89. package/plugins/src/base/rules/eager/falsifiable-checks.md +26 -0
  90. package/plugins/src/base/rules/eager/stale-state-claims.md +26 -0
  91. package/plugins/src/base/rules/eager/verification.md +1 -0
  92. package/plugins/src/base/rules/reference/falsifiable-checks.md +92 -0
  93. package/plugins/src/base/rules/reference/stale-state-claims.md +88 -0
  94. package/plugins/src/base/skills/lisa-codify-verification/SKILL.md +23 -7
package/package.json CHANGED
@@ -115,7 +115,7 @@
115
115
  "brace-expansion": ">=5.0.8"
116
116
  },
117
117
  "name": "@codyswann/lisa",
118
- "version": "2.313.2",
118
+ "version": "2.315.0",
119
119
  "description": "Claude Code governance framework that applies guardrails, guidance, and automated enforcement to projects",
120
120
  "main": "dist/index.js",
121
121
  "exports": {
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "lisa",
3
- "version": "2.313.2",
3
+ "version": "2.315.0",
4
4
  "description": "Universal governance — agents, skills, commands, hooks, and rules for all projects",
5
5
  "author": {
6
6
  "name": "Cody Swann"
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "lisa",
3
- "version": "2.313.2",
3
+ "version": "2.315.0",
4
4
  "description": "Universal governance: agents, skills, commands, hooks, and rules for all projects.",
5
5
  "author": {
6
6
  "name": "Cody Swann"
@@ -135,9 +135,22 @@ Run only the new test, using whatever per-test invocation the project supports:
135
135
 
136
136
  Confirm:
137
137
  1. The test PASSES against the current code (the change being shipped)
138
- 2. The test would have FAILED before the change (sanity check by mentally reverting, or for bug fixes, by running against the pre-fix commit if cheap)
138
+ 2. The test ACTUALLY FAILS without the change observed, not reasoned about
139
139
 
140
- For a bug fix, step 2 is mandatory and easy: check out the failing commit, run the new test, see it fail, return to the fix branch. This proves the test actually guards the regression.
140
+ **Step 2 is mandatory for every codified test, and "mentally reverting" does not satisfy it.** Mental reversion is the exact mechanism by which non-functional guards ship: the author believes the assertion is load-bearing, and it is not. Break the guarded property for real, run the test, and read the failure. See `.claude/rules/falsifiable-checks.md` for the four observed ways a check passes while asserting nothing.
141
+
142
+ Do it one of these ways, in order of preference:
143
+
144
+ - **Run against the pre-fix commit** (bug fixes): check out the failing commit, run the new test, see it fail, return to the fix branch.
145
+ - **Break the property in place**: delete the field, revert the line, flip the condition; run; restore. Prefer this when there is no single pre-fix commit.
146
+ - **Unit-test the checker against synthetic bad input**: required when the input is generated, schema-validated, or cached — a revert can silently fail to change what the test reads (a generator that errors leaves the previous artifact in place, and the test then "passes" on stale input).
147
+
148
+ Two properties the failure itself must have:
149
+
150
+ - It must **name the right location**. A failure that does not localize is weak evidence the test is measuring the intended thing.
151
+ - It must not be satisfiable by the test's own fixture. If the assertion can be met by data the test supplies rather than by the artifact under test, add an explicit assertion against the real artifact (the document, the config, the component's actual output) — or reuse an existing source-bound test.
152
+
153
+ Record the falsification in the codification report (what you broke, how it failed). A codified test whose failure has not been observed is reported as **unvalidated**, not as a regression gate.
141
154
 
142
155
  ### 5. Wire it into the suite
143
156
 
@@ -162,13 +175,13 @@ Append to the verification report (or PR description):
162
175
  ```markdown
163
176
  ### Codified Verifications
164
177
 
165
- | # | Verification | Framework | Test file | Status |
166
- |---|--------------|-----------|-----------|--------|
167
- | 1 | <description> | Playwright | `e2e/checkout.spec.ts::displays order confirmation after checkout` | PASS |
168
- | 2 | <same journey, native surface> | Maestro | `.maestro/flows/checkout-confirmation.yaml` | PASS |
178
+ | # | Verification | Framework | Test file | Status | Falsified by |
179
+ |---|--------------|-----------|-----------|--------|--------------|
180
+ | 1 | <description> | Playwright | `e2e/checkout.spec.ts::displays order confirmation after checkout` | PASS | removed the confirmation render → failed at `checkout.spec.ts:42` |
181
+ | 2 | <same journey, native surface> | Maestro | `.maestro/flows/checkout-confirmation.yaml` | PASS | same break → flow failed on the confirmation assertion |
169
182
  ```
170
183
 
171
- This evidence shows the verification is now guarded.
184
+ This evidence shows the verification is now guarded. **The `Falsified by` column is required** — it names the deliberate break and the observed failure. `UNVALIDATED` is the only permitted alternative, and it means the test is not yet a regression gate.
172
185
 
173
186
  ## Output
174
187
 
@@ -183,6 +196,9 @@ If codification was skipped, an explicit reason recorded in the report (one of t
183
196
  ## Rules
184
197
 
185
198
  - Never claim a verification is codified without running the new test and observing it pass
199
+ - Never claim it is codified without observing it **FAIL** on a real break — mental reversion is not observation, and a test whose failure was never seen is unvalidated, not a gate (`.claude/rules/falsifiable-checks.md`)
200
+ - Never let the assertion be satisfiable by the test's own fixture instead of the artifact under test — bind it to the real document/config/output
201
+ - Never trust a revert-to-verify on generated, schema-validated, or cached input without confirming the input actually changed; a failed generator silently leaves the old artifact and the test "passes" on stale bytes
186
202
  - Never disable, skip, or `.skip()` the new test "temporarily" to make CI green — fix the test or fix the underlying change
187
203
  - Never use `expect(true).toBe(true)` placeholders or smoke-only assertions that don't actually exercise the verified behavior
188
204
  - Never reuse the verification's manual artifact (screenshot, curl output) as a "test" — those are evidence, not regression coverage
@@ -21,6 +21,8 @@ You decide what has to be true for this change to be trusted, and design the tes
21
21
 
22
22
  Do not write tests against the implementation's shape — they pass through a rewrite that breaks behaviour, which is the opposite of the job. Do not treat a coverage number as evidence of anything; it counts lines reached, not defects that would be caught.
23
23
 
24
+ Do not hand on a test you have not watched fail. A test that cannot fail is worse than a missing one: it reports the defect as absent and ends the search. Break the behaviour, watch the assertion fail and name the right place, restore. Watch especially for the assertion that is satisfiable by the test's own fixture rather than by the artifact under test — that one passes no matter what the production code does. `.claude/rules/falsifiable-checks.md` has the four observed shapes.
25
+
24
26
  ## What you hand on
25
27
 
26
- The matrix, the edge cases with the reason each is interesting, the TDD sequence, and the commands that run it all. Where behaviour is user-visible, say which runner proves it end to end.
28
+ The matrix, the edge cases with the reason each is interesting, the TDD sequence, and the commands that run it all. Where behaviour is user-visible, say which runner proves it end to end. For each test, what break makes it fail — an assertion whose failure mode you cannot name is not yet designed.
@@ -12,7 +12,7 @@ skills:
12
12
 
13
13
  You are a verification specialist. Your job is to **prove empirically** that work is done -- not by reading code, but by running the actual system and observing the results.
14
14
 
15
- Read `.claude/rules/verification.md` at the start of every investigation for the full verification framework, types, and lifecycle. Read `.claude/rules/claim-evidence-mapping.md` alongside it: it binds every claim to the **boundary** it asserts and every boundary to the evidence **kinds** that reach it. The verdict you write is what `spec-conformance-specialist` cross-checks — record each claim's `boundary`, its `required_evidence_kinds`, its `evidence_refs`, and its `not_established` list so a boundary mismatch is catchable rather than invisible.
15
+ Read `.claude/rules/verification.md` at the start of every investigation for the full verification framework, types, and lifecycle. Read `.claude/rules/falsifiable-checks.md` alongside it: every check YOU author — probe, script, codified spec, sweep — is subject to it, and a check that has not been shown capable of failing is reported as *unvalidated*, never as passing. Read `.claude/rules/claim-evidence-mapping.md` too: it binds every claim to the **boundary** it asserts and every boundary to the evidence **kinds** that reach it. The verdict you write is what `spec-conformance-specialist` cross-checks — record each claim's `boundary`, its `required_evidence_kinds`, its `evidence_refs`, and its `not_established` list so a boundary mismatch is catchable rather than invisible.
16
16
 
17
17
  ## Core Philosophy
18
18
 
@@ -128,6 +128,9 @@ For every empirical verification that produced PASS evidence, invoke the `codify
128
128
  - Follow the verification lifecycle: confirm quality gates, classify, check tooling, fail fast, plan, execute, codify, spec conformance, loop
129
129
  - Every passing empirical verification must be codified as a regression test via `codify-verification` before declaring done (skip allowed only for PR / Documentation / Deploy / Investigate-Only)
130
130
  - Tests, typecheck, lint, and format are quality gates (prerequisites), NOT verification — never report them as verification evidence
131
+ - Falsify every check you author before reporting a clean result: break the guarded property, confirm the check fails and NAMES the right location, restore. Report what you broke alongside the result — see `.claude/rules/falsifiable-checks.md`
132
+ - A zero-hit sweep is meaningless until the detector has found known instances on a ref where the defect still exists; a passing probe needs a deliberate bite control that MUST report a problem
133
+ - State each clean result's blind spot (presence vs. value, reachability, class completeness) — a negative result describes what the check can perceive, not the code
131
134
  - Discover existing project scripts and tools before creating new ones
132
135
  - Every verification must produce observable output -- a status code, a response body, a UI state, a test result
133
136
  - Verification scripts must be runnable locally without CI/CD dependencies
@@ -16,6 +16,7 @@ Do not reason your way to a confident-sounding answer from documentation, prior
16
16
  - Presenting a guess, recollection, or doc summary as established fact when it was cheap to verify and you did not.
17
17
  - "Should work" / "probably" / "the docs say" as the basis for a load-bearing decision an experiment could have settled.
18
18
  - Skipping the probe because the answer "seems obvious" — those are exactly the ones that quietly drift from reality.
19
+ - Treating a recorded "not yet" / "pending" / "blocked" / "human-gated" note as current state. It is a claim about the day it was written; probe the live state before planning around it, escalating it, or reporting it as a blocker (`stale-state-claims`).
19
20
 
20
21
  This is the inquiry counterpart to the `verification` rule (which proves completed work behaves correctly). Both reject "it looks correct" as evidence.
21
22
 
@@ -0,0 +1,26 @@
1
+ # Falsifiable Checks — A Check That Cannot Fail Is Not Evidence (load-bearing)
2
+
3
+ A passing test, a clean lint run, a zero-hit sweep, a green ratchet: each is evidence **only if that check is known to be capable of failing.** Otherwise it is a checkbox, and a green suite that protects nothing is worse than no suite — it actively suppresses the search for the defect.
4
+
5
+ **Before reporting any clean, zero, or passing result from a check you authored or modified, prove it fails on known-bad input.** Break the thing being guarded, confirm the check fails *and names the right location*, then restore. That falsification is part of the deliverable, not an optional extra.
6
+
7
+ This is the instrument-validity counterpart to the `verification` rule (which proves the *software* behaves) and `empirical-inquiry` (which proves a *fact*). All three reject "it looks correct." This one rejects "the check said so."
8
+
9
+ ## The four ways a check silently measures nothing
10
+
11
+ Each has been observed in real runs; each reported success while asserting nothing:
12
+
13
+ 1. **Self-matching guard** — the check's own explanatory comment, docstring, or ticket prose contains the token it searches for, so it matches itself. It passes with the guarded field deleted.
14
+ 2. **Fixture-validated assertion** — the assertion reads the test's own fixture rather than the artifact under test. Common when the production path consumes a *raw* input the fixture supplies directly.
15
+ 3. **Stale-artifact pass** — the revert-to-verify step silently failed (generator errored, cache served old output, build skipped), so the check re-read unchanged input and "passed".
16
+ 4. **Wrong-baseline sweep** — the detector ran against already-fixed state, so its zero is uninformative. Validate detectors against a ref where the defect still exists.
17
+
18
+ ## Mandatory
19
+
20
+ - **Falsify before reporting.** No clean result is reportable until the check has been shown to fail on a deliberate break. **"Mentally reverting" does not count** — reasoning that the assertion *would* fail is precisely the step that lets a non-functional guard ship, because the author already believes it is load-bearing. Run the break.
21
+ - **Say how it was falsified.** "0 findings" alone is not a result; state what you broke and that the check caught it. A gate whose falsification is untested must be reported as *unvalidated*, not as passing.
22
+ - **Prefer structural over textual checks.** Parse the AST/structure instead of matching source text: text matching cannot distinguish a field from a comment, an alias, or a nested occurrence, and it produces false positives that mask the real ones.
23
+ - **A negative result is scoped to what the check can see.** State the blind spot. A presence check cannot see a wrong value; a per-file check cannot see a cross-file interaction; fixing one instance of a class is not fixing the class — sweep the class.
24
+ - **When revert-to-verify is unreliable** (generated artifacts, schema-validated inputs, caches), unit-test the checker directly against synthetic known-bad input instead.
25
+
26
+ Full prose, worked examples, and the reporting template: [reference/falsifiable-checks.md](../reference/falsifiable-checks.md).
@@ -0,0 +1,26 @@
1
+ # Stale State Claims — "Not Yet" Expires, the Note Does Not (load-bearing)
2
+
3
+ A comment, docstring, config annotation, or work-item note that records a **temporary** state — "not yet", "pending", "waiting on X", "human-gated", "will flip once Y ships" — was accurate the day it was written and is believed long after it stopped being true. Nobody revisits prose when the condition clears.
4
+
5
+ It then misdirects with full authority: work gets skipped as blocked when nothing blocks it, re-planned as undone when it already shipped, or escalated to a person who made that exact decision weeks ago.
6
+
7
+ This is the sibling of `falsifiable-checks`. That rule is about **instruments that cannot fail**; this one is about **assertions of state that have expired**. Both read as authoritative and neither announces its own decay.
8
+
9
+ ## The four ways a recorded state outlives its truth
10
+
11
+ Each has been observed; each was believed long after it went false:
12
+
13
+ 1. **Expired blocker note** — prose asserting a not-yet condition that has since cleared. An endpoint left unset under a comment saying its routing was "not yet promoted" had been serving live for days; an agent planned work to enable what was already enabled.
14
+ 2. **Stale gate marker** — a "pending human approval / not provisioned yet" annotation whose precondition is satisfied. One held for twelve days after the thing it waited on existed, and the decision was escalated to a person surprised to be asked.
15
+ 3. **Prediction buried by closure** — a warning recorded in a comment on an item that then closes. Closure deletes the warning: one correctly predicted a defect, the item closed 25 minutes later, and the defect sat unnoticed for eight days while a dependent item was parked waiting on it.
16
+ 4. **Silent waiting gate** — a queue or approval state with no surfacing mechanism, which holds work for exactly as long as nobody happens to look. One held a security-relevant fix undeployed for two days.
17
+
18
+ ## Mandatory
19
+
20
+ - **A recorded blocker is a claim about the past — check the present before acting on it.** Before planning around it, escalating it, or reporting it as a blocker, probe the live state (`empirical-inquiry`). One command is cheap; inheriting a false premise is not.
21
+ - **Whoever clears the condition deletes the claim.** Removing the block is not done until the note describing it is gone. A cleared gate whose comment survives is the next reader's false premise, and you are the last person who knows it is false.
22
+ - **Prefer expiry-resistant forms.** State what **is** true rather than what is pending. Where a temporal claim is unavoidable, anchor it to something that fails when it goes stale — a check, a linked work item — never to prose nobody re-reads.
23
+ - **A prediction on a closing item becomes tracked work or is retracted.** If the warning is real it is a work item with an explicit blocking link (`tracked-work`); if it is not real, retract it. Prose on a closed item is neither, and the one-way lifecycle means only `claim-archaeology` can recover it.
24
+ - **A gate that can hold work silently is a defect in the gate.** Any waiting state needs a surfacing mechanism — notification, dashboard, scheduled sweep. Fix the gate; do not resolve to look more often.
25
+
26
+ Full prose, worked examples, and the rewrite patterns: [reference/stale-state-claims.md](../reference/stale-state-claims.md).
@@ -10,6 +10,7 @@
10
10
 
11
11
  - **Never claim success without runtime evidence.** "The code looks correct" is not evidence.
12
12
  - **If all you did was run tests, typecheck, and lint — you have NOT verified.**
13
+ - **A check that cannot fail is not evidence either.** Every gate, probe, sweep, and codified spec you author is subject to the `falsifiable-checks` rule: break the guarded property, observe the check fail and name the location, then restore — and report that falsification with the result. "Mentally reverting" does not count, and an unfalsified gate is reported as *unvalidated*, never as passing.
13
14
  - **Browser-controller neutrality.** For UI work, control a live browser and perform the Validation Journey as a human would. An in-app Browser/Chrome tool, interactive Playwright control (MCP, API, or ad hoc script), CDP, computer use, the optional Lisa-owned Kane adapter, or an equivalent controller is acceptable. Kane requires explicit upload approval, a passing `lisa kane probe`, an allow-listed non-production environment, and mutation policy `full`; its provider failure is not a product failure. Do not block merely because one preferred backend is unavailable when another interactive controller can drive the browser. Running an automated Playwright or Maestro test alone is still a quality gate, not the initial empirical evidence; after the live journey passes, codify it in the applicable native runner(s). Kane never replaces those regression gates.
14
15
  - **Before starting implementation, state your verification plan** — how you will USE the resulting software to prove it works. A plan that only lists `test`/`typecheck`/`lint` commands is not a plan. Do not begin until confirmed.
15
16
  - **After verifying empirically, codify it as a regression test** via the `codify-verification` skill — Playwright for UI, integration test for API/DB/auth, benchmark for performance. Codification is mandatory for every verification type except PR/Documentation/Deploy and Investigate-Only spikes. For **frontend work**, codification is dual-runner: a Playwright spec in the project's Playwright test runner AND a Maestro flow in the Maestro test runner whenever the project supports Maestro (`.maestro/` directory, `maestro:test` script, or Maestro CI workflow) — both encoding the same verified journey, neither a substitute for the other.
@@ -0,0 +1,92 @@
1
+ # Falsifiable Checks — Reference
2
+
3
+ Eager head: [eager/falsifiable-checks.md](../eager/falsifiable-checks.md).
4
+
5
+ ## Why this rule exists
6
+
7
+ The `verification` rule prevents the failure "I claimed it works without using it." This rule prevents a subtler one, one layer down: **I used a check, the check said clean, and the check was incapable of saying anything else.**
8
+
9
+ It is the more dangerous failure of the two, because the first leaves you uncertain while the second leaves you *confidently wrong* — and it terminates the investigation. A guard that cannot fail does not merely omit protection; it manufactures evidence that the defect is absent, so nobody looks again.
10
+
11
+ The empirical origin: a single defect-sweep run produced four false-passing checks and zero false fixes. Every fix was correct; every one of the four *instruments* was broken. Reviewers then found three real defects the checks had cleared. The error concentrated entirely in verification, which is why the countermeasure belongs at the check level rather than the code level.
12
+
13
+ ## The four failure modes, in detail
14
+
15
+ ### 1. Self-matching guard
16
+
17
+ A guard searches source text for the token it protects. The fix that satisfies the guard carries a comment explaining *why* that token is required — and the comment contains the token. The guard matches its own justification.
18
+
19
+ Observed: a guard asserting every `bioData` GraphQL selection includes `ggPlayerKey` passed with `ggPlayerKey` **deleted**, because the explanatory comment above the selection said the word. It was caught only by reverting the fix.
20
+
21
+ Countermeasures, in order of preference:
22
+ - Parse structurally (comments never enter an AST).
23
+ - Failing that, strip comments before matching — but note this is itself error-prone: naive `#`-to-end-of-line stripping also destroys `#` inside string literals.
24
+
25
+ ### 2. Fixture-validated assertion
26
+
27
+ The test writes a fixture through the production path and asserts on the result — but the production path reads the *fixture* for the property under test, not the artifact the test claims to be guarding.
28
+
29
+ Observed: a cache-normalization test asserted that two entities keyed apart. It passed with the key field deleted from the query, because the cache computes its key from the **raw incoming object** (which the fixture supplied complete) rather than from the query's selection set. The test validated its own fixture.
30
+
31
+ Countermeasures:
32
+ - Add an explicit assertion binding the test to the real artifact (the document, the config, the container's actual output).
33
+ - Prefer an existing source-bound test when one exists — often a config/props snapshot already reads the real object.
34
+ - Say so in the file: label a fixture as a fixture, so the next reader does not mistake it for the binding.
35
+
36
+ ### 3. Stale-artifact pass
37
+
38
+ The revert-to-verify step appears to run but silently does not change the input the check reads.
39
+
40
+ Observed: proving a guard bites by mutating a `.graphql` document and re-running codegen produced a **false pass** — the mutation made the document invalid against the schema, so codegen errored, left the previously generated file in place, and the test re-read unchanged input. The falsification attempt itself was the thing that failed.
41
+
42
+ Countermeasures:
43
+ - Assert the input actually changed (diff the generated artifact, check the generator's exit code, confirm the digest moved).
44
+ - When the artifact is generated, schema-validated, or cached, do not rely on revert-to-verify at all — unit-test the checker against synthetic known-bad input you construct in-process.
45
+
46
+ ### 4. Wrong-baseline sweep
47
+
48
+ A detector reports zero hits, but ran against state where the defect was already fixed — so zero carries no information.
49
+
50
+ Observed: an uncalled-method detector returned 0 hits across 4,656 files, run on the branch where both instances were already fixed. Re-run against the pre-fix ref it found exactly the 2 known instances, which is what made the zero meaningful.
51
+
52
+ Countermeasure: every detector reporting a zero must first be shown to find known instances on a ref where they exist (`origin/<base>`, the pre-fix commit, or a synthetic fixture).
53
+
54
+ ## What a negative result is scoped to
55
+
56
+ A clean result is a statement about what the check can perceive, not about the code. State the boundary:
57
+
58
+ - **Presence vs. value.** A check that a field exists cannot see that its value is wrong. A field whose value is a stringified function is *present*. Real instance: `new Date().toISOString` (missing call parens) passed a presence-diff, produced a byte-identical constant id on every call, and collided every optimistic cache entry.
59
+ - **Reachability.** A textual match is a candidate, not a defect. Candidates die on: dead code (nothing references the fragment/function), configuration that bypasses the mechanism (`fetchPolicy: "no-cache"` never normalizes), and upstream guards (a button that disables itself makes a state-based re-entry guard redundant). In one sweep, 14 candidates reduced to 3 real and then to 1 user-facing.
60
+ - **Class completeness.** Fixing one instance of a defect class is not fixing the class. Real instance: a cache-id collision was fixed in one file while an instance of the *same class* sat two lines from an active edit in another; a reviewer caught it. After identifying a class, sweep for it — and prefer a repo-wide guard over a local fix so the class cannot regrow.
61
+
62
+ ## How to apply
63
+
64
+ 1. Author the check.
65
+ 2. Deliberately break the guarded property.
66
+ 3. Confirm the check **fails and names the right file/line**. A failure that does not localize is weak evidence the check is measuring the right thing.
67
+ 4. Restore, and confirm green again.
68
+ 5. Report the falsification alongside the result.
69
+
70
+ For generated or validated inputs, replace steps 2–4 with a direct unit test of the checker against synthetic bad input.
71
+
72
+ **A check whose failure has never been observed is reported as `unvalidated`, not as passing.** "Mentally reverting" does not satisfy step 3 — the author of a guard already believes it is load-bearing, so reasoning about the failure reproduces the belief rather than testing it. `unvalidated` is a legitimate state to report and land; silently presenting an unfalsified gate as a passing one is not.
73
+
74
+ ## Reporting template
75
+
76
+ > `<check name>`: <result>. Falsified by <the deliberate break>, which failed as
77
+ > `<observed failure, ideally the located name>`. Blind spots: <what this check
78
+ > cannot see>.
79
+
80
+ Concretely:
81
+
82
+ > Repo-wide keyFields guard: 0 violations across 27 documents. Falsified by moving
83
+ > `Username` one level deeper into `Attributes`; the guard failed naming
84
+ > `activity-feeds/operations.graphql:28`. Blind spot: matches by field name, not
85
+ > resolved schema type.
86
+
87
+ ## Interaction with other rules
88
+
89
+ - **`verification`** — proves the software behaves as a user needs. This rule proves the proof is real. Codified regression tests added under `codify-verification` are subject to this rule: a codified spec that cannot fail is not a regression gate.
90
+ - **`empirical-inquiry`** — settles an uncertain fact with the cheapest probe. A probe is a check, so it inherits the falsification requirement, including deliberate bite controls (cases that MUST report a problem) when the probe's job is to detect problems.
91
+ - **`claim-evidence-mapping`** — an unfalsified gate cannot back a claim.
92
+ - **`stale-state-claims`** — the sibling failure from the opposite direction: an assertion of state that *could* fail but is never re-evaluated, because nothing re-runs prose. Binding a temporal claim to a check is the recommended fix there, which puts that check under this rule.
@@ -0,0 +1,88 @@
1
+ # Stale State Claims — Reference
2
+
3
+ Eager head: [eager/stale-state-claims.md](../eager/stale-state-claims.md).
4
+
5
+ ## Why this rule exists
6
+
7
+ Most bad documentation is *wrong*. This kind was **right** — and that is what makes it dangerous. A note saying "not wired up yet" earned its credibility honestly, so the next reader has no reason to doubt it. There is no defect to find, no test that goes red, no reviewer who objects. The claim simply keeps being read after the world moved.
8
+
9
+ The cost is not confusion, it is confident misdirection in a specific direction: **toward doing nothing**. An expired blocker never causes someone to break production. It causes work to be skipped, re-planned, or escalated — outcomes that look like caution and are indistinguishable from good judgment at the moment they happen.
10
+
11
+ The `falsifiable-checks` rule covers instruments that cannot fail. This one covers assertions that *could* fail but are never re-evaluated, because nothing re-runs prose. Both produce the same end state — a confident answer with no live evidence behind it — from opposite directions.
12
+
13
+ ## The four failure modes, in detail
14
+
15
+ ### 1. Expired blocker note
16
+
17
+ A comment, docstring, config value, or README line records that something is not available yet. The condition clears. The note does not.
18
+
19
+ Observed: a deployment stage's endpoint was left unset with an adjacent comment explaining that its routing had "not yet been promoted." The routing had been live for days. An agent read the comment, believed it, and planned work to enable a capability that was already enabled — then reported the capability as pending.
20
+
21
+ Countermeasures:
22
+ - Treat any not-yet claim as a **hypothesis with an expiry you cannot see**. Resolve it with one live probe (a request, a query, a status read) before it enters a plan, a status report, or an escalation.
23
+ - Cite the probe, not the comment: "verified live at <time>" is evidence; "the comment says" is hearsay about the past.
24
+ - When you are the one who promotes/enables/provisions the thing, the note describing its absence is part of the change. Grep for it.
25
+
26
+ ### 2. Stale gate marker
27
+
28
+ A config entry, flag, or checklist item is annotated as awaiting a human decision or an unfulfilled prerequisite. The prerequisite arrives, or the decision gets made elsewhere. The marker persists and keeps routing work to a gate that is no longer closed.
29
+
30
+ Observed: an entry marked human-gated because no client had been provisioned. The client had existed for twelve days. The gate was honored anyway and the question escalated to a person who had already decided it and was surprised to be asked.
31
+
32
+ This mode is corrosive twice over: it wastes the human's attention, and it teaches the agent that gates are noise — which is exactly the wrong lesson to carry into a gate that is still real.
33
+
34
+ Countermeasures:
35
+ - Before honoring a gate, verify the condition that justifies it still holds. A gate whose stated reason is falsifiable and false is not a gate.
36
+ - Record gates as **conditions**, not as verdicts: "gated until <checkable condition>" can be evaluated; "human-gated" cannot.
37
+ - When escalating, state the evidence that the gate is still live. If you cannot, you are escalating a comment.
38
+
39
+ ### 3. Prediction buried by closure
40
+
41
+ Someone records a correct warning — "this will fail until X is fixed", "this leaves Y broken" — as a comment on a work item, and the item then closes. Closure is a filter: closed items are not read. The prediction is deleted in every practical sense while feeling like it was recorded.
42
+
43
+ Observed: a comment correctly predicted a defect. The item closed 25 minutes later with no follow-up filed. The defect sat unnoticed for **eight days**, blocking a dependent item that was itself parked waiting for it — two items, both stalled, and the explanation was already written down in a place nobody would look.
44
+
45
+ Countermeasures:
46
+ - A prediction is either **real work or a retraction**. If real, it is a tracked leaf with an explicit blocking link to what it blocks (`tracked-work`), created **before** the parent item closes. If not real, say so in the same thread so the next reader is not left holding an unresolved warning.
47
+ - Never let closing be the last action on an item that carries an open prediction. Check the comment thread as part of closing.
48
+ - Blocking links are the mechanism that makes the parked dependent item legible. A prediction with no link is invisible from the side that is actually waiting.
49
+
50
+ ### 4. Silent waiting gate
51
+
52
+ An approval, queue, or review state holds work and emits nothing. Its duration is therefore set by how often somebody happens to look, which is not a property of the work's urgency.
53
+
54
+ Observed: a deploy approval gate held a security-relevant fix undeployed for two days. Nothing was broken, nobody was wrong, and nothing surfaced that the gate was waiting.
55
+
56
+ Countermeasures:
57
+ - Every waiting state needs a surfacing mechanism proportional to what it can hold: notification, dashboard row, scheduled sweep, or an automatic expiry.
58
+ - **Fix the gate, not your habits.** "Remember to check" is not a mechanism; it is the same failure with a person's name on it.
59
+ - When you add a gate, state how long it may silently hold work and what surfaces it. If the answer is "indefinitely" and "nothing", the gate is not finished.
60
+
61
+ ## Expiry-resistant forms
62
+
63
+ Prefer, in order:
64
+
65
+ 1. **State what is true.** "Serves from <path>" outlives "not yet serving from <path>", because it stays wrong-detectably wrong instead of quietly-stale.
66
+ 2. **Bind the claim to a check.** A test, guard, or assertion that fails when the temporal claim goes false converts a silent expiry into a red build. (That check is itself subject to `falsifiable-checks` — a guard that cannot fail re-creates the problem it was added to solve.)
67
+ 3. **Bind the claim to a work item.** "Blocked by <ref>" is resolvable by anyone; "waiting on the migration" is resolvable only by whoever wrote it.
68
+ 4. **Timestamp and scope it.** If prose is genuinely the only option, write "as of <date>" and name the condition that ends it. A dated claim at least advertises its own age.
69
+
70
+ Avoid: bare "not yet", "TODO once", "temporarily", "for now", "pending" with no owner, condition, date, or link. Each is a claim that can only be falsified by someone who already knows the truth — which is the one person who does not need to read it.
71
+
72
+ ## How to apply
73
+
74
+ 1. Reading a note that asserts a pending/blocked/not-yet state: **do not act on it yet.**
75
+ 2. Identify the cheapest live probe that settles the underlying fact.
76
+ 3. Run it. Report the observation, not the note.
77
+ 4. If the note is stale, **delete or correct it in the same change** — leaving it is handing the next reader the trap you just escaped.
78
+ 5. If the note is accurate, upgrade it while you are there: attach the condition, the link, or the check that will make it self-expiring.
79
+
80
+ And when you are the one clearing a condition: sweep for the notes that described it. The change is not complete while the repository still asserts the old state.
81
+
82
+ ## Interaction with other rules
83
+
84
+ - **`falsifiable-checks`** — the sibling failure. That rule prevents an instrument that cannot fail; this one prevents an assertion that never gets re-evaluated. A check bound to a temporal claim (form 2 above) sits under both rules at once.
85
+ - **`empirical-inquiry`** — supplies the discipline this rule depends on: settle the fact with the cheapest probe rather than reasoning from what is written down. A stale note is exactly the "confident-sounding answer from a prior assumption" that rule forbids.
86
+ - **`verification`** — a status derived from a comment is not runtime evidence. Reporting "still blocked" on the strength of a note is the same error as reporting "works" on the strength of the code looking correct.
87
+ - **`tracked-work`** — the destination for any prediction that survives its item's closure: one live leaf, explicit blocking link, carried on the branch and PR.
88
+ - **`claim-archaeology`** — the recovery path once mode 3 has already happened. Lifecycles are one-way, so a warning lost to closure resurfaces only as a fresh item whose ancestry has to be reconstructed. Archaeology is the cleanup; this rule is the prevention.
@@ -135,9 +135,22 @@ Run only the new test, using whatever per-test invocation the project supports:
135
135
 
136
136
  Confirm:
137
137
  1. The test PASSES against the current code (the change being shipped)
138
- 2. The test would have FAILED before the change (sanity check by mentally reverting, or for bug fixes, by running against the pre-fix commit if cheap)
138
+ 2. The test ACTUALLY FAILS without the change observed, not reasoned about
139
139
 
140
- For a bug fix, step 2 is mandatory and easy: check out the failing commit, run the new test, see it fail, return to the fix branch. This proves the test actually guards the regression.
140
+ **Step 2 is mandatory for every codified test, and "mentally reverting" does not satisfy it.** Mental reversion is the exact mechanism by which non-functional guards ship: the author believes the assertion is load-bearing, and it is not. Break the guarded property for real, run the test, and read the failure. See `.claude/rules/falsifiable-checks.md` for the four observed ways a check passes while asserting nothing.
141
+
142
+ Do it one of these ways, in order of preference:
143
+
144
+ - **Run against the pre-fix commit** (bug fixes): check out the failing commit, run the new test, see it fail, return to the fix branch.
145
+ - **Break the property in place**: delete the field, revert the line, flip the condition; run; restore. Prefer this when there is no single pre-fix commit.
146
+ - **Unit-test the checker against synthetic bad input**: required when the input is generated, schema-validated, or cached — a revert can silently fail to change what the test reads (a generator that errors leaves the previous artifact in place, and the test then "passes" on stale input).
147
+
148
+ Two properties the failure itself must have:
149
+
150
+ - It must **name the right location**. A failure that does not localize is weak evidence the test is measuring the intended thing.
151
+ - It must not be satisfiable by the test's own fixture. If the assertion can be met by data the test supplies rather than by the artifact under test, add an explicit assertion against the real artifact (the document, the config, the component's actual output) — or reuse an existing source-bound test.
152
+
153
+ Record the falsification in the codification report (what you broke, how it failed). A codified test whose failure has not been observed is reported as **unvalidated**, not as a regression gate.
141
154
 
142
155
  ### 5. Wire it into the suite
143
156
 
@@ -162,13 +175,13 @@ Append to the verification report (or PR description):
162
175
  ```markdown
163
176
  ### Codified Verifications
164
177
 
165
- | # | Verification | Framework | Test file | Status |
166
- |---|--------------|-----------|-----------|--------|
167
- | 1 | <description> | Playwright | `e2e/checkout.spec.ts::displays order confirmation after checkout` | PASS |
168
- | 2 | <same journey, native surface> | Maestro | `.maestro/flows/checkout-confirmation.yaml` | PASS |
178
+ | # | Verification | Framework | Test file | Status | Falsified by |
179
+ |---|--------------|-----------|-----------|--------|--------------|
180
+ | 1 | <description> | Playwright | `e2e/checkout.spec.ts::displays order confirmation after checkout` | PASS | removed the confirmation render → failed at `checkout.spec.ts:42` |
181
+ | 2 | <same journey, native surface> | Maestro | `.maestro/flows/checkout-confirmation.yaml` | PASS | same break → flow failed on the confirmation assertion |
169
182
  ```
170
183
 
171
- This evidence shows the verification is now guarded.
184
+ This evidence shows the verification is now guarded. **The `Falsified by` column is required** — it names the deliberate break and the observed failure. `UNVALIDATED` is the only permitted alternative, and it means the test is not yet a regression gate.
172
185
 
173
186
  ## Output
174
187
 
@@ -183,6 +196,9 @@ If codification was skipped, an explicit reason recorded in the report (one of t
183
196
  ## Rules
184
197
 
185
198
  - Never claim a verification is codified without running the new test and observing it pass
199
+ - Never claim it is codified without observing it **FAIL** on a real break — mental reversion is not observation, and a test whose failure was never seen is unvalidated, not a gate (`.claude/rules/falsifiable-checks.md`)
200
+ - Never let the assertion be satisfiable by the test's own fixture instead of the artifact under test — bind it to the real document/config/output
201
+ - Never trust a revert-to-verify on generated, schema-validated, or cached input without confirming the input actually changed; a failed generator silently leaves the old artifact and the test "passes" on stale bytes
186
202
  - Never disable, skip, or `.skip()` the new test "temporarily" to make CI green — fix the test or fix the underlying change
187
203
  - Never use `expect(true).toBe(true)` placeholders or smoke-only assertions that don't actually exercise the verified behavior
188
204
  - Never reuse the verification's manual artifact (screenshot, curl output) as a "test" — those are evidence, not regression coverage
@@ -21,6 +21,8 @@ You decide what has to be true for this change to be trusted, and design the tes
21
21
 
22
22
  Do not write tests against the implementation's shape — they pass through a rewrite that breaks behaviour, which is the opposite of the job. Do not treat a coverage number as evidence of anything; it counts lines reached, not defects that would be caught.
23
23
 
24
+ Do not hand on a test you have not watched fail. A test that cannot fail is worse than a missing one: it reports the defect as absent and ends the search. Break the behaviour, watch the assertion fail and name the right place, restore. Watch especially for the assertion that is satisfiable by the test's own fixture rather than by the artifact under test — that one passes no matter what the production code does. `.claude/rules/falsifiable-checks.md` has the four observed shapes.
25
+
24
26
  ## What you hand on
25
27
 
26
- The matrix, the edge cases with the reason each is interesting, the TDD sequence, and the commands that run it all. Where behaviour is user-visible, say which runner proves it end to end.
28
+ The matrix, the edge cases with the reason each is interesting, the TDD sequence, and the commands that run it all. Where behaviour is user-visible, say which runner proves it end to end. For each test, what break makes it fail — an assertion whose failure mode you cannot name is not yet designed.
@@ -12,7 +12,7 @@ skills:
12
12
 
13
13
  You are a verification specialist. Your job is to **prove empirically** that work is done -- not by reading code, but by running the actual system and observing the results.
14
14
 
15
- Read `.claude/rules/verification.md` at the start of every investigation for the full verification framework, types, and lifecycle. Read `.claude/rules/claim-evidence-mapping.md` alongside it: it binds every claim to the **boundary** it asserts and every boundary to the evidence **kinds** that reach it. The verdict you write is what `spec-conformance-specialist` cross-checks — record each claim's `boundary`, its `required_evidence_kinds`, its `evidence_refs`, and its `not_established` list so a boundary mismatch is catchable rather than invisible.
15
+ Read `.claude/rules/verification.md` at the start of every investigation for the full verification framework, types, and lifecycle. Read `.claude/rules/falsifiable-checks.md` alongside it: every check YOU author — probe, script, codified spec, sweep — is subject to it, and a check that has not been shown capable of failing is reported as *unvalidated*, never as passing. Read `.claude/rules/claim-evidence-mapping.md` too: it binds every claim to the **boundary** it asserts and every boundary to the evidence **kinds** that reach it. The verdict you write is what `spec-conformance-specialist` cross-checks — record each claim's `boundary`, its `required_evidence_kinds`, its `evidence_refs`, and its `not_established` list so a boundary mismatch is catchable rather than invisible.
16
16
 
17
17
  ## Core Philosophy
18
18
 
@@ -128,6 +128,9 @@ For every empirical verification that produced PASS evidence, invoke the `codify
128
128
  - Follow the verification lifecycle: confirm quality gates, classify, check tooling, fail fast, plan, execute, codify, spec conformance, loop
129
129
  - Every passing empirical verification must be codified as a regression test via `codify-verification` before declaring done (skip allowed only for PR / Documentation / Deploy / Investigate-Only)
130
130
  - Tests, typecheck, lint, and format are quality gates (prerequisites), NOT verification — never report them as verification evidence
131
+ - Falsify every check you author before reporting a clean result: break the guarded property, confirm the check fails and NAMES the right location, restore. Report what you broke alongside the result — see `.claude/rules/falsifiable-checks.md`
132
+ - A zero-hit sweep is meaningless until the detector has found known instances on a ref where the defect still exists; a passing probe needs a deliberate bite control that MUST report a problem
133
+ - State each clean result's blind spot (presence vs. value, reachability, class completeness) — a negative result describes what the check can perceive, not the code
131
134
  - Discover existing project scripts and tools before creating new ones
132
135
  - Every verification must produce observable output -- a status code, a response body, a UI state, a test result
133
136
  - Verification scripts must be runnable locally without CI/CD dependencies
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "lisa",
3
- "version": "2.313.2",
3
+ "version": "2.315.0",
4
4
  "description": "Universal governance — agents, skills, commands, hooks, and rules for all projects",
5
5
  "author": {
6
6
  "name": "Cody Swann"
@@ -135,9 +135,22 @@ Run only the new test, using whatever per-test invocation the project supports:
135
135
 
136
136
  Confirm:
137
137
  1. The test PASSES against the current code (the change being shipped)
138
- 2. The test would have FAILED before the change (sanity check by mentally reverting, or for bug fixes, by running against the pre-fix commit if cheap)
138
+ 2. The test ACTUALLY FAILS without the change observed, not reasoned about
139
139
 
140
- For a bug fix, step 2 is mandatory and easy: check out the failing commit, run the new test, see it fail, return to the fix branch. This proves the test actually guards the regression.
140
+ **Step 2 is mandatory for every codified test, and "mentally reverting" does not satisfy it.** Mental reversion is the exact mechanism by which non-functional guards ship: the author believes the assertion is load-bearing, and it is not. Break the guarded property for real, run the test, and read the failure. See `.claude/rules/falsifiable-checks.md` for the four observed ways a check passes while asserting nothing.
141
+
142
+ Do it one of these ways, in order of preference:
143
+
144
+ - **Run against the pre-fix commit** (bug fixes): check out the failing commit, run the new test, see it fail, return to the fix branch.
145
+ - **Break the property in place**: delete the field, revert the line, flip the condition; run; restore. Prefer this when there is no single pre-fix commit.
146
+ - **Unit-test the checker against synthetic bad input**: required when the input is generated, schema-validated, or cached — a revert can silently fail to change what the test reads (a generator that errors leaves the previous artifact in place, and the test then "passes" on stale input).
147
+
148
+ Two properties the failure itself must have:
149
+
150
+ - It must **name the right location**. A failure that does not localize is weak evidence the test is measuring the intended thing.
151
+ - It must not be satisfiable by the test's own fixture. If the assertion can be met by data the test supplies rather than by the artifact under test, add an explicit assertion against the real artifact (the document, the config, the component's actual output) — or reuse an existing source-bound test.
152
+
153
+ Record the falsification in the codification report (what you broke, how it failed). A codified test whose failure has not been observed is reported as **unvalidated**, not as a regression gate.
141
154
 
142
155
  ### 5. Wire it into the suite
143
156
 
@@ -162,13 +175,13 @@ Append to the verification report (or PR description):
162
175
  ```markdown
163
176
  ### Codified Verifications
164
177
 
165
- | # | Verification | Framework | Test file | Status |
166
- |---|--------------|-----------|-----------|--------|
167
- | 1 | <description> | Playwright | `e2e/checkout.spec.ts::displays order confirmation after checkout` | PASS |
168
- | 2 | <same journey, native surface> | Maestro | `.maestro/flows/checkout-confirmation.yaml` | PASS |
178
+ | # | Verification | Framework | Test file | Status | Falsified by |
179
+ |---|--------------|-----------|-----------|--------|--------------|
180
+ | 1 | <description> | Playwright | `e2e/checkout.spec.ts::displays order confirmation after checkout` | PASS | removed the confirmation render → failed at `checkout.spec.ts:42` |
181
+ | 2 | <same journey, native surface> | Maestro | `.maestro/flows/checkout-confirmation.yaml` | PASS | same break → flow failed on the confirmation assertion |
169
182
  ```
170
183
 
171
- This evidence shows the verification is now guarded.
184
+ This evidence shows the verification is now guarded. **The `Falsified by` column is required** — it names the deliberate break and the observed failure. `UNVALIDATED` is the only permitted alternative, and it means the test is not yet a regression gate.
172
185
 
173
186
  ## Output
174
187
 
@@ -183,6 +196,9 @@ If codification was skipped, an explicit reason recorded in the report (one of t
183
196
  ## Rules
184
197
 
185
198
  - Never claim a verification is codified without running the new test and observing it pass
199
+ - Never claim it is codified without observing it **FAIL** on a real break — mental reversion is not observation, and a test whose failure was never seen is unvalidated, not a gate (`.claude/rules/falsifiable-checks.md`)
200
+ - Never let the assertion be satisfiable by the test's own fixture instead of the artifact under test — bind it to the real document/config/output
201
+ - Never trust a revert-to-verify on generated, schema-validated, or cached input without confirming the input actually changed; a failed generator silently leaves the old artifact and the test "passes" on stale bytes
186
202
  - Never disable, skip, or `.skip()` the new test "temporarily" to make CI green — fix the test or fix the underlying change
187
203
  - Never use `expect(true).toBe(true)` placeholders or smoke-only assertions that don't actually exercise the verified behavior
188
204
  - Never reuse the verification's manual artifact (screenshot, curl output) as a "test" — those are evidence, not regression coverage
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "lisa-cdk",
3
- "version": "2.313.2",
3
+ "version": "2.315.0",
4
4
  "description": "AWS CDK-specific plugin",
5
5
  "author": {
6
6
  "name": "Cody Swann"
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "lisa-cdk",
3
- "version": "2.313.2",
3
+ "version": "2.315.0",
4
4
  "description": "AWS CDK-specific Lisa plugin.",
5
5
  "author": {
6
6
  "name": "Cody Swann"
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "lisa-cdk",
3
- "version": "2.313.2",
3
+ "version": "2.315.0",
4
4
  "description": "AWS CDK-specific plugin",
5
5
  "author": {
6
6
  "name": "Cody Swann"
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "lisa-cdk",
3
- "version": "2.313.2",
3
+ "version": "2.315.0",
4
4
  "description": "AWS CDK-specific plugin",
5
5
  "author": {
6
6
  "name": "Cody Swann"
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "lisa-cdk",
3
- "version": "2.313.2",
3
+ "version": "2.315.0",
4
4
  "description": "AWS CDK-specific plugin",
5
5
  "author": {
6
6
  "name": "Cody Swann"
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "lisa",
3
- "version": "2.313.2",
3
+ "version": "2.315.0",
4
4
  "description": "Universal governance — agents, skills, commands, hooks, and rules for all projects",
5
5
  "author": {
6
6
  "name": "Cody Swann"