@codyswann/lisa 2.299.4 → 2.300.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/core/upstream-evidence-manifest.js +4 -4
- package/package.json +1 -1
- package/plugins/lisa/.claude-plugin/plugin.json +1 -1
- package/plugins/lisa/.codex-plugin/plugin.json +1 -1
- package/plugins/lisa/.codex-plugin/skills/lisa-implement/SKILL.md +9 -0
- package/plugins/lisa/.codex-plugin/skills/lisa-verification-lifecycle/SKILL.md +9 -0
- package/plugins/lisa/skills/lisa-implement/SKILL.md +9 -0
- package/plugins/lisa/skills/lisa-verification-lifecycle/SKILL.md +9 -0
- package/plugins/lisa-agy/plugin.json +1 -1
- package/plugins/lisa-agy/skills/lisa-implement/SKILL.md +9 -0
- package/plugins/lisa-agy/skills/lisa-verification-lifecycle/SKILL.md +9 -0
- package/plugins/lisa-cdk/.claude-plugin/plugin.json +1 -1
- package/plugins/lisa-cdk/.codex-plugin/plugin.json +1 -1
- package/plugins/lisa-cdk-agy/plugin.json +1 -1
- package/plugins/lisa-cdk-copilot/.claude-plugin/plugin.json +1 -1
- package/plugins/lisa-cdk-cursor/.claude-plugin/plugin.json +1 -1
- package/plugins/lisa-copilot/.claude-plugin/plugin.json +1 -1
- package/plugins/lisa-copilot/skills/lisa-implement/SKILL.md +9 -0
- package/plugins/lisa-copilot/skills/lisa-verification-lifecycle/SKILL.md +9 -0
- package/plugins/lisa-cursor/.claude-plugin/plugin.json +1 -1
- package/plugins/lisa-cursor/skills/lisa-implement/SKILL.md +9 -0
- package/plugins/lisa-cursor/skills/lisa-verification-lifecycle/SKILL.md +9 -0
- package/plugins/lisa-expo/.claude-plugin/plugin.json +1 -1
- package/plugins/lisa-expo/.codex-plugin/plugin.json +1 -1
- package/plugins/lisa-expo/.codex-plugin/skills/playwright-selectors/SKILL.md +16 -0
- package/plugins/lisa-expo/skills/playwright-selectors/SKILL.md +16 -0
- package/plugins/lisa-expo-agy/plugin.json +1 -1
- package/plugins/lisa-expo-agy/skills/playwright-selectors/SKILL.md +16 -0
- package/plugins/lisa-expo-copilot/.claude-plugin/plugin.json +1 -1
- package/plugins/lisa-expo-copilot/skills/playwright-selectors/SKILL.md +16 -0
- package/plugins/lisa-expo-cursor/.claude-plugin/plugin.json +1 -1
- package/plugins/lisa-expo-cursor/skills/playwright-selectors/SKILL.md +16 -0
- package/plugins/lisa-harper-fabric/.claude-plugin/plugin.json +1 -1
- package/plugins/lisa-harper-fabric/.codex-plugin/plugin.json +1 -1
- package/plugins/lisa-harper-fabric-agy/plugin.json +1 -1
- package/plugins/lisa-harper-fabric-copilot/.claude-plugin/plugin.json +1 -1
- package/plugins/lisa-harper-fabric-cursor/.claude-plugin/plugin.json +1 -1
- package/plugins/lisa-nestjs/.claude-plugin/plugin.json +1 -1
- package/plugins/lisa-nestjs/.codex-plugin/plugin.json +1 -1
- package/plugins/lisa-nestjs-agy/plugin.json +1 -1
- package/plugins/lisa-nestjs-copilot/.claude-plugin/plugin.json +1 -1
- package/plugins/lisa-nestjs-cursor/.claude-plugin/plugin.json +1 -1
- package/plugins/lisa-openclaw/.claude-plugin/plugin.json +1 -1
- package/plugins/lisa-openclaw/.codex-plugin/plugin.json +1 -1
- package/plugins/lisa-openclaw-agy/plugin.json +1 -1
- package/plugins/lisa-openclaw-copilot/.claude-plugin/plugin.json +1 -1
- package/plugins/lisa-openclaw-cursor/.claude-plugin/plugin.json +1 -1
- package/plugins/lisa-phaser/.claude-plugin/plugin.json +1 -1
- package/plugins/lisa-phaser/.codex-plugin/plugin.json +1 -1
- package/plugins/lisa-phaser-agy/plugin.json +1 -1
- package/plugins/lisa-phaser-copilot/.claude-plugin/plugin.json +1 -1
- package/plugins/lisa-phaser-cursor/.claude-plugin/plugin.json +1 -1
- package/plugins/lisa-rails/.claude-plugin/plugin.json +1 -1
- package/plugins/lisa-rails/.codex-plugin/plugin.json +1 -1
- package/plugins/lisa-rails-agy/plugin.json +1 -1
- package/plugins/lisa-rails-copilot/.claude-plugin/plugin.json +1 -1
- package/plugins/lisa-rails-cursor/.claude-plugin/plugin.json +1 -1
- package/plugins/lisa-typescript/.claude-plugin/plugin.json +1 -1
- package/plugins/lisa-typescript/.codex-plugin/plugin.json +1 -1
- package/plugins/lisa-typescript-agy/plugin.json +1 -1
- package/plugins/lisa-typescript-copilot/.claude-plugin/plugin.json +1 -1
- package/plugins/lisa-typescript-cursor/.claude-plugin/plugin.json +1 -1
- package/plugins/lisa-wiki/.claude-plugin/plugin.json +1 -1
- package/plugins/lisa-wiki/.codex-plugin/plugin.json +1 -1
- package/plugins/lisa-wiki-agy/plugin.json +1 -1
- package/plugins/lisa-wiki-copilot/.claude-plugin/plugin.json +1 -1
- package/plugins/lisa-wiki-cursor/.claude-plugin/plugin.json +1 -1
- package/plugins/src/base/skills/lisa-implement/SKILL.md +9 -0
- package/plugins/src/base/skills/lisa-verification-lifecycle/SKILL.md +9 -0
- package/plugins/src/expo/skills/playwright-selectors/SKILL.md +16 -0
- package/ui/index.html +3846 -168
|
@@ -428,7 +428,7 @@ export const UPSTREAM_EVIDENCE_MANIFEST = Object.freeze({
|
|
|
428
428
|
"plugins/src/base/skills/lisa-github-write-issue/SKILL.md": "ee54b3d14cc5dd3a68a8c36590102dd02f49334c9493ad52fa8c7f35157073b1",
|
|
429
429
|
"plugins/src/base/skills/lisa-github-write-prd/SKILL.md": "bc919c52e2f42129d41a3f40279d2def99486ec58bed2dab51ca7166b427a947",
|
|
430
430
|
"plugins/src/base/skills/lisa-health/SKILL.md": "dfdb08a863e78bff42671793dcec29cfae0654db18ebf66a4aa77aeb56ddb775",
|
|
431
|
-
"plugins/src/base/skills/lisa-implement/SKILL.md": "
|
|
431
|
+
"plugins/src/base/skills/lisa-implement/SKILL.md": "d01ce8e1af7f4e857810b0c9039481b717ea013bca18e2dc75102a47076d89a0",
|
|
432
432
|
"plugins/src/base/skills/lisa-improve-code-complexity/SKILL.md": "24ab5b193b409db6ee6bee981a1c0a48d08991782d7846116ad01658c8bc1ae8",
|
|
433
433
|
"plugins/src/base/skills/lisa-improve-harness/SKILL.md": "bafda5d2f9b86c48cd16bf0015528fdf5891675221d662fdc533aaad66aad669",
|
|
434
434
|
"plugins/src/base/skills/lisa-improve-max-lines-per-function/SKILL.md": "95c875950c9848fdaf520a18b9bbb264e33d7347be10170add768e01d9cd8e52",
|
|
@@ -555,7 +555,7 @@ export const UPSTREAM_EVIDENCE_MANIFEST = Object.freeze({
|
|
|
555
555
|
"plugins/src/base/skills/lisa-usage-accounting/SKILL.md": "74ef854c172c74f3c1b916c547ed6945abd9852d453146268e276a8ebc35be4e",
|
|
556
556
|
"plugins/src/base/skills/lisa-use-the-product/SKILL.md": "ab3f3ab475b7c97c3409f502b80d688816b1ff4b8e3dfeec7715d93ca41e3fa0",
|
|
557
557
|
"plugins/src/base/skills/lisa-validate-tracker-mapping/SKILL.md": "ab45fdc00294c4ca9698093cb294a2aaf39a73d714fe1aa6ad686c6c34e157ba",
|
|
558
|
-
"plugins/src/base/skills/lisa-verification-lifecycle/SKILL.md": "
|
|
558
|
+
"plugins/src/base/skills/lisa-verification-lifecycle/SKILL.md": "93c5a31e51a7eb7924a2e019660f94d0de60c88247846bb545f291e5d48f5a72",
|
|
559
559
|
"plugins/src/base/skills/lisa-verify-prd/SKILL.md": "a9deea13d2c201e9b265ee3d36f6dd17358fc35e82ae801e72dd1a3145e6fef8",
|
|
560
560
|
"plugins/src/base/skills/lisa-verify-workflow-change/SKILL.md": "d89740a4dc350a9ac0d26b1b955bf74f009775b60ee0275df3f9a50fe8be3048",
|
|
561
561
|
"plugins/src/base/skills/lisa-verify/SKILL.md": "bd97bd5965c05eafff25fcfeabe0fa7e763f96d5dcab8bf959a78679696c33f1",
|
|
@@ -672,7 +672,7 @@ export const UPSTREAM_EVIDENCE_MANIFEST = Object.freeze({
|
|
|
672
672
|
"plugins/src/expo/skills/owasp-zap/SKILL.md": "8b78e25e8885d87c799835a7a2733b739e1c971df95cb24ff449710e873983cf",
|
|
673
673
|
"plugins/src/expo/skills/play-store-access/SKILL.md": "ef5f628dfe99b11b3c7216f541c2c0b297a7038730603020accd5c08b798094c",
|
|
674
674
|
"plugins/src/expo/skills/playwright-ci-debugging/SKILL.md": "5160df3f83df8ff469fa51e384564e5264f493ec90a85e75eff5839bb0bd7157",
|
|
675
|
-
"plugins/src/expo/skills/playwright-selectors/SKILL.md": "
|
|
675
|
+
"plugins/src/expo/skills/playwright-selectors/SKILL.md": "2fbc6b2d0b5765a60dad30dfdbd88abaa6916ff442e32ff638fec9508a8c29d8",
|
|
676
676
|
"plugins/src/expo/skills/reduce-complexity/SKILL.md": "301a43335f398507652b1c37198d76f693b180519c90f379d6e53121a695a38e",
|
|
677
677
|
"plugins/src/expo/skills/reduce-complexity/references/extraction-strategies.md": "64cd766bca8623536be7d7c5e5184a9df089bf4bdc5f2b35f8af01feac0a2f49",
|
|
678
678
|
"plugins/src/expo/skills/reduce-complexity/references/refactoring-patterns.md": "fb90a31054bf7760b3865dac8fd058deeaaa681124123173d703794eb7b08468",
|
|
@@ -1040,7 +1040,7 @@ export const UPSTREAM_EVIDENCE_MANIFEST = Object.freeze({
|
|
|
1040
1040
|
"typescript/merge/.oxlintrc.json": "4debc093acfd263eb81ae15894073ebf098513204f45f40b83fb8fa5353fa1f8",
|
|
1041
1041
|
"typescript/package-lisa/package.lisa.json": "e661329de6f19550953e3522c0dfbc643397ca9a27e975f23437b64bd50b475c",
|
|
1042
1042
|
"ui/README.md": "deeb35e767ea5dd2883268835ea3ad21cbad9fa63ec8d8ff5e200f0e2a7d2751",
|
|
1043
|
-
"ui/index.html": "
|
|
1043
|
+
"ui/index.html": "1ba31656cacc1382128ec8ca5996e1dc2c5470533f88efed981ae7ba9dd9a088",
|
|
1044
1044
|
});
|
|
1045
1045
|
/** Exact paths tracked by the public Lisa repository at generation time. */
|
|
1046
1046
|
export const UPSTREAM_SURFACE_MANIFEST = Object.freeze({
|
package/package.json
CHANGED
|
@@ -115,7 +115,7 @@
|
|
|
115
115
|
"brace-expansion": ">=5.0.6"
|
|
116
116
|
},
|
|
117
117
|
"name": "@codyswann/lisa",
|
|
118
|
-
"version": "2.
|
|
118
|
+
"version": "2.300.0",
|
|
119
119
|
"description": "Claude Code governance framework that applies guardrails, guidance, and automated enforcement to projects",
|
|
120
120
|
"main": "dist/index.js",
|
|
121
121
|
"exports": {
|
|
@@ -139,6 +139,15 @@ The team lead may not waive, defer, demote, or phrase this regression spec as "o
|
|
|
139
139
|
|
|
140
140
|
Completion evidence for the regression spec must prove execution, not mere existence. A green CI run is insufficient unless the PR evidence includes a CI log line, reporter output, or equivalent record naming the new spec and showing that it ran and passed. Guard explicitly against `test.skip`, suite-level environment gates, shard filters, and "0 tests" passes.
|
|
141
141
|
|
|
142
|
+
**Observe the new spec green at least once BEFORE you ship it — ordering matters.** A spec that has never been seen passing anywhere is not regression coverage; it is an untested artifact, and shipping it can break a gate that the current branch does not even run (a spec skipped on the integration branch may be a required gate on the release branch, where it fails for the next author instead of for you). Run it locally first. Only fall back to CI execution proof when local execution is genuinely impossible, and treat that impossibility as a finding to record, not a step to skip.
|
|
143
|
+
|
|
144
|
+
Two traps make a local run lie, so check both before trusting or blaming a local result:
|
|
145
|
+
|
|
146
|
+
- **The run may not be testing your code.** Browser harnesses commonly point at a deployed environment unless a CI-only flag is set (for example a Playwright config that defines `webServer` only when `process.env.CI` is set, so a local run silently exercises the deployed app rather than the working tree). A "local pass" obtained this way is a statement about deployed code, and a local failure may be the deployed code failing, not your change. Confirm which artifact is under test before drawing any conclusion.
|
|
147
|
+
- **A red local run may not be your spec.** Before concluding that a new spec is broken, run a **known-good sibling spec, unmodified, in the same mode as a control**. If the control also fails, the surface is not exercisable in that environment and the result says nothing about your spec. If the control passes and yours fails, the defect is yours. Running the control is cheap and is the only reliable way to separate "my spec is wrong" from "this harness cannot run here" — asserting either without it produces confident, wrong conclusions in both directions.
|
|
148
|
+
|
|
149
|
+
When the control shows the surface is not locally exercisable, that is exit 2 below (a genuine technical blocker): ship the spec only with a linked follow-up that names it as unproven and carries the obligation to confirm it, and say plainly in the PR that the spec has never been observed green. Do not describe an unproven spec as regression coverage.
|
|
150
|
+
|
|
142
151
|
If the required regression spec is still in flight on an auto-merge-enabled PR, pause auto-merge or use an equivalent merge gate until the spec commit is pushed and its execution proof is available. The flow must not allow the PR to merge before this non-demotable deliverable is satisfied or formally blocked through the linked follow-up path above.
|
|
143
152
|
|
|
144
153
|
Using the general-purpose agent in Team Lead session, determine how you will know that the task is fully complete. Write this as an **effective completion condition** — one an independent verifier could confirm from observed output alone, not from your assertion that it works. A strong condition has:
|
|
@@ -76,6 +76,15 @@ Evidence output must explicitly label each verification result as either `verifi
|
|
|
76
76
|
|
|
77
77
|
For a required user-visible regression spec, evidence must prove execution, not only existence. Record a CI log line, reporter output, or equivalent artifact that names the new spec and shows it ran and passed in the PR. A green CI run without named execution proof is not enough; explicitly check for `test.skip`, suite-level environment gates, shard filters, and "0 tests" passes.
|
|
78
78
|
|
|
79
|
+
Prefer observing the new spec green **locally, before it ships**; CI execution proof is the fallback, not the first resort. A spec never seen passing anywhere is an untested artifact rather than coverage, and it can fail later on a branch whose gates differ from this one.
|
|
80
|
+
|
|
81
|
+
Before trusting or blaming any local run of a browser/device spec, establish two things:
|
|
82
|
+
|
|
83
|
+
- **Which artifact is under test.** Harnesses often target a deployed environment unless a CI-only flag is set (e.g. a Playwright config defining `webServer` only when `process.env.CI` is set). A pass obtained that way describes deployed code, not the working tree.
|
|
84
|
+
- **Whether the surface is exercisable at all**, by running a **known-good sibling spec unmodified as a control** in the same mode. Control fails too → the environment cannot exercise that surface and the run carries no information about the new spec. Control passes and the new spec fails → the defect is in the new spec.
|
|
85
|
+
|
|
86
|
+
Record the control result alongside the verdict. A browser-boundary claim marked `not-established` for environmental reasons is only credible with that control; without it the claim is an assumption. When a verdict is later found to rest on one of these traps, state the retraction explicitly in the verdict rather than silently revising it.
|
|
87
|
+
|
|
79
88
|
If auto-merge is enabled while the regression spec is still in flight, disable auto-merge or apply an equivalent merge gate until the spec commit is pushed and its CI execution proof is available. Do not let the PR merge before the required regression deliverable is satisfied or formally blocked through the linked follow-up path.
|
|
80
89
|
|
|
81
90
|
### 7. Codify
|
|
@@ -139,6 +139,15 @@ The team lead may not waive, defer, demote, or phrase this regression spec as "o
|
|
|
139
139
|
|
|
140
140
|
Completion evidence for the regression spec must prove execution, not mere existence. A green CI run is insufficient unless the PR evidence includes a CI log line, reporter output, or equivalent record naming the new spec and showing that it ran and passed. Guard explicitly against `test.skip`, suite-level environment gates, shard filters, and "0 tests" passes.
|
|
141
141
|
|
|
142
|
+
**Observe the new spec green at least once BEFORE you ship it — ordering matters.** A spec that has never been seen passing anywhere is not regression coverage; it is an untested artifact, and shipping it can break a gate that the current branch does not even run (a spec skipped on the integration branch may be a required gate on the release branch, where it fails for the next author instead of for you). Run it locally first. Only fall back to CI execution proof when local execution is genuinely impossible, and treat that impossibility as a finding to record, not a step to skip.
|
|
143
|
+
|
|
144
|
+
Two traps make a local run lie, so check both before trusting or blaming a local result:
|
|
145
|
+
|
|
146
|
+
- **The run may not be testing your code.** Browser harnesses commonly point at a deployed environment unless a CI-only flag is set (for example a Playwright config that defines `webServer` only when `process.env.CI` is set, so a local run silently exercises the deployed app rather than the working tree). A "local pass" obtained this way is a statement about deployed code, and a local failure may be the deployed code failing, not your change. Confirm which artifact is under test before drawing any conclusion.
|
|
147
|
+
- **A red local run may not be your spec.** Before concluding that a new spec is broken, run a **known-good sibling spec, unmodified, in the same mode as a control**. If the control also fails, the surface is not exercisable in that environment and the result says nothing about your spec. If the control passes and yours fails, the defect is yours. Running the control is cheap and is the only reliable way to separate "my spec is wrong" from "this harness cannot run here" — asserting either without it produces confident, wrong conclusions in both directions.
|
|
148
|
+
|
|
149
|
+
When the control shows the surface is not locally exercisable, that is exit 2 below (a genuine technical blocker): ship the spec only with a linked follow-up that names it as unproven and carries the obligation to confirm it, and say plainly in the PR that the spec has never been observed green. Do not describe an unproven spec as regression coverage.
|
|
150
|
+
|
|
142
151
|
If the required regression spec is still in flight on an auto-merge-enabled PR, pause auto-merge or use an equivalent merge gate until the spec commit is pushed and its execution proof is available. The flow must not allow the PR to merge before this non-demotable deliverable is satisfied or formally blocked through the linked follow-up path above.
|
|
143
152
|
|
|
144
153
|
Using the general-purpose agent in Team Lead session, determine how you will know that the task is fully complete. Write this as an **effective completion condition** — one an independent verifier could confirm from observed output alone, not from your assertion that it works. A strong condition has:
|
|
@@ -76,6 +76,15 @@ Evidence output must explicitly label each verification result as either `verifi
|
|
|
76
76
|
|
|
77
77
|
For a required user-visible regression spec, evidence must prove execution, not only existence. Record a CI log line, reporter output, or equivalent artifact that names the new spec and shows it ran and passed in the PR. A green CI run without named execution proof is not enough; explicitly check for `test.skip`, suite-level environment gates, shard filters, and "0 tests" passes.
|
|
78
78
|
|
|
79
|
+
Prefer observing the new spec green **locally, before it ships**; CI execution proof is the fallback, not the first resort. A spec never seen passing anywhere is an untested artifact rather than coverage, and it can fail later on a branch whose gates differ from this one.
|
|
80
|
+
|
|
81
|
+
Before trusting or blaming any local run of a browser/device spec, establish two things:
|
|
82
|
+
|
|
83
|
+
- **Which artifact is under test.** Harnesses often target a deployed environment unless a CI-only flag is set (e.g. a Playwright config defining `webServer` only when `process.env.CI` is set). A pass obtained that way describes deployed code, not the working tree.
|
|
84
|
+
- **Whether the surface is exercisable at all**, by running a **known-good sibling spec unmodified as a control** in the same mode. Control fails too → the environment cannot exercise that surface and the run carries no information about the new spec. Control passes and the new spec fails → the defect is in the new spec.
|
|
85
|
+
|
|
86
|
+
Record the control result alongside the verdict. A browser-boundary claim marked `not-established` for environmental reasons is only credible with that control; without it the claim is an assumption. When a verdict is later found to rest on one of these traps, state the retraction explicitly in the verdict rather than silently revising it.
|
|
87
|
+
|
|
79
88
|
If auto-merge is enabled while the regression spec is still in flight, disable auto-merge or apply an equivalent merge gate until the spec commit is pushed and its CI execution proof is available. Do not let the PR merge before the required regression deliverable is satisfied or formally blocked through the linked follow-up path.
|
|
80
89
|
|
|
81
90
|
### 7. Codify
|
|
@@ -139,6 +139,15 @@ The team lead may not waive, defer, demote, or phrase this regression spec as "o
|
|
|
139
139
|
|
|
140
140
|
Completion evidence for the regression spec must prove execution, not mere existence. A green CI run is insufficient unless the PR evidence includes a CI log line, reporter output, or equivalent record naming the new spec and showing that it ran and passed. Guard explicitly against `test.skip`, suite-level environment gates, shard filters, and "0 tests" passes.
|
|
141
141
|
|
|
142
|
+
**Observe the new spec green at least once BEFORE you ship it — ordering matters.** A spec that has never been seen passing anywhere is not regression coverage; it is an untested artifact, and shipping it can break a gate that the current branch does not even run (a spec skipped on the integration branch may be a required gate on the release branch, where it fails for the next author instead of for you). Run it locally first. Only fall back to CI execution proof when local execution is genuinely impossible, and treat that impossibility as a finding to record, not a step to skip.
|
|
143
|
+
|
|
144
|
+
Two traps make a local run lie, so check both before trusting or blaming a local result:
|
|
145
|
+
|
|
146
|
+
- **The run may not be testing your code.** Browser harnesses commonly point at a deployed environment unless a CI-only flag is set (for example a Playwright config that defines `webServer` only when `process.env.CI` is set, so a local run silently exercises the deployed app rather than the working tree). A "local pass" obtained this way is a statement about deployed code, and a local failure may be the deployed code failing, not your change. Confirm which artifact is under test before drawing any conclusion.
|
|
147
|
+
- **A red local run may not be your spec.** Before concluding that a new spec is broken, run a **known-good sibling spec, unmodified, in the same mode as a control**. If the control also fails, the surface is not exercisable in that environment and the result says nothing about your spec. If the control passes and yours fails, the defect is yours. Running the control is cheap and is the only reliable way to separate "my spec is wrong" from "this harness cannot run here" — asserting either without it produces confident, wrong conclusions in both directions.
|
|
148
|
+
|
|
149
|
+
When the control shows the surface is not locally exercisable, that is exit 2 below (a genuine technical blocker): ship the spec only with a linked follow-up that names it as unproven and carries the obligation to confirm it, and say plainly in the PR that the spec has never been observed green. Do not describe an unproven spec as regression coverage.
|
|
150
|
+
|
|
142
151
|
If the required regression spec is still in flight on an auto-merge-enabled PR, pause auto-merge or use an equivalent merge gate until the spec commit is pushed and its execution proof is available. The flow must not allow the PR to merge before this non-demotable deliverable is satisfied or formally blocked through the linked follow-up path above.
|
|
143
152
|
|
|
144
153
|
Using the general-purpose agent in Team Lead session, determine how you will know that the task is fully complete. Write this as an **effective completion condition** — one an independent verifier could confirm from observed output alone, not from your assertion that it works. A strong condition has:
|
|
@@ -76,6 +76,15 @@ Evidence output must explicitly label each verification result as either `verifi
|
|
|
76
76
|
|
|
77
77
|
For a required user-visible regression spec, evidence must prove execution, not only existence. Record a CI log line, reporter output, or equivalent artifact that names the new spec and shows it ran and passed in the PR. A green CI run without named execution proof is not enough; explicitly check for `test.skip`, suite-level environment gates, shard filters, and "0 tests" passes.
|
|
78
78
|
|
|
79
|
+
Prefer observing the new spec green **locally, before it ships**; CI execution proof is the fallback, not the first resort. A spec never seen passing anywhere is an untested artifact rather than coverage, and it can fail later on a branch whose gates differ from this one.
|
|
80
|
+
|
|
81
|
+
Before trusting or blaming any local run of a browser/device spec, establish two things:
|
|
82
|
+
|
|
83
|
+
- **Which artifact is under test.** Harnesses often target a deployed environment unless a CI-only flag is set (e.g. a Playwright config defining `webServer` only when `process.env.CI` is set). A pass obtained that way describes deployed code, not the working tree.
|
|
84
|
+
- **Whether the surface is exercisable at all**, by running a **known-good sibling spec unmodified as a control** in the same mode. Control fails too → the environment cannot exercise that surface and the run carries no information about the new spec. Control passes and the new spec fails → the defect is in the new spec.
|
|
85
|
+
|
|
86
|
+
Record the control result alongside the verdict. A browser-boundary claim marked `not-established` for environmental reasons is only credible with that control; without it the claim is an assumption. When a verdict is later found to rest on one of these traps, state the retraction explicitly in the verdict rather than silently revising it.
|
|
87
|
+
|
|
79
88
|
If auto-merge is enabled while the regression spec is still in flight, disable auto-merge or apply an equivalent merge gate until the spec commit is pushed and its CI execution proof is available. Do not let the PR merge before the required regression deliverable is satisfied or formally blocked through the linked follow-up path.
|
|
80
89
|
|
|
81
90
|
### 7. Codify
|
|
@@ -139,6 +139,15 @@ The team lead may not waive, defer, demote, or phrase this regression spec as "o
|
|
|
139
139
|
|
|
140
140
|
Completion evidence for the regression spec must prove execution, not mere existence. A green CI run is insufficient unless the PR evidence includes a CI log line, reporter output, or equivalent record naming the new spec and showing that it ran and passed. Guard explicitly against `test.skip`, suite-level environment gates, shard filters, and "0 tests" passes.
|
|
141
141
|
|
|
142
|
+
**Observe the new spec green at least once BEFORE you ship it — ordering matters.** A spec that has never been seen passing anywhere is not regression coverage; it is an untested artifact, and shipping it can break a gate that the current branch does not even run (a spec skipped on the integration branch may be a required gate on the release branch, where it fails for the next author instead of for you). Run it locally first. Only fall back to CI execution proof when local execution is genuinely impossible, and treat that impossibility as a finding to record, not a step to skip.
|
|
143
|
+
|
|
144
|
+
Two traps make a local run lie, so check both before trusting or blaming a local result:
|
|
145
|
+
|
|
146
|
+
- **The run may not be testing your code.** Browser harnesses commonly point at a deployed environment unless a CI-only flag is set (for example a Playwright config that defines `webServer` only when `process.env.CI` is set, so a local run silently exercises the deployed app rather than the working tree). A "local pass" obtained this way is a statement about deployed code, and a local failure may be the deployed code failing, not your change. Confirm which artifact is under test before drawing any conclusion.
|
|
147
|
+
- **A red local run may not be your spec.** Before concluding that a new spec is broken, run a **known-good sibling spec, unmodified, in the same mode as a control**. If the control also fails, the surface is not exercisable in that environment and the result says nothing about your spec. If the control passes and yours fails, the defect is yours. Running the control is cheap and is the only reliable way to separate "my spec is wrong" from "this harness cannot run here" — asserting either without it produces confident, wrong conclusions in both directions.
|
|
148
|
+
|
|
149
|
+
When the control shows the surface is not locally exercisable, that is exit 2 below (a genuine technical blocker): ship the spec only with a linked follow-up that names it as unproven and carries the obligation to confirm it, and say plainly in the PR that the spec has never been observed green. Do not describe an unproven spec as regression coverage.
|
|
150
|
+
|
|
142
151
|
If the required regression spec is still in flight on an auto-merge-enabled PR, pause auto-merge or use an equivalent merge gate until the spec commit is pushed and its execution proof is available. The flow must not allow the PR to merge before this non-demotable deliverable is satisfied or formally blocked through the linked follow-up path above.
|
|
143
152
|
|
|
144
153
|
Using the general-purpose agent in Team Lead session, determine how you will know that the task is fully complete. Write this as an **effective completion condition** — one an independent verifier could confirm from observed output alone, not from your assertion that it works. A strong condition has:
|
|
@@ -76,6 +76,15 @@ Evidence output must explicitly label each verification result as either `verifi
|
|
|
76
76
|
|
|
77
77
|
For a required user-visible regression spec, evidence must prove execution, not only existence. Record a CI log line, reporter output, or equivalent artifact that names the new spec and shows it ran and passed in the PR. A green CI run without named execution proof is not enough; explicitly check for `test.skip`, suite-level environment gates, shard filters, and "0 tests" passes.
|
|
78
78
|
|
|
79
|
+
Prefer observing the new spec green **locally, before it ships**; CI execution proof is the fallback, not the first resort. A spec never seen passing anywhere is an untested artifact rather than coverage, and it can fail later on a branch whose gates differ from this one.
|
|
80
|
+
|
|
81
|
+
Before trusting or blaming any local run of a browser/device spec, establish two things:
|
|
82
|
+
|
|
83
|
+
- **Which artifact is under test.** Harnesses often target a deployed environment unless a CI-only flag is set (e.g. a Playwright config defining `webServer` only when `process.env.CI` is set). A pass obtained that way describes deployed code, not the working tree.
|
|
84
|
+
- **Whether the surface is exercisable at all**, by running a **known-good sibling spec unmodified as a control** in the same mode. Control fails too → the environment cannot exercise that surface and the run carries no information about the new spec. Control passes and the new spec fails → the defect is in the new spec.
|
|
85
|
+
|
|
86
|
+
Record the control result alongside the verdict. A browser-boundary claim marked `not-established` for environmental reasons is only credible with that control; without it the claim is an assumption. When a verdict is later found to rest on one of these traps, state the retraction explicitly in the verdict rather than silently revising it.
|
|
87
|
+
|
|
79
88
|
If auto-merge is enabled while the regression spec is still in flight, disable auto-merge or apply an equivalent merge gate until the spec commit is pushed and its CI execution proof is available. Do not let the PR merge before the required regression deliverable is satisfied or formally blocked through the linked follow-up path.
|
|
80
89
|
|
|
81
90
|
### 7. Codify
|
|
@@ -139,6 +139,15 @@ The team lead may not waive, defer, demote, or phrase this regression spec as "o
|
|
|
139
139
|
|
|
140
140
|
Completion evidence for the regression spec must prove execution, not mere existence. A green CI run is insufficient unless the PR evidence includes a CI log line, reporter output, or equivalent record naming the new spec and showing that it ran and passed. Guard explicitly against `test.skip`, suite-level environment gates, shard filters, and "0 tests" passes.
|
|
141
141
|
|
|
142
|
+
**Observe the new spec green at least once BEFORE you ship it — ordering matters.** A spec that has never been seen passing anywhere is not regression coverage; it is an untested artifact, and shipping it can break a gate that the current branch does not even run (a spec skipped on the integration branch may be a required gate on the release branch, where it fails for the next author instead of for you). Run it locally first. Only fall back to CI execution proof when local execution is genuinely impossible, and treat that impossibility as a finding to record, not a step to skip.
|
|
143
|
+
|
|
144
|
+
Two traps make a local run lie, so check both before trusting or blaming a local result:
|
|
145
|
+
|
|
146
|
+
- **The run may not be testing your code.** Browser harnesses commonly point at a deployed environment unless a CI-only flag is set (for example a Playwright config that defines `webServer` only when `process.env.CI` is set, so a local run silently exercises the deployed app rather than the working tree). A "local pass" obtained this way is a statement about deployed code, and a local failure may be the deployed code failing, not your change. Confirm which artifact is under test before drawing any conclusion.
|
|
147
|
+
- **A red local run may not be your spec.** Before concluding that a new spec is broken, run a **known-good sibling spec, unmodified, in the same mode as a control**. If the control also fails, the surface is not exercisable in that environment and the result says nothing about your spec. If the control passes and yours fails, the defect is yours. Running the control is cheap and is the only reliable way to separate "my spec is wrong" from "this harness cannot run here" — asserting either without it produces confident, wrong conclusions in both directions.
|
|
148
|
+
|
|
149
|
+
When the control shows the surface is not locally exercisable, that is exit 2 below (a genuine technical blocker): ship the spec only with a linked follow-up that names it as unproven and carries the obligation to confirm it, and say plainly in the PR that the spec has never been observed green. Do not describe an unproven spec as regression coverage.
|
|
150
|
+
|
|
142
151
|
If the required regression spec is still in flight on an auto-merge-enabled PR, pause auto-merge or use an equivalent merge gate until the spec commit is pushed and its execution proof is available. The flow must not allow the PR to merge before this non-demotable deliverable is satisfied or formally blocked through the linked follow-up path above.
|
|
143
152
|
|
|
144
153
|
Using the general-purpose agent in Team Lead session, determine how you will know that the task is fully complete. Write this as an **effective completion condition** — one an independent verifier could confirm from observed output alone, not from your assertion that it works. A strong condition has:
|
|
@@ -76,6 +76,15 @@ Evidence output must explicitly label each verification result as either `verifi
|
|
|
76
76
|
|
|
77
77
|
For a required user-visible regression spec, evidence must prove execution, not only existence. Record a CI log line, reporter output, or equivalent artifact that names the new spec and shows it ran and passed in the PR. A green CI run without named execution proof is not enough; explicitly check for `test.skip`, suite-level environment gates, shard filters, and "0 tests" passes.
|
|
78
78
|
|
|
79
|
+
Prefer observing the new spec green **locally, before it ships**; CI execution proof is the fallback, not the first resort. A spec never seen passing anywhere is an untested artifact rather than coverage, and it can fail later on a branch whose gates differ from this one.
|
|
80
|
+
|
|
81
|
+
Before trusting or blaming any local run of a browser/device spec, establish two things:
|
|
82
|
+
|
|
83
|
+
- **Which artifact is under test.** Harnesses often target a deployed environment unless a CI-only flag is set (e.g. a Playwright config defining `webServer` only when `process.env.CI` is set). A pass obtained that way describes deployed code, not the working tree.
|
|
84
|
+
- **Whether the surface is exercisable at all**, by running a **known-good sibling spec unmodified as a control** in the same mode. Control fails too → the environment cannot exercise that surface and the run carries no information about the new spec. Control passes and the new spec fails → the defect is in the new spec.
|
|
85
|
+
|
|
86
|
+
Record the control result alongside the verdict. A browser-boundary claim marked `not-established` for environmental reasons is only credible with that control; without it the claim is an assumption. When a verdict is later found to rest on one of these traps, state the retraction explicitly in the verdict rather than silently revising it.
|
|
87
|
+
|
|
79
88
|
If auto-merge is enabled while the regression spec is still in flight, disable auto-merge or apply an equivalent merge gate until the spec commit is pushed and its CI execution proof is available. Do not let the PR merge before the required regression deliverable is satisfied or formally blocked through the linked follow-up path.
|
|
80
89
|
|
|
81
90
|
### 7. Codify
|
|
@@ -271,6 +271,22 @@ export default defineConfig({
|
|
|
271
271
|
});
|
|
272
272
|
```
|
|
273
273
|
|
|
274
|
+
> **Run local verification with `CI=1`.** In a config shaped like the one above,
|
|
275
|
+
> `webServer` exists only when `CI` is set and `baseURL` otherwise points at the
|
|
276
|
+
> **deployed** environment. So a plain `npx playwright test <spec>` does not test your
|
|
277
|
+
> working tree at all — it tests whatever is deployed. A pass proves nothing about your
|
|
278
|
+
> change, and a failure may be the deployed app failing rather than your code. Use
|
|
279
|
+
> `CI=1 npx playwright test <spec>` whenever you are verifying uncommitted work.
|
|
280
|
+
>
|
|
281
|
+
> **Use a control spec before concluding anything from a red run.** If a spec fails
|
|
282
|
+
> locally, re-run a **known-good sibling spec, unmodified, in the same mode**. If the
|
|
283
|
+
> control fails too, that surface is not exercisable in your environment and the run
|
|
284
|
+
> says nothing about your spec — don't "fix" a spec that was never broken. If the
|
|
285
|
+
> control passes, the defect is genuinely yours.
|
|
286
|
+
>
|
|
287
|
+
> Also read reporter totals carefully: `setup` / auth projects count as passing tests.
|
|
288
|
+
> "2 passed" can mean two auth setups and zero assertions.
|
|
289
|
+
|
|
274
290
|
---
|
|
275
291
|
|
|
276
292
|
## Writing Robust Tests
|
|
@@ -271,6 +271,22 @@ export default defineConfig({
|
|
|
271
271
|
});
|
|
272
272
|
```
|
|
273
273
|
|
|
274
|
+
> **Run local verification with `CI=1`.** In a config shaped like the one above,
|
|
275
|
+
> `webServer` exists only when `CI` is set and `baseURL` otherwise points at the
|
|
276
|
+
> **deployed** environment. So a plain `npx playwright test <spec>` does not test your
|
|
277
|
+
> working tree at all — it tests whatever is deployed. A pass proves nothing about your
|
|
278
|
+
> change, and a failure may be the deployed app failing rather than your code. Use
|
|
279
|
+
> `CI=1 npx playwright test <spec>` whenever you are verifying uncommitted work.
|
|
280
|
+
>
|
|
281
|
+
> **Use a control spec before concluding anything from a red run.** If a spec fails
|
|
282
|
+
> locally, re-run a **known-good sibling spec, unmodified, in the same mode**. If the
|
|
283
|
+
> control fails too, that surface is not exercisable in your environment and the run
|
|
284
|
+
> says nothing about your spec — don't "fix" a spec that was never broken. If the
|
|
285
|
+
> control passes, the defect is genuinely yours.
|
|
286
|
+
>
|
|
287
|
+
> Also read reporter totals carefully: `setup` / auth projects count as passing tests.
|
|
288
|
+
> "2 passed" can mean two auth setups and zero assertions.
|
|
289
|
+
|
|
274
290
|
---
|
|
275
291
|
|
|
276
292
|
## Writing Robust Tests
|
|
@@ -271,6 +271,22 @@ export default defineConfig({
|
|
|
271
271
|
});
|
|
272
272
|
```
|
|
273
273
|
|
|
274
|
+
> **Run local verification with `CI=1`.** In a config shaped like the one above,
|
|
275
|
+
> `webServer` exists only when `CI` is set and `baseURL` otherwise points at the
|
|
276
|
+
> **deployed** environment. So a plain `npx playwright test <spec>` does not test your
|
|
277
|
+
> working tree at all — it tests whatever is deployed. A pass proves nothing about your
|
|
278
|
+
> change, and a failure may be the deployed app failing rather than your code. Use
|
|
279
|
+
> `CI=1 npx playwright test <spec>` whenever you are verifying uncommitted work.
|
|
280
|
+
>
|
|
281
|
+
> **Use a control spec before concluding anything from a red run.** If a spec fails
|
|
282
|
+
> locally, re-run a **known-good sibling spec, unmodified, in the same mode**. If the
|
|
283
|
+
> control fails too, that surface is not exercisable in your environment and the run
|
|
284
|
+
> says nothing about your spec — don't "fix" a spec that was never broken. If the
|
|
285
|
+
> control passes, the defect is genuinely yours.
|
|
286
|
+
>
|
|
287
|
+
> Also read reporter totals carefully: `setup` / auth projects count as passing tests.
|
|
288
|
+
> "2 passed" can mean two auth setups and zero assertions.
|
|
289
|
+
|
|
274
290
|
---
|
|
275
291
|
|
|
276
292
|
## Writing Robust Tests
|
|
@@ -271,6 +271,22 @@ export default defineConfig({
|
|
|
271
271
|
});
|
|
272
272
|
```
|
|
273
273
|
|
|
274
|
+
> **Run local verification with `CI=1`.** In a config shaped like the one above,
|
|
275
|
+
> `webServer` exists only when `CI` is set and `baseURL` otherwise points at the
|
|
276
|
+
> **deployed** environment. So a plain `npx playwright test <spec>` does not test your
|
|
277
|
+
> working tree at all — it tests whatever is deployed. A pass proves nothing about your
|
|
278
|
+
> change, and a failure may be the deployed app failing rather than your code. Use
|
|
279
|
+
> `CI=1 npx playwright test <spec>` whenever you are verifying uncommitted work.
|
|
280
|
+
>
|
|
281
|
+
> **Use a control spec before concluding anything from a red run.** If a spec fails
|
|
282
|
+
> locally, re-run a **known-good sibling spec, unmodified, in the same mode**. If the
|
|
283
|
+
> control fails too, that surface is not exercisable in your environment and the run
|
|
284
|
+
> says nothing about your spec — don't "fix" a spec that was never broken. If the
|
|
285
|
+
> control passes, the defect is genuinely yours.
|
|
286
|
+
>
|
|
287
|
+
> Also read reporter totals carefully: `setup` / auth projects count as passing tests.
|
|
288
|
+
> "2 passed" can mean two auth setups and zero assertions.
|
|
289
|
+
|
|
274
290
|
---
|
|
275
291
|
|
|
276
292
|
## Writing Robust Tests
|
|
@@ -271,6 +271,22 @@ export default defineConfig({
|
|
|
271
271
|
});
|
|
272
272
|
```
|
|
273
273
|
|
|
274
|
+
> **Run local verification with `CI=1`.** In a config shaped like the one above,
|
|
275
|
+
> `webServer` exists only when `CI` is set and `baseURL` otherwise points at the
|
|
276
|
+
> **deployed** environment. So a plain `npx playwright test <spec>` does not test your
|
|
277
|
+
> working tree at all — it tests whatever is deployed. A pass proves nothing about your
|
|
278
|
+
> change, and a failure may be the deployed app failing rather than your code. Use
|
|
279
|
+
> `CI=1 npx playwright test <spec>` whenever you are verifying uncommitted work.
|
|
280
|
+
>
|
|
281
|
+
> **Use a control spec before concluding anything from a red run.** If a spec fails
|
|
282
|
+
> locally, re-run a **known-good sibling spec, unmodified, in the same mode**. If the
|
|
283
|
+
> control fails too, that surface is not exercisable in your environment and the run
|
|
284
|
+
> says nothing about your spec — don't "fix" a spec that was never broken. If the
|
|
285
|
+
> control passes, the defect is genuinely yours.
|
|
286
|
+
>
|
|
287
|
+
> Also read reporter totals carefully: `setup` / auth projects count as passing tests.
|
|
288
|
+
> "2 passed" can mean two auth setups and zero assertions.
|
|
289
|
+
|
|
274
290
|
---
|
|
275
291
|
|
|
276
292
|
## Writing Robust Tests
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "lisa-openclaw",
|
|
3
|
-
"version": "2.
|
|
3
|
+
"version": "2.300.0",
|
|
4
4
|
"description": "Connect staff roles to Telegram or Slack via OpenClaw — facilitator/specialist hub-and-spoke routing and repo-coding topics, for Claude Code and Codex",
|
|
5
5
|
"author": {
|
|
6
6
|
"name": "Cody Swann"
|