@michelj/context-guard 0.4.2 → 0.4.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +74 -6
- package/README.zh-CN.md +74 -6
- package/{skills/context-guard/SKILL.md → SKILL.md} +216 -77
- package/bin/context-guard-skill.js +88 -3
- package/hooks.json +25 -1
- package/package.json +7 -3
- package/{skills/context-guard/references → references}/context-template.md +52 -1
- package/references/feature-chain-methodology.md +228 -0
- package/{skills/context-guard/scripts → scripts}/context_guard.py +3119 -121
- package/{skills/context-guard/scripts → scripts}/context_guard_hook.py +344 -12
- package/{skills/context-guard/tests → tests}/BC-20260701-090.sh +8 -7
- package/{skills/context-guard/tests → tests}/BC-20260702-096.sh +6 -3
- package/tests/BC-20260706-098.sh +66 -0
- package/tests/BC-20260707-099.sh +47 -0
- package/tests/BC-20260707-100.sh +46 -0
- package/tests/BC-20260707-101.sh +47 -0
- package/tests/BC-20260707-102.sh +68 -0
- package/tests/BC-20260707-103.sh +59 -0
- package/tests/BC-20260707-104.sh +103 -0
- package/tests/BC-20260707-105.sh +109 -0
- package/tests/BC-20260707-106.sh +80 -0
- package/tests/BC-20260707-107.sh +74 -0
- package/tests/BC-20260707-108.sh +48 -0
- package/tests/BC-20260707-109.sh +56 -0
- package/tests/BC-20260707-110.sh +71 -0
- package/tests/BC-20260707-111.sh +70 -0
- package/tests/BC-20260707-112.sh +45 -0
- package/tests/BC-20260707-113.sh +73 -0
- package/tests/BC-20260707-115.sh +77 -0
- package/tests/BC-20260707-116.sh +77 -0
- package/tests/BC-20260707-118.sh +115 -0
- package/tests/BC-20260707-119.sh +47 -0
- package/tests/BC-20260707-120.sh +60 -0
- package/tests/BC-20260707-121.sh +66 -0
- package/tests/BC-20260707-122.sh +48 -0
- package/tests/BC-20260707-123.sh +43 -0
- package/tests/BC-20260707-124.sh +56 -0
- package/tests/BC-20260707-125.sh +64 -0
- package/tests/BC-20260707-126.sh +80 -0
- package/tests/BC-20260707-127.sh +88 -0
- package/tests/BC-20260707-129.sh +59 -0
- package/tests/BC-20260707-130.sh +69 -0
- package/tests/BC-20260707-131.sh +140 -0
- package/tests/BC-20260707-132.sh +150 -0
- package/tests/BC-20260707-133.sh +70 -0
- package/tests/BC-20260708-136.sh +210 -0
- package/tests/BC-20260708-137.sh +106 -0
- package/tests/BC-20260708-138.sh +168 -0
- package/tests/BC-20260708-139.sh +79 -0
- package/tests/BC-20260709-002.sh +63 -0
- package/tests/BC-20260709-003.sh +239 -0
- package/tests/BC-20260709-006.sh +76 -0
- package/tests/BC-20260709-008.sh +168 -0
- package/tests/BC-20260710-001.sh +61 -0
- package/tests/BC-20260710-002.sh +111 -0
- package/tests/npm-install-smoke.sh +53 -0
- package/skills/context-guard/README.md +0 -234
- package/skills/context-guard/README.zh-CN.md +0 -234
- /package/{skills/context-guard/agents → agents}/openai.yaml +0 -0
- /package/{skills/context-guard/references → references}/register-template.md +0 -0
- /package/{skills/context-guard/references → references}/task-case-template.md +0 -0
- /package/{skills/context-guard/tests → tests}/BC-20260618-063.sh +0 -0
- /package/{skills/context-guard/tests → tests}/BC-20260618-065.sh +0 -0
- /package/{skills/context-guard/tests → tests}/BC-20260626-080.sh +0 -0
- /package/{skills/context-guard/tests → tests}/BC-20260626-081.sh +0 -0
- /package/{skills/context-guard/tests → tests}/BC-20260626-082.sh +0 -0
- /package/{skills/context-guard/tests → tests}/BC-20260626-083.sh +0 -0
- /package/{skills/context-guard/tests → tests}/BC-20260627-084.sh +0 -0
- /package/{skills/context-guard/tests → tests}/BC-20260630-086.sh +0 -0
- /package/{skills/context-guard/tests → tests}/BC-20260630-087.sh +0 -0
- /package/{skills/context-guard/tests → tests}/BC-20260630-088.sh +0 -0
- /package/{skills/context-guard/tests → tests}/BC-20260630-089.sh +0 -0
|
@@ -7,15 +7,24 @@ const path = require("path");
|
|
|
7
7
|
const { spawnSync } = require("child_process");
|
|
8
8
|
|
|
9
9
|
const packageRoot = path.resolve(__dirname, "..");
|
|
10
|
-
const sourceSkillDir =
|
|
10
|
+
const sourceSkillDir = packageRoot;
|
|
11
11
|
const sourceHooksPath = path.join(packageRoot, "hooks.json");
|
|
12
12
|
const pythonScript = path.join(sourceSkillDir, "scripts", "context_guard.py");
|
|
13
|
+
const skillInstallEntries = [
|
|
14
|
+
"SKILL.md",
|
|
15
|
+
"README.md",
|
|
16
|
+
"README.zh-CN.md",
|
|
17
|
+
"agents",
|
|
18
|
+
"references",
|
|
19
|
+
"scripts",
|
|
20
|
+
"tests"
|
|
21
|
+
];
|
|
13
22
|
|
|
14
23
|
function usage() {
|
|
15
24
|
console.log(`Context Guard Skill
|
|
16
25
|
|
|
17
26
|
Usage:
|
|
18
|
-
context-guard install [--target <dir>] [--with-hooks] [--hooks-target <file>]
|
|
27
|
+
context-guard install [--target <dir>] [--with-hooks] [--hooks-target <file>] [--config-target <file>]
|
|
19
28
|
context-guard path
|
|
20
29
|
context-guard <context_guard.py command> [args...]
|
|
21
30
|
|
|
@@ -54,11 +63,18 @@ function defaultHooksTarget() {
|
|
|
54
63
|
return path.join(codexHome, "hooks.json");
|
|
55
64
|
}
|
|
56
65
|
|
|
66
|
+
function defaultConfigTarget() {
|
|
67
|
+
const codexHome = process.env.CODEX_HOME || path.join(os.homedir(), ".codex");
|
|
68
|
+
return path.join(codexHome, "config.toml");
|
|
69
|
+
}
|
|
70
|
+
|
|
57
71
|
function parseInstallArgs(args) {
|
|
58
72
|
const options = {
|
|
59
73
|
target: defaultSkillTarget(),
|
|
60
74
|
withHooks: false,
|
|
61
75
|
hooksTarget: defaultHooksTarget(),
|
|
76
|
+
configTarget: defaultConfigTarget(),
|
|
77
|
+
configTargetExplicit: false,
|
|
62
78
|
dryRun: false
|
|
63
79
|
};
|
|
64
80
|
|
|
@@ -74,6 +90,14 @@ function parseInstallArgs(args) {
|
|
|
74
90
|
const value = args[++i];
|
|
75
91
|
if (!value) fail("--hooks-target requires a file path");
|
|
76
92
|
options.hooksTarget = path.resolve(expandHome(value));
|
|
93
|
+
if (!options.configTargetExplicit) {
|
|
94
|
+
options.configTarget = path.join(path.dirname(options.hooksTarget), "config.toml");
|
|
95
|
+
}
|
|
96
|
+
} else if (arg === "--config-target") {
|
|
97
|
+
const value = args[++i];
|
|
98
|
+
if (!value) fail("--config-target requires a file path");
|
|
99
|
+
options.configTarget = path.resolve(expandHome(value));
|
|
100
|
+
options.configTargetExplicit = true;
|
|
77
101
|
} else if (arg === "--dry-run") {
|
|
78
102
|
options.dryRun = true;
|
|
79
103
|
} else if (arg === "-h" || arg === "--help") {
|
|
@@ -86,6 +110,60 @@ function parseInstallArgs(args) {
|
|
|
86
110
|
return options;
|
|
87
111
|
}
|
|
88
112
|
|
|
113
|
+
function migrateHooksFeatureConfig(configTarget, dryRun) {
|
|
114
|
+
const original = fs.existsSync(configTarget) ? fs.readFileSync(configTarget, "utf8") : "";
|
|
115
|
+
const lines = original ? original.replace(/\r\n/g, "\n").split("\n") : [];
|
|
116
|
+
const sectionStart = lines.findIndex((line) => /^\s*\[features\]\s*(?:#.*)?$/.test(line));
|
|
117
|
+
let nextLines = lines.slice();
|
|
118
|
+
|
|
119
|
+
if (sectionStart === -1) {
|
|
120
|
+
while (nextLines.length && nextLines[nextLines.length - 1] === "") nextLines.pop();
|
|
121
|
+
if (nextLines.length) nextLines.push("");
|
|
122
|
+
nextLines.push("[features]", "hooks = true", "");
|
|
123
|
+
} else {
|
|
124
|
+
let sectionEnd = nextLines.length;
|
|
125
|
+
for (let i = sectionStart + 1; i < nextLines.length; i += 1) {
|
|
126
|
+
if (/^\s*\[[^\]]+\]/.test(nextLines[i])) {
|
|
127
|
+
sectionEnd = i;
|
|
128
|
+
break;
|
|
129
|
+
}
|
|
130
|
+
}
|
|
131
|
+
const hooksIndex = nextLines.findIndex(
|
|
132
|
+
(line, index) => index > sectionStart && index < sectionEnd && /^\s*hooks\s*=/.test(line)
|
|
133
|
+
);
|
|
134
|
+
const legacyIndexes = [];
|
|
135
|
+
nextLines.forEach((line, index) => {
|
|
136
|
+
if (index > sectionStart && index < sectionEnd && /^\s*codex_hooks\s*=/.test(line)) legacyIndexes.push(index);
|
|
137
|
+
});
|
|
138
|
+
|
|
139
|
+
if (hooksIndex >= 0) {
|
|
140
|
+
nextLines[hooksIndex] = nextLines[hooksIndex].replace(/^\s*hooks\s*=.*$/, "hooks = true");
|
|
141
|
+
for (const index of legacyIndexes.reverse()) nextLines.splice(index, 1);
|
|
142
|
+
} else if (legacyIndexes.length) {
|
|
143
|
+
const first = legacyIndexes.shift();
|
|
144
|
+
nextLines[first] = nextLines[first].replace(/^\s*codex_hooks\s*=.*$/, "hooks = true");
|
|
145
|
+
for (const index of legacyIndexes.reverse()) nextLines.splice(index, 1);
|
|
146
|
+
} else {
|
|
147
|
+
nextLines.splice(sectionStart + 1, 0, "hooks = true");
|
|
148
|
+
}
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
const next = `${nextLines.join("\n").replace(/\n*$/, "")}\n`;
|
|
152
|
+
if (next === original.replace(/\r\n/g, "\n")) return;
|
|
153
|
+
if (dryRun) {
|
|
154
|
+
console.log(`[context-guard-skill] would enable hooks in ${configTarget}`);
|
|
155
|
+
return;
|
|
156
|
+
}
|
|
157
|
+
fs.mkdirSync(path.dirname(configTarget), { recursive: true });
|
|
158
|
+
if (fs.existsSync(configTarget)) {
|
|
159
|
+
const backupPath = `${configTarget}.bak-${new Date().toISOString().replace(/[:.]/g, "-")}`;
|
|
160
|
+
fs.copyFileSync(configTarget, backupPath);
|
|
161
|
+
console.log(`[context-guard-skill] backed up config: ${backupPath}`);
|
|
162
|
+
}
|
|
163
|
+
fs.writeFileSync(configTarget, next);
|
|
164
|
+
console.log(`[context-guard-skill] enabled hooks: ${configTarget}`);
|
|
165
|
+
}
|
|
166
|
+
|
|
89
167
|
function copySkill(target, dryRun) {
|
|
90
168
|
if (!fs.existsSync(path.join(sourceSkillDir, "SKILL.md"))) {
|
|
91
169
|
fail(`source skill folder is missing: ${sourceSkillDir}`);
|
|
@@ -96,7 +174,13 @@ function copySkill(target, dryRun) {
|
|
|
96
174
|
}
|
|
97
175
|
fs.rmSync(target, { recursive: true, force: true });
|
|
98
176
|
fs.mkdirSync(path.dirname(target), { recursive: true });
|
|
99
|
-
fs.
|
|
177
|
+
fs.mkdirSync(target, { recursive: true });
|
|
178
|
+
for (const entry of skillInstallEntries) {
|
|
179
|
+
const from = path.join(sourceSkillDir, entry);
|
|
180
|
+
if (!fs.existsSync(from)) continue;
|
|
181
|
+
const to = path.join(target, entry);
|
|
182
|
+
fs.cpSync(from, to, { recursive: true });
|
|
183
|
+
}
|
|
100
184
|
console.log(`[context-guard-skill] installed skill: ${target}`);
|
|
101
185
|
}
|
|
102
186
|
|
|
@@ -159,6 +243,7 @@ function install(args) {
|
|
|
159
243
|
const options = parseInstallArgs(args);
|
|
160
244
|
copySkill(options.target, options.dryRun);
|
|
161
245
|
if (options.withHooks) {
|
|
246
|
+
migrateHooksFeatureConfig(options.configTarget, options.dryRun);
|
|
162
247
|
installHooks(options.target, options.hooksTarget, options.dryRun);
|
|
163
248
|
}
|
|
164
249
|
}
|
package/hooks.json
CHANGED
|
@@ -13,6 +13,18 @@
|
|
|
13
13
|
]
|
|
14
14
|
}
|
|
15
15
|
],
|
|
16
|
+
"SubagentStart": [
|
|
17
|
+
{
|
|
18
|
+
"hooks": [
|
|
19
|
+
{
|
|
20
|
+
"type": "command",
|
|
21
|
+
"command": "python3 ./scripts/context_guard_hook.py subagent-start",
|
|
22
|
+
"timeout": 10,
|
|
23
|
+
"statusMessage": "Initializing subagent context"
|
|
24
|
+
}
|
|
25
|
+
]
|
|
26
|
+
}
|
|
27
|
+
],
|
|
16
28
|
"UserPromptSubmit": [
|
|
17
29
|
{
|
|
18
30
|
"hooks": [
|
|
@@ -25,13 +37,25 @@
|
|
|
25
37
|
]
|
|
26
38
|
}
|
|
27
39
|
],
|
|
40
|
+
"SubagentStop": [
|
|
41
|
+
{
|
|
42
|
+
"hooks": [
|
|
43
|
+
{
|
|
44
|
+
"type": "command",
|
|
45
|
+
"command": "python3 ./scripts/context_guard_hook.py subagent-stop",
|
|
46
|
+
"timeout": 900,
|
|
47
|
+
"statusMessage": "Checking subagent context and Test Hub"
|
|
48
|
+
}
|
|
49
|
+
]
|
|
50
|
+
}
|
|
51
|
+
],
|
|
28
52
|
"Stop": [
|
|
29
53
|
{
|
|
30
54
|
"hooks": [
|
|
31
55
|
{
|
|
32
56
|
"type": "command",
|
|
33
57
|
"command": "python3 ./scripts/context_guard_hook.py stop",
|
|
34
|
-
"timeout":
|
|
58
|
+
"timeout": 900,
|
|
35
59
|
"statusMessage": "Checking context checkpoint"
|
|
36
60
|
}
|
|
37
61
|
]
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@michelj/context-guard",
|
|
3
|
-
"version": "0.4.
|
|
3
|
+
"version": "0.4.3",
|
|
4
4
|
"description": "Install and run the Context Guard Codex skill.",
|
|
5
5
|
"repository": {
|
|
6
6
|
"type": "git",
|
|
@@ -17,13 +17,17 @@
|
|
|
17
17
|
"files": [
|
|
18
18
|
"README.md",
|
|
19
19
|
"README.zh-CN.md",
|
|
20
|
+
"SKILL.md",
|
|
20
21
|
"hooks.json",
|
|
22
|
+
"agents/",
|
|
21
23
|
"bin/",
|
|
22
|
-
"
|
|
24
|
+
"references/",
|
|
25
|
+
"scripts/",
|
|
26
|
+
"tests/"
|
|
23
27
|
],
|
|
24
28
|
"scripts": {
|
|
25
29
|
"postinstall": "node bin/postinstall.js",
|
|
26
|
-
"test": "bash tests/npm-install-smoke.sh && python3 -m py_compile
|
|
30
|
+
"test": "bash tests/npm-install-smoke.sh && python3 -m py_compile scripts/context_guard.py scripts/context_guard_hook.py"
|
|
27
31
|
},
|
|
28
32
|
"engines": {
|
|
29
33
|
"node": ">=18"
|
|
@@ -6,9 +6,12 @@ Use this structure for project context:
|
|
|
6
6
|
<opened Codex project root>/
|
|
7
7
|
.codex/context/
|
|
8
8
|
├── index.md
|
|
9
|
+
├── user-messages.md
|
|
9
10
|
├── roadmap.md
|
|
10
11
|
├── bad-cases.md
|
|
11
12
|
├── preferences.json
|
|
13
|
+
├── private/
|
|
14
|
+
│ └── secrets.local.json
|
|
12
15
|
├── roadmap/
|
|
13
16
|
│ ├── roadmap.html
|
|
14
17
|
│ ├── roadmap-details.html
|
|
@@ -21,6 +24,7 @@ Use this structure for project context:
|
|
|
21
24
|
├── task-cases/
|
|
22
25
|
├── test-hub/
|
|
23
26
|
│ ├── registry.json
|
|
27
|
+
│ ├── feature-chains.json
|
|
24
28
|
│ ├── last-run.json
|
|
25
29
|
│ └── runs/
|
|
26
30
|
├── bad-case-tests/
|
|
@@ -29,6 +33,8 @@ Use this structure for project context:
|
|
|
29
33
|
|
|
30
34
|
The opened Codex project root is the local folder selected in Codex or the local workspace root for the current thread. Do not place this folder inside a skill installation directory, remote SSH path, chat/thread-specific folder, or temporary execution directory unless the user explicitly asks that location to own its own context.
|
|
31
35
|
|
|
36
|
+
The `private/` folder is local-only sensitive memory. It must be covered by `.codex/.gitignore`, use restrictive local permissions where possible, and never be projected into Roadmap HTML, agent-readable exports, README files, logs, or final answers.
|
|
37
|
+
|
|
32
38
|
## preferences.json
|
|
33
39
|
|
|
34
40
|
```json
|
|
@@ -84,6 +90,39 @@ Keep Quick Scan to these four lines unless the user explicitly asks for a fuller
|
|
|
84
90
|
Keep only concise summaries here. Move detailed stale context to `.codex/context/archive/`.
|
|
85
91
|
```
|
|
86
92
|
|
|
93
|
+
## user-messages.md
|
|
94
|
+
|
|
95
|
+
```md
|
|
96
|
+
# User Message Memory
|
|
97
|
+
|
|
98
|
+
This file preserves concise user wording that future Codex turns may need. It is agent-readable context, not a public transcript.
|
|
99
|
+
|
|
100
|
+
## Recent User Signals
|
|
101
|
+
|
|
102
|
+
### YYYY-MM-DD HH:MM:SS
|
|
103
|
+
|
|
104
|
+
- Mode: verbatim | summary | redacted | ephemeral
|
|
105
|
+
- User message: short original user wording, concise summary for large inputs, or redacted message when secrets are present
|
|
106
|
+
- Secret pointer: USER-SECRET-YYYYMMDD-HHMMSS | ephemeral-not-stored | omitted when not relevant
|
|
107
|
+
- Use: Preserve this wording when deciding task direction, constraints, credentials, preferences, bad-case intake, or roadmap `User request` fields.
|
|
108
|
+
|
|
109
|
+
## Durable User Constraints
|
|
110
|
+
|
|
111
|
+
- User preference or rule that should affect future turns.
|
|
112
|
+
|
|
113
|
+
## Secret Pointers
|
|
114
|
+
|
|
115
|
+
- USER-SECRET-YYYYMMDD-HHMMSS: redacted purpose only; raw value lives in `.codex/context/private/secrets.local.json`.
|
|
116
|
+
```
|
|
117
|
+
|
|
118
|
+
Rules:
|
|
119
|
+
|
|
120
|
+
- Record short user prompts near-verbatim when they contain requirements, preferences, constraints, route changes, bad-case reports, server details, or credentials needed for the task.
|
|
121
|
+
- Summarize large pasted files, logs, attachments, or generated blobs instead of copying them wholesale.
|
|
122
|
+
- Never store raw secrets outside `.codex/context/private/` or a secure OS credential store.
|
|
123
|
+
- Do not persist OTP or short-lived one-time verification codes; record only `ephemeral-not-stored`.
|
|
124
|
+
- Promote durable user requirements into task context or roadmap `User request:` fields, then keep this file as a concise source of recent user wording.
|
|
125
|
+
|
|
87
126
|
## roadmap.md
|
|
88
127
|
|
|
89
128
|
```md
|
|
@@ -216,12 +255,19 @@ Use task cases for realistic multi-step verification flows. They should catch bu
|
|
|
216
255
|
- Use `Level: major` for significant milestones shown as main route cards; use `Level: checkpoint` for minor progress that should live in details.
|
|
217
256
|
- Use `Branch:` for forked or parallel routes. Missing `Branch:` means `Main`; use `Parent:` to point to the node where a branch forked.
|
|
218
257
|
- In the human overview, visible card numbers should be consecutive per route group after checkpoint filtering, not source node numbers with gaps.
|
|
219
|
-
- If the human overview has multiple route groups, show all route lines together with parent/fork markers and
|
|
258
|
+
- If the human overview has multiple route groups, show all route lines together with parent/fork markers; keep tests and bad cases in clicked node details, Test Hub, and agent-readable exports by default.
|
|
220
259
|
- Each node should be concise enough for Codex to scan quickly: outcome, decision, next step, linked bad cases.
|
|
221
260
|
- Link nodes to bad cases and test-chain notes instead of duplicating full details.
|
|
222
261
|
- Treat human-facing test coverage as human-designed bad-case recurrence detection. Single-route overview hides the test lane; multi-route compact test routes should be generated only from user-approved tests with explicit `Run policy`, approved task-case checkpoints, or approved test registry entries, not from ordinary linked bad-case guards or roadmap node checkpoint logs.
|
|
223
262
|
- In branch or multi-route views, align each visible test item to the roadmap node whose approved bad-case test or approved task-case checkpoint it covers. If a route has no approved tests, do not show a test route for it. Empty test slots should be subtle timeline placeholders only when the route has at least one approved test elsewhere.
|
|
224
263
|
- Prefer a task-oriented case in `.codex/context/task-cases/` when realistic workflow phases matter more than isolated bug checks. Bad-case guards should often point to a task-case checkpoint that covers them.
|
|
264
|
+
- Prefer a feature chain in `.codex/context/test-hub/feature-chains.json` when one user-visible feature or workflow can cover several bad cases through checkpoint coverage. Attach new bad cases to existing chain checkpoints before proposing a new chain.
|
|
265
|
+
- Before proposing a new feature chain for a bad case, run `context_guard.py feature-chain-plan --root <project> --query "<bad case or feature text>"`; this is a read-only intake to either review an existing chain or propose a `feature-chain-propose` skeleton with checkpoint coverage state. Do not use `feature-chain-add` as the planner skeleton.
|
|
266
|
+
- Before approving automation, or when proposed feature chains sound similar, run `context_guard.py feature-chain-overlap --root <project>`; this is a read-only duplicate-chain audit to help merge or extend an existing workflow instead of creating another always-run test.
|
|
267
|
+
- `feature-chain-add` creates proposed chains by default. After the user confirms the business flow and test design, use `feature-chain-approve --chain-id <id> --command-text "<approved command>"` to promote the same chain. Do not hand-edit `feature-chains.json` or create a duplicate approved chain to bypass the approval gate.
|
|
268
|
+
- If the user says an approved feature chain should not run every time, use `feature-chain-set-policy --chain-id <id> --run-policy <policy> --reason <reason>` to update the same chain. Do not delete or duplicate feature chains just to change cadence.
|
|
269
|
+
- After editing feature chains, run `context_guard.py validate-feature-chains --root <project>` to catch missing entries, exits, approved automation, checkpoint checks, linked bad-case coverage, and artifact policy issues. This validates structure only; it does not approve business test design.
|
|
270
|
+
- Feature-chain commands may emit `CG_CHECKPOINT:<checkpoint title or id>:PASS` or `CG_CHECKPOINT:<checkpoint title or id>:FAIL:<short reason>`; Test Hub treats any `FAIL` marker as a failed chain and reports that checkpoint. Marker names must match registered checkpoint titles or ids; unknown markers are test-chain failures.
|
|
225
271
|
- Task-case scripts or agents should log the phase/checkpoint that failed, so Codex can locate the broken workflow step without re-debugging the whole task.
|
|
226
272
|
- Before writing any new durable task-case script or active task case, ask the user to confirm with only the business path: from what state to what state, the main task, and the major risk. Keep technical phases/checkpoints/logs inside the task-case file, not in the confirmation prompt. If confirmation is unavailable, keep the case `proposed` and avoid broad new scripts.
|
|
227
273
|
- When the user explicitly asks to create, write, generate, design, or add a test/test task/task case, start the user-visible response with `测试创建识别:...` or the folder-language equivalent, then summarize the test target from what state to what state and the main risk it catches.
|
|
@@ -239,6 +285,8 @@ Use task cases for realistic multi-step verification flows. They should catch bu
|
|
|
239
285
|
- Keep source records in the configured `.codex/context/preferences.json` record language. The HTML roadmap should follow that preference and should not show a visible language selector by default.
|
|
240
286
|
- During goal mode or long-running autonomous work, keep the active goal aligned to the current task, add compact goal checkpoints during meaningful phase changes, and record bad cases as soon as they appear.
|
|
241
287
|
- Treat `.codex/context/index.md`, `.codex/context/roadmap.md`, `.codex/context/bad-cases.md`, and task context files as the source of truth.
|
|
288
|
+
- Treat `.codex/context/user-messages.md` as the source for recent user wording and durable user signals. Use it to populate roadmap `User request:` fields and avoid asking the user to repeat short instructions.
|
|
289
|
+
- Treat `.codex/context/private/` as local-only sensitive memory. Never expose raw secrets in human-facing HTML or git-tracked context.
|
|
242
290
|
- Treat `.codex/context/roadmap/roadmap.html` as a human-facing view only. Codex should not use it for context intake or bad-case management.
|
|
243
291
|
- Treat `.codex/context/roadmap/roadmap.md` and `.codex/context/roadmap/roadmap.json` as stable agent-readable exports for quick scanning, route lookup, bad-case lookup, and recurrence-guard lookup, not as primary editable sources.
|
|
244
292
|
- Keep `NODE-...`, `BC-...`, and `CTX-...` IDs in source files for linking, but hide them in the default human-facing HTML. Show short natural-language node and bad-case labels instead.
|
|
@@ -280,6 +328,9 @@ Use task cases for realistic multi-step verification flows. They should catch bu
|
|
|
280
328
|
## Pruning Rules
|
|
281
329
|
|
|
282
330
|
- Do not record normal implementation chatter.
|
|
331
|
+
- Do not drop short user messages that contain requirements, constraints, preferences, credentials, route changes, or bad-case reports.
|
|
332
|
+
- Do not paste huge files or logs into user-message memory; summarize them.
|
|
333
|
+
- Do not store raw secrets in public context files, roadmap HTML, exports, logs, or final answers.
|
|
283
334
|
- Do not record every command; record only commands that prove a checkpoint or guard a bad case.
|
|
284
335
|
- Do not let roadmap node `Test chain:` history replace bad-case recurrence guards in user-facing roadmap output.
|
|
285
336
|
- Do not split a real workflow into many unrelated bug-level tests when one task case with checkpoints would reveal the failure location more clearly.
|
|
@@ -0,0 +1,228 @@
|
|
|
1
|
+
# Feature Chain Methodology
|
|
2
|
+
|
|
3
|
+
Use feature chains when the goal is to prevent fixed bad cases from reappearing without creating one durable test per bad case.
|
|
4
|
+
|
|
5
|
+
## Core Idea
|
|
6
|
+
|
|
7
|
+
The durable test unit is a user-visible feature or workflow. A bad case is coverage attached to one checkpoint inside that workflow.
|
|
8
|
+
|
|
9
|
+
```text
|
|
10
|
+
Feature chain
|
|
11
|
+
Entry: the real trigger users or Codex will perform
|
|
12
|
+
Checkpoint 1: expected intermediate state
|
|
13
|
+
Covers: BC-...
|
|
14
|
+
Checkpoint 2: expected transition or output
|
|
15
|
+
Covers: BC-..., BC-...
|
|
16
|
+
Exit check: strict final green condition
|
|
17
|
+
```
|
|
18
|
+
|
|
19
|
+
## Creation Rule
|
|
20
|
+
|
|
21
|
+
When a bad case appears:
|
|
22
|
+
|
|
23
|
+
1. Identify the feature entry that can reproduce or guard the symptom.
|
|
24
|
+
2. Search existing feature chains for the same entry, workflow, component, route, or service. Use the read-only planning helper first:
|
|
25
|
+
|
|
26
|
+
```bash
|
|
27
|
+
python3 ~/.agents/skills/context-guard/scripts/context_guard.py feature-chain-plan --root <project> --query "<bad case or feature text>"
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
The query can be natural language or a `BC-...` ID. The planner is read-only: it either says `action: review-existing-chain` with match evidence and an after-confirmation attach-command skeleton, or `action: propose-new-chain` with a compact user-confirmation prompt and a `feature-chain-propose` command skeleton. The proposed skeleton must contain a checkpoint and explicit coverage state: use `--bad-cases` for a real bad-case seed, or `--coverage-pending-reason` for a user-described test target that has no concrete bad case yet. If the input is only a `BC-...` ID, the prompt should describe the bad case by title or summary instead of asking the user to confirm an opaque ID. It must not create a feature chain, attach a bad case, or approve automation; run the skeleton only after user confirmation.
|
|
31
|
+
|
|
32
|
+
3. Use `feature-chain-suggest` when you only need raw candidate chains/checkpoints. If only a bad-case ID is available, the helper expands it from `bad-cases.md` first so matching uses the case title, summary, phenomenon, trigger, cause, and tags rather than the opaque ID alone.
|
|
33
|
+
4. Use `feature-chain-coverage` when you need the whole register view. It should show covered cases, unassigned candidates, and possible existing chains for visible candidates without changing the registry. Strong suggestions should include short match evidence terms; if the evidence is weak or unreadable, treat the suggestion as planning noise rather than coverage.
|
|
34
|
+
5. Use `feature-chain-candidates` when the unassigned list is too large. It groups unassigned bad cases by shared feature tags, prefers more specific tag combinations over broad single tags, suppresses repeated groups that add little new coverage, and proposes a small set of candidate feature chains. Treat `new coverage` as the key signal: a candidate with high total count but low new coverage may be a subcase of an earlier chain. It must not create, attach, or approve anything; it only helps choose which user-visible flow is worth designing.
|
|
35
|
+
6. Use `feature-chain-overlap` before approving automation, or whenever several proposed chains sound similar. It is read-only and flags pairs that likely describe the same workflow. If it reports overlap, merge the intent or extend one chain before creating another always-run test.
|
|
36
|
+
7. If a chain exists and the match is semantically correct, attach the bad case to the closest checkpoint and tighten that checkpoint.
|
|
37
|
+
8. If no chain exists, propose a new chain in one short business-facing sentence and wait for user confirmation before approving or automating it.
|
|
38
|
+
|
|
39
|
+
For natural-language test requests, keep the user's business intent but remove the request wrapper before writing the confirmation prompt. For example, `写一个测试,检验每次开发完成后 Markdown 编辑器里的单行、多行和矩阵公式都能正常渲染` should become a compact subject such as `Markdown 编辑器里的单行、多行和矩阵公式能正常渲染`, not a verbatim copy of the whole chat sentence. This keeps the prompt useful for human confirmation while avoiding agent-invented workflow details.
|
|
40
|
+
|
|
41
|
+
When the user already gives a workflow shape, preserve it. A request like `创建一个测试任务:从编辑器输入 Markdown 到预览正确渲染,主要验证公式渲染回归` should be confirmed as `从「编辑器输入 Markdown」到「预览正确渲染」,主要验证「公式渲染回归」`. Do not replace explicit entry/exit/risk wording with generic "相关入口到正确结果" language.
|
|
42
|
+
|
|
43
|
+
For this explicit shape, `feature-chain-plan` may prefill the after-confirmation `feature-chain-propose` skeleton with the stated entry and exit check, and may print the stated risk as a suggested checkpoint. This is still only a confirmation aid: it must not create the chain, approve automation, or invent missing checkpoint details before the user confirms the business flow.
|
|
44
|
+
|
|
45
|
+
CLI rule: `feature-chain-add` creates `status: proposed` by default. Do not treat this as an approved test. `feature-chain-add --test-status approved` is not allowed for `every-dev-completion` chains because it skips the user confirmation and approval dry-run gates. Use `feature-chain-approve` on the same proposed chain instead.
|
|
46
|
+
|
|
47
|
+
After the user confirms a candidate flow, use `feature-chain-propose` when you need to record a safe draft with seed bad-case coverage:
|
|
48
|
+
|
|
49
|
+
```bash
|
|
50
|
+
python3 ~/.agents/skills/context-guard/scripts/context_guard.py feature-chain-propose \
|
|
51
|
+
--root <project> \
|
|
52
|
+
--title "<confirmed feature title>" \
|
|
53
|
+
--entry "<confirmed user-visible entry>" \
|
|
54
|
+
--exit-check "<confirmed strict final green condition>" \
|
|
55
|
+
--node-title "<confirmed checkpoint>" \
|
|
56
|
+
--bad-cases "BC-..., BC-..." \
|
|
57
|
+
--check "<checkpoint recurrence check>"
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
This command creates only a `proposed` chain. It records the user-confirmed shape and seed bad cases, but it does not add an executable command and must not enter `dev-complete` until `feature-chain-approve` is run with user-approved automation.
|
|
61
|
+
|
|
62
|
+
If the user confirms the feature flow before any concrete bad case exists, keep the same command but replace `--bad-cases ...` with `--coverage-pending-reason "<why there is no linked bad case yet>"`. This records the draft so it is not lost, but it is not recurrence coverage and cannot be approved for `every-dev-completion` until a real bad case is attached to a checkpoint.
|
|
63
|
+
|
|
64
|
+
When a later bad case appears, run `feature-chain-plan` first. If it points to a coverage-pending chain/checkpoint, attach the bad case there with `feature-chain-attach-bc` and tighten the checkpoint text. The attach step should clear the pending-coverage note, because the checkpoint now has real bad-case coverage.
|
|
65
|
+
|
|
66
|
+
The expected lifecycle is:
|
|
67
|
+
|
|
68
|
+
1. `feature-chain-plan` turns user wording or a bad-case ID into a read-only confirmation prompt.
|
|
69
|
+
2. After user confirmation, `feature-chain-propose` records a non-executable draft with either linked bad cases or a coverage-pending reason.
|
|
70
|
+
3. Later bad cases are routed through `feature-chain-plan` and attached to the nearest existing checkpoint when semantically correct.
|
|
71
|
+
4. `feature-chain-summary` gives the fast coverage map; `feature-chain-overlap` checks duplicate workflow coverage before approval.
|
|
72
|
+
5. `feature-chain-approve` is the only path into the always-run set and must pass the checkpoint dry run.
|
|
73
|
+
6. `dev-complete` runs the approved chain with structured checkpoint markers and cleans success artifacts.
|
|
74
|
+
|
|
75
|
+
This lifecycle is the core experiment: fewer feature chains should cover more bad-case recurrence checks without Codex inventing a broad test suite.
|
|
76
|
+
|
|
77
|
+
Multi-project trials are the sanity check for this method. These trials must start from fresh sandbox projects instead of reusing an existing project, existing context folder, or previous test registry; otherwise the result may only prove that old context happened to work. The sandbox themes should also be genuinely different, such as a life utility, a creative tool, and a game or interaction, not merely three variants of the same engineering workflow. In small linear flows, one feature chain with two to four checkpoints can cover about three related bad cases and localize the failed phase. Treat planner checkpoint suggestions as hints, not business truth: Codex or the user must attach each bad case to the real phase where it can recur. When the workflow has queues, retries, multiple workers, recovery branches, or cross-process cleanup, upgrade the design to a task case instead of stretching a simple feature chain.
|
|
78
|
+
|
|
79
|
+
A single-chain trial is not enough to validate Context Guard itself. A system-level regression should include at least two independent approved feature chains in one fresh project, then prove that `dev-complete` runs both, reports one chain failure without hiding the other chain's pass result, preserves the failing evidence, and returns to all-pass after the same chain is fixed. This checks the Test Hub orchestration layer rather than only the lifecycle of one feature chain.
|
|
80
|
+
|
|
81
|
+
Before approving automation, use a dry run when the proposed command or checkpoint markers need validation:
|
|
82
|
+
|
|
83
|
+
```bash
|
|
84
|
+
python3 ~/.agents/skills/context-guard/scripts/context_guard.py feature-chain-dry-run \
|
|
85
|
+
--root <project> \
|
|
86
|
+
--chain-id FC-YYYYMMDD-001 \
|
|
87
|
+
--command-text "<candidate command>"
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
Dry run executes the candidate command against the proposed chain's registered checkpoints, reports missing/failed/unknown checkpoint markers, cleans success artifacts, and preserves failure evidence under `.codex/context/test-hub/dry-runs/`. It does not approve the chain, does not write the command into `feature-chains.json`, and does not add the chain to `dev-complete`.
|
|
91
|
+
|
|
92
|
+
Dry-run evidence paths must be unique per run. Fast repeated or parallel dry runs must not reuse or overwrite a previous failure directory, because the preserved evidence is what lets Codex locate the failed checkpoint without reinterpreting the whole task.
|
|
93
|
+
|
|
94
|
+
## Approval Rule
|
|
95
|
+
|
|
96
|
+
After the user confirms the feature flow and test design, promote the existing proposed chain with:
|
|
97
|
+
|
|
98
|
+
```bash
|
|
99
|
+
python3 ~/.agents/skills/context-guard/scripts/context_guard.py feature-chain-approve \
|
|
100
|
+
--root <project> \
|
|
101
|
+
--chain-id FC-YYYYMMDD-001 \
|
|
102
|
+
--command-text "<approved command>"
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
Approval is a safety gate and the only supported path from proposed feature-chain automation to `every-dev-completion`. It refuses a chain that has no checkpoint, no checkpoint check text, no linked bad-case coverage, or no automated command when the run policy is `every-dev-completion`. For `every-dev-completion` automation, approval must also run a dry-run preflight before mutating the registry. If the command misses a required checkpoint marker, emits an unknown marker, emits a `FAIL` marker, times out, or hits a blocker, approval fails and the chain stays `proposed`. Do not bypass this by hand-editing `feature-chains.json`, using `feature-chain-add --test-status approved`, or creating a second approved chain.
|
|
106
|
+
|
|
107
|
+
## Policy Rule
|
|
108
|
+
|
|
109
|
+
After approval, the user's cadence still wins. If the user says a feature chain should not run after every development turn, update the existing chain instead of deleting or duplicating it:
|
|
110
|
+
|
|
111
|
+
```bash
|
|
112
|
+
python3 ~/.agents/skills/context-guard/scripts/context_guard.py feature-chain-set-policy \
|
|
113
|
+
--root <project> \
|
|
114
|
+
--chain-id FC-YYYYMMDD-001 \
|
|
115
|
+
--run-policy relevant-only \
|
|
116
|
+
--reason "User said this chain is only needed when touching GPU monitor flows."
|
|
117
|
+
```
|
|
118
|
+
|
|
119
|
+
Keep the reason short and human-readable. Use `disabled-with-reason` only when the user asks to disable the chain or it cannot be run safely.
|
|
120
|
+
|
|
121
|
+
## Design Heuristic
|
|
122
|
+
|
|
123
|
+
A good chain has:
|
|
124
|
+
|
|
125
|
+
- one clear entry point
|
|
126
|
+
- one realistic flow, not a pile of unrelated checks
|
|
127
|
+
- two to five checkpoints
|
|
128
|
+
- strict red and green conditions
|
|
129
|
+
- failure localization, so the runner reports which checkpoint broke
|
|
130
|
+
- cleanup-on-pass and preserve-on-fail behavior
|
|
131
|
+
|
|
132
|
+
Avoid:
|
|
133
|
+
|
|
134
|
+
- one script per bad case when one feature flow can cover them
|
|
135
|
+
- broad suites that run unrelated product areas
|
|
136
|
+
- checks that only prove code executed but cannot catch the old symptom
|
|
137
|
+
- durable tests written from agent guesses without user confirmation
|
|
138
|
+
|
|
139
|
+
## Storage
|
|
140
|
+
|
|
141
|
+
Store chain metadata in:
|
|
142
|
+
|
|
143
|
+
```text
|
|
144
|
+
.codex/context/test-hub/feature-chains.json
|
|
145
|
+
```
|
|
146
|
+
|
|
147
|
+
Store large scenario specs in `.codex/context/task-cases/` only when the workflow needs richer phases, logs, or human-readable execution notes.
|
|
148
|
+
|
|
149
|
+
## Execution
|
|
150
|
+
|
|
151
|
+
Approved chains with `status: approved | active | stable` and `run_policy: every-dev-completion` are part of Test Hub. At development completion, run them through:
|
|
152
|
+
|
|
153
|
+
```bash
|
|
154
|
+
python3 ~/.agents/skills/context-guard/scripts/context_guard.py dev-complete --root <project>
|
|
155
|
+
```
|
|
156
|
+
|
|
157
|
+
The runner should execute the approved command with minimal Codex reinterpretation, clean success artifacts, preserve failure evidence, and report the failed checkpoint or blocker.
|
|
158
|
+
|
|
159
|
+
A feature-chain experiment is not complete just because one happy path passes. It should prove the closed loop: one chain covers multiple bad cases, a failing checkpoint preserves evidence with an actionable reason, and the fixed path passes while cleaning temporary artifacts.
|
|
160
|
+
|
|
161
|
+
When an approved chain fails, do not design a new test to prove the same workflow. Treat the failed checkpoint as the recurrence signal, fix the cause, and rerun the same approved chain. A good runner makes this loop cheap by emitting readable checkpoint markers, preserving only the useful failure evidence, and cleaning success artifacts after the rerun passes.
|
|
162
|
+
|
|
163
|
+
Feature-chain commands can report phase-level status with lightweight markers:
|
|
164
|
+
|
|
165
|
+
```text
|
|
166
|
+
CG_CHECKPOINT:<checkpoint title or id>:PASS
|
|
167
|
+
CG_CHECKPOINT:<checkpoint title or id>:FAIL:<short reason>
|
|
168
|
+
```
|
|
169
|
+
|
|
170
|
+
Test Hub treats any `FAIL` marker as a failed feature chain, even if the command exits 0. Prefer these markers when one command covers several checkpoints, because the preserved result will point to the broken workflow step instead of only saying that the whole command failed.
|
|
171
|
+
|
|
172
|
+
Marker names must match registered checkpoint titles or ids. Unknown markers fail the chain because they usually mean the script no longer matches the approved workflow. Keep non-English checkpoint titles readable and distinct; do not collapse them into generic ids.
|
|
173
|
+
|
|
174
|
+
Approved feature-chain commands must report every registered checkpoint unless a checkpoint is explicitly optional (`optional: true` or `required: false`). Missing markers fail the chain, because an unreported checkpoint was not proven to run.
|
|
175
|
+
|
|
176
|
+
If the user or business flow says one checkpoint should not be required every time, change that checkpoint explicitly instead of weakening the whole chain:
|
|
177
|
+
|
|
178
|
+
```bash
|
|
179
|
+
python3 ~/.agents/skills/context-guard/scripts/context_guard.py feature-chain-set-checkpoint \
|
|
180
|
+
--root <project> \
|
|
181
|
+
--chain-id FC-YYYYMMDD-001 \
|
|
182
|
+
--node-title "前端打开监控页" \
|
|
183
|
+
--required optional \
|
|
184
|
+
--reason "Only runs in browser integration environment."
|
|
185
|
+
```
|
|
186
|
+
|
|
187
|
+
Use `--required required` to restore the checkpoint to every-run coverage. This keeps the chain strict by default while allowing intentional, documented exceptions.
|
|
188
|
+
|
|
189
|
+
Audit required/optional coverage without opening JSON:
|
|
190
|
+
|
|
191
|
+
```bash
|
|
192
|
+
python3 ~/.agents/skills/context-guard/scripts/context_guard.py feature-chain-list --root <project> --verbose
|
|
193
|
+
```
|
|
194
|
+
|
|
195
|
+
Audit the compact coverage map before creating new coverage:
|
|
196
|
+
|
|
197
|
+
```bash
|
|
198
|
+
python3 ~/.agents/skills/context-guard/scripts/context_guard.py feature-chain-summary --root <project>
|
|
199
|
+
```
|
|
200
|
+
|
|
201
|
+
This is the quickest way to see whether a small number of feature chains already covers several bad cases, and which checkpoints are still waiting for a real bad case before approval.
|
|
202
|
+
Treat the `coverage density`, `reuse signal`, and `next:` lines as decision aids: they should push Codex toward reusing or extending an existing workflow when possible, not toward creating another standalone test.
|
|
203
|
+
|
|
204
|
+
Audit possible duplicate feature chains before approval:
|
|
205
|
+
|
|
206
|
+
```bash
|
|
207
|
+
python3 ~/.agents/skills/context-guard/scripts/context_guard.py feature-chain-overlap --root <project>
|
|
208
|
+
```
|
|
209
|
+
|
|
210
|
+
This is a route-choice guard, not a test runner. It compares existing chain wording and linked bad cases, then prints pairs that may be the same workflow. Use it to avoid turning one business flow into multiple always-run tests.
|
|
211
|
+
|
|
212
|
+
Audit bad-case coverage across chains without mutating records:
|
|
213
|
+
|
|
214
|
+
```bash
|
|
215
|
+
python3 ~/.agents/skills/context-guard/scripts/context_guard.py feature-chain-coverage --root <project>
|
|
216
|
+
```
|
|
217
|
+
|
|
218
|
+
Use this to decide whether a new bad case should attach to an existing chain or remain a standalone guard. Treat unassigned cases as candidates, not as required new tests.
|
|
219
|
+
|
|
220
|
+
## Quality Gate
|
|
221
|
+
|
|
222
|
+
After editing feature chains, run:
|
|
223
|
+
|
|
224
|
+
```bash
|
|
225
|
+
python3 ~/.agents/skills/context-guard/scripts/context_guard.py validate-feature-chains --root <project>
|
|
226
|
+
```
|
|
227
|
+
|
|
228
|
+
This gate checks structure, not business judgment. It catches approved chains that are missing an entry point, exit check, automated command, checkpoint nodes, checkpoint check text, linked bad-case coverage, or a clear artifact policy. It does not create tests, approve tests, or decide whether the user's workflow deserves a durable chain.
|