universal-dev-standards 6.12.0 → 6.13.0-beta.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bundled/ai/standards/developer-memory.ai.yaml +24 -2
- package/bundled/ai/standards/turn-completion-integrity.ai.yaml +34 -2
- package/bundled/core/developer-memory.md +58 -2
- package/bundled/core/turn-completion-integrity.md +44 -2
- package/bundled/hooks/check-turn-completion-codex.mjs +143 -0
- package/bundled/hooks/check-turn-completion-gemini.mjs +81 -0
- package/bundled/hooks/check-turn-completion.mjs +24 -149
- package/bundled/hooks/turn-completion/engine.mjs +206 -0
- package/bundled/hooks/turn-completion/locales/en.mjs +50 -2
- package/bundled/hooks/turn-completion/locales/zh-TW.mjs +54 -3
- package/bundled/locales/zh-CN/CHANGELOG.md +27 -3
- package/bundled/locales/zh-CN/README.md +1 -1
- package/bundled/locales/zh-CN/SECURITY.md +1 -0
- package/bundled/locales/zh-CN/core/turn-completion-integrity.md +43 -6
- package/bundled/locales/zh-CN/docs/CLI-INIT-OPTIONS.md +24 -5
- package/bundled/locales/zh-TW/CHANGELOG.md +27 -3
- package/bundled/locales/zh-TW/README.md +1 -1
- package/bundled/locales/zh-TW/SECURITY.md +1 -0
- package/bundled/locales/zh-TW/core/turn-completion-integrity.md +43 -6
- package/bundled/locales/zh-TW/docs/CLI-INIT-OPTIONS.md +24 -5
- package/package.json +1 -1
- package/src/commands/init.js +26 -0
- package/src/commands/uninstall.js +1 -1
- package/src/installers/hooks-installer.js +108 -0
- package/src/uninstallers/hook-uninstaller.js +190 -23
- package/standards-registry.json +7 -7
|
@@ -13,8 +13,8 @@ standard:
|
|
|
13
13
|
- "Respect noise control levels and cooldown periods"
|
|
14
14
|
|
|
15
15
|
meta:
|
|
16
|
-
version: "1.
|
|
17
|
-
updated: "2026-
|
|
16
|
+
version: "1.2.0"
|
|
17
|
+
updated: "2026-09-25"
|
|
18
18
|
source: core/developer-memory.md
|
|
19
19
|
description: Structured system for capturing, retrieving, and surfacing developer experience insights across conversations and projects
|
|
20
20
|
|
|
@@ -128,6 +128,20 @@ standard:
|
|
|
128
128
|
trigger: "confidence < 0.2 OR 180+ days unsurfaced with confidence < 0.5"
|
|
129
129
|
- type: staleness
|
|
130
130
|
trigger: "versioned type with outdated version_bound"
|
|
131
|
+
- type: code-reference
|
|
132
|
+
trigger: "recalled memory cites a file path or symbol (function/class) that no longer exists at that location, or has moved"
|
|
133
|
+
timing: "runs before proactive-surfacing (check before a memory is shown, not after)"
|
|
134
|
+
scope: "file paths and symbol names (functions/classes) only; file:line is out of scope (line numbers drift on unrelated edits — different staleness semantics; see DEC-115 OQ-1)"
|
|
135
|
+
modes:
|
|
136
|
+
degraded:
|
|
137
|
+
when: "no graph engine configured (any tool, no setup — the default)"
|
|
138
|
+
how: "the assistant verifies the referenced path/symbol itself (Glob/Grep/Read) before using the memory — same mechanism as the memory-as-hint-not-conclusion rule"
|
|
139
|
+
source: "knowledge-graph-memory 1.0.0 §2.1 Degraded Mode"
|
|
140
|
+
engine:
|
|
141
|
+
when: "a graph engine is indexed (e.g. EngramGraph)"
|
|
142
|
+
how: "engine query (e.g. `egr refs check`) reports each citation's state: present / moved (with new location) / missing / unresolvable"
|
|
143
|
+
source: "knowledge-graph-memory 1.0.0 §2.2 Service Mode"
|
|
144
|
+
unresolvable_rule: "unresolvable (e.g. a cross-repo reference outside the indexed scope) MUST NOT be treated as present or missing — surface it as its own state"
|
|
131
145
|
- type: revision
|
|
132
146
|
trigger: "2+ needs-revision feedback"
|
|
133
147
|
|
|
@@ -135,6 +149,11 @@ standard:
|
|
|
135
149
|
- id: proactive-surfacing
|
|
136
150
|
trigger: conversation start or matching code pattern/error/decision detected
|
|
137
151
|
instruction: >
|
|
152
|
+
Before surfacing, run the code-reference staleness check
|
|
153
|
+
(review.checks type: code-reference) on each candidate memory's
|
|
154
|
+
cited file path / symbol. Do not surface a memory whose citation
|
|
155
|
+
resolves to missing without flagging it as stale; unresolvable is
|
|
156
|
+
not the same as present or missing.
|
|
138
157
|
Surface top 3 relevant memories when relevance > 0.7.
|
|
139
158
|
Max 5 per trigger. Cooldown 7 days per entry.
|
|
140
159
|
If > 5 matches, summarize into grouped insight.
|
|
@@ -225,3 +244,6 @@ standard:
|
|
|
225
244
|
Category: {category} | Tags: {tags}
|
|
226
245
|
Insight: {insight}
|
|
227
246
|
Context: {context}
|
|
247
|
+
|
|
248
|
+
related_standards:
|
|
249
|
+
- knowledge-graph-memory
|
|
@@ -3,8 +3,8 @@
|
|
|
3
3
|
|
|
4
4
|
id: turn-completion-integrity
|
|
5
5
|
meta:
|
|
6
|
-
version: "1.
|
|
7
|
-
updated: "2026-09-
|
|
6
|
+
version: "1.4.0"
|
|
7
|
+
updated: "2026-09-25"
|
|
8
8
|
source: core/turn-completion-integrity.md
|
|
9
9
|
description: An agent must not end a turn having stated a next action it did not take; enforced at turn end, not by instruction
|
|
10
10
|
related:
|
|
@@ -117,6 +117,36 @@ enforcement:
|
|
|
117
117
|
severity: warning
|
|
118
118
|
timeout_ms: 2000
|
|
119
119
|
|
|
120
|
+
# The check is enforced only where an adapter exists AND a hook is wired into
|
|
121
|
+
# that harness's own config. This is separate from `enforcement:` above, which
|
|
122
|
+
# only describes the Claude Code entry the generic installer walks — Codex and
|
|
123
|
+
# Gemini CLI are installed by their own installCodexHooks/installGeminiHooks
|
|
124
|
+
# functions (cli/src/installers/hooks-installer.js), each writing that
|
|
125
|
+
# harness's own config file and output contract.
|
|
126
|
+
supported_harnesses:
|
|
127
|
+
- harness: claude-code
|
|
128
|
+
event: Stop
|
|
129
|
+
config_file: ".claude/settings.json"
|
|
130
|
+
block_shape: '{"decision":"block","reason":...}, exit 0; silence allows'
|
|
131
|
+
script: scripts/hooks/check-turn-completion.mjs
|
|
132
|
+
- harness: codex
|
|
133
|
+
event: Stop
|
|
134
|
+
config_file: ".codex/hooks.json"
|
|
135
|
+
block_shape: '{"decision":"block","reason":...}, exit 0 — plain text or empty stdout documented as invalid for this event'
|
|
136
|
+
script: scripts/hooks/check-turn-completion-codex.mjs
|
|
137
|
+
known_limit: "R9 (human-directed stop exemption) is best-effort — Codex gives the assistant's last message directly but not the human's; a failed transcript read leaves the human side empty rather than skipping detection"
|
|
138
|
+
- harness: gemini-cli
|
|
139
|
+
event: AfterAgent
|
|
140
|
+
config_file: ".gemini/settings.json"
|
|
141
|
+
block_shape: '{"decision":"deny","reason":...}, exit 0 — the documented preferred path over exit code 2'
|
|
142
|
+
script: scripts/hooks/check-turn-completion-gemini.mjs
|
|
143
|
+
- harness: cursor
|
|
144
|
+
status: evaluated-not-supported
|
|
145
|
+
reason: "Whether Cursor's stop hook can actually block a turn was unresolved as of writing; shipping an adapter against an unverified contract repeats the exact failure R3 exists to prevent"
|
|
146
|
+
- harness: any-other
|
|
147
|
+
status: inactive
|
|
148
|
+
reason: "Same silent-by-default failure as an unsupported language (R8). uds init --with-hooks reports which harnesses it wired a hook into."
|
|
149
|
+
|
|
120
150
|
checklist:
|
|
121
151
|
- The check runs at turn end, not as an instruction to the agent
|
|
122
152
|
- Every failure path exits without blocking
|
|
@@ -129,3 +159,5 @@ checklist:
|
|
|
129
159
|
- The check recognises its own block message and does not read it as the human's
|
|
130
160
|
- The check recognises the itemized blocker ending R2 defines, and does not block it
|
|
131
161
|
- The attribution search excludes the check's own headings and scaffolding
|
|
162
|
+
- Each supported harness's block contract is verified against that harness's own docs, not assumed from another harness
|
|
163
|
+
- The installer only writes a harness's hook config for a harness the adopter selected
|
|
@@ -2,8 +2,8 @@
|
|
|
2
2
|
|
|
3
3
|
> **Language**: English | [繁體中文](../locales/zh-TW/core/developer-memory.md)
|
|
4
4
|
|
|
5
|
-
**Version**: 1.
|
|
6
|
-
**Last Updated**: 2026-
|
|
5
|
+
**Version**: 1.2.0
|
|
6
|
+
**Last Updated**: 2026-09-25
|
|
7
7
|
**Applicability**: All software projects using AI assistants
|
|
8
8
|
**Scope**: universal
|
|
9
9
|
**Owning Spec**: XSPEC-291 (developer-memory has no dedicated feature XSPEC; XSPEC-291 owns it)
|
|
@@ -250,6 +250,26 @@ recorded so each value is auditable rather than arbitrary.
|
|
|
250
250
|
| `validity.type == "versioned"` AND version is outdated | Flag as stale |
|
|
251
251
|
| `validity.type == "temporal"` AND `expires_at` passed | Flag as expired |
|
|
252
252
|
| `validity.type == "evergreen"` | Skip version check |
|
|
253
|
+
| Recalled memory cites a file path or symbol (function/class) that no longer exists at that location, or has moved | Flag as `code-reference` stale — see below |
|
|
254
|
+
|
|
255
|
+
#### Code-Reference Staleness (`code-reference`)
|
|
256
|
+
|
|
257
|
+
Code moves; a memory recorded against an old location does not update itself. This check catches the case where the memory's insight is still true but its citation — a file path, or a function/class name — is not.
|
|
258
|
+
|
|
259
|
+
This reuses the **two operating modes [Knowledge Graph Memory](knowledge-graph-memory.md) §2 already defines**, rather than inventing a third:
|
|
260
|
+
|
|
261
|
+
| Mode | When | How |
|
|
262
|
+
|------|------|-----|
|
|
263
|
+
| Degraded (no graph engine) | Any tool, no extra setup — the default | Before using a memory, the assistant confirms the cited path/symbol still exists itself (`Glob`/`Grep`/`Read`) — the same mechanism already required by §5 Memory Verification Principle |
|
|
264
|
+
| Engine (a graph engine is indexed, e.g. [EngramGraph](https://github.com/AsiaOstrich/EngramGraph)'s `egr refs check`) | When available | The engine reports each citation's state: `present` / `moved` (with its new location) / `missing` / `unresolvable` |
|
|
265
|
+
|
|
266
|
+
> A correct implementation produces the same answer shape in both modes (Knowledge Graph Memory §2.2) — engine mode is faster and more complete, not different in kind.
|
|
267
|
+
|
|
268
|
+
**`unresolvable` MUST NOT be treated as `present` or `missing`.** It means the checker could not determine an answer for that citation (e.g. a cross-repo reference outside the indexed scope, per Knowledge Graph Memory §2.2's cross-domain caveat) — that uncertainty must reach the assistant as its own state, not collapse silently into a false positive (treated as still valid) or a false negative (treated as gone).
|
|
269
|
+
|
|
270
|
+
**Timing**: this check runs as part of `proactive-surfacing` (§4.1) — before a memory is shown, not after. A memory whose citation resolves to `missing` is not silently surfaced as if nothing changed.
|
|
271
|
+
|
|
272
|
+
**Scope (first batch)**: file paths and symbol names (function/class names) only. `file:line` references are explicitly **out of scope** — a line number drifts on every unrelated edit to the file, which is a different kind of staleness from "this file/symbol no longer exists" (see DEC-115 OQ-1; revisited by 2027-01-31).
|
|
253
273
|
|
|
254
274
|
#### Revision Suggestions
|
|
255
275
|
|
|
@@ -280,6 +300,7 @@ recorded so each value is auditable rather than arbitrary.
|
|
|
280
300
|
| Cooldown period | 7 days per entry | Prevent repetitive suggestions |
|
|
281
301
|
| Max per trigger | 3–5 entries | Information overload prevention |
|
|
282
302
|
| Overflow handling | AI summarizes into grouped insight | When > 5 matches found |
|
|
303
|
+
| Code-reference check | Run before surfacing (§3.4 `code-reference` staleness, degraded or engine mode) | Do not surface a memory whose file/symbol citation no longer resolves without flagging it |
|
|
283
304
|
|
|
284
305
|
#### Surfacing Format
|
|
285
306
|
|
|
@@ -541,6 +562,7 @@ by_category:
|
|
|
541
562
|
- [AI Instruction Standards](ai-instruction-standards.md) — Token-efficient format for memory system instructions
|
|
542
563
|
- [AI-Friendly Architecture](ai-friendly-architecture.md) — Project structure enabling memory integration
|
|
543
564
|
- [Documentation Writing Standards](documentation-writing-standards.md) — Writing quality for memory entries
|
|
565
|
+
- [Knowledge Graph Memory](knowledge-graph-memory.md) — Source of the degraded/engine dual-mode reused by `code-reference` staleness (§3.4)
|
|
544
566
|
|
|
545
567
|
---
|
|
546
568
|
|
|
@@ -587,10 +609,44 @@ If user feedback reveals:
|
|
|
587
609
|
|
|
588
610
|
---
|
|
589
611
|
|
|
612
|
+
## 11. Tool Adoption Example: Code-Reference Check (Non-Normative)
|
|
613
|
+
|
|
614
|
+
This section illustrates one way to wire the `code-reference` staleness check (§3.4) into a tool's own automation. It is documentation only — UDS does not ship this hook, and `uds init`/`uds update` do not install it.
|
|
615
|
+
|
|
616
|
+
### Claude Code
|
|
617
|
+
|
|
618
|
+
Claude Code's automatic memory lives outside the repo, at `~/.claude/projects/<project-path>/memory/`. A `SessionStart` hook can run the check against that directory before the session's first memory surfacing, falling back to degraded mode when no graph engine is present:
|
|
619
|
+
|
|
620
|
+
```bash
|
|
621
|
+
#!/usr/bin/env bash
|
|
622
|
+
# Illustrative only — not installed by any UDS command.
|
|
623
|
+
MEMORY_DIR="$HOME/.claude/projects/$(pwd | tr '/' '-')/memory"
|
|
624
|
+
|
|
625
|
+
if command -v egr >/dev/null 2>&1 && [ -f .engram/graph.db ]; then
|
|
626
|
+
# Engine mode (§3.4): egr resolves each cited path/symbol.
|
|
627
|
+
egr refs check "$MEMORY_DIR"
|
|
628
|
+
else
|
|
629
|
+
# Degraded mode (§3.4): no engine configured — nothing to run up
|
|
630
|
+
# front; the assistant verifies each citation itself before use (§5).
|
|
631
|
+
echo "[developer-memory] no graph engine detected; degraded mode applies"
|
|
632
|
+
fi
|
|
633
|
+
```
|
|
634
|
+
|
|
635
|
+
### Other Tools
|
|
636
|
+
|
|
637
|
+
A tool without a session-start hook still satisfies this standard through either:
|
|
638
|
+
|
|
639
|
+
- a rule in the repo's own instruction file (`CLAUDE.md`, `AGENTS.md`, `.cursorrules`, `.clinerules`, `.windsurfrules`, `copilot-instructions.md`, `GEMINI.md`, `.roo/rules/`, etc.) instructing the assistant to verify a memory's cited path/symbol before relying on it; or
|
|
640
|
+
- a pre-commit check that scans the repo's own instruction file(s) for citations that no longer resolve.
|
|
641
|
+
|
|
642
|
+
---
|
|
643
|
+
|
|
590
644
|
## Version History
|
|
591
645
|
|
|
592
646
|
| Version | Date | Changes |
|
|
593
647
|
|---------|------|---------|
|
|
648
|
+
| 1.2.0 | 2026-09-25 | Added: `code-reference` staleness check to Review (§3.4) — a memory citing a moved/missing file path or symbol is flagged before surfacing; reuses Knowledge Graph Memory's degraded/engine dual mode instead of inventing a third; `unresolvable` may not be read as present or missing; scope is file paths and symbol names only (`file:line` explicitly out of scope, DEC-115 OQ-1); §11 adds a non-normative Claude Code adoption example (DEC-115-L1) |
|
|
649
|
+
| 1.1.1 | 2026-06-18 | Added: `Owning Spec` header pointing to XSPEC-291 (patch, no behavioral change; XSPEC-291 §11 disposition) |
|
|
594
650
|
| 1.1.0 | 2026-06-18 | Added: rationale column + configurable note to Retirement Suggestions thresholds (XSPEC-292 T8) |
|
|
595
651
|
| 1.0.0 | 2026-02-07 | Initial standard: schema, 4 operations, proactive protocol, noise control, architecture decision (Always-On Protocol) |
|
|
596
652
|
|
|
@@ -2,8 +2,8 @@
|
|
|
2
2
|
|
|
3
3
|
> **Language**: English | [繁體中文](../locales/zh-TW/core/turn-completion-integrity.md)
|
|
4
4
|
|
|
5
|
-
**Version**: 1.
|
|
6
|
-
**Last Updated**: 2026-09-
|
|
5
|
+
**Version**: 1.4.0
|
|
6
|
+
**Last Updated**: 2026-09-25
|
|
7
7
|
**Applicability**: Any harness where an agent ends a turn and hands control back to a human
|
|
8
8
|
**Scope**: universal
|
|
9
9
|
**Industry Standards**: none claimed — derived from observed failures, see Evidence
|
|
@@ -146,6 +146,46 @@ prevent, one level up.
|
|
|
146
146
|
|
|
147
147
|
---
|
|
148
148
|
|
|
149
|
+
## Supported harnesses
|
|
150
|
+
|
|
151
|
+
The check is enforced only where a harness adapter exists and a hook is
|
|
152
|
+
actually wired into that harness's own config. As of v1.4.0:
|
|
153
|
+
|
|
154
|
+
| Harness | Event | Config file | Block contract |
|
|
155
|
+
|---|---|---|---|
|
|
156
|
+
| Claude Code | Stop | `.claude/settings.json` | stdout `{"decision":"block","reason":...}`, exit 0; silence allows |
|
|
157
|
+
| Codex | Stop | `.codex/hooks.json` | stdout `{"decision":"block","reason":...}`, exit 0 — plain text or empty stdout is documented as invalid for this event |
|
|
158
|
+
| Gemini CLI | AfterAgent | `.gemini/settings.json` | stdout `{"decision":"deny","reason":...}`, exit 0 — the documented preferred path over exit code 2 |
|
|
159
|
+
|
|
160
|
+
Codex's R9 exemption is best-effort, not silent failure: Codex's Stop payload
|
|
161
|
+
gives the assistant's final message directly but not the human's, so reading
|
|
162
|
+
the human side requires parsing a transcript file. Confirmed against a real
|
|
163
|
+
codex-cli 0.156.1 installation (2026-09-26): a user message in
|
|
164
|
+
`~/.codex/sessions/**/*.jsonl` is a `{"type":"response_item","payload":
|
|
165
|
+
{"type":"message","role":"user","content":[{"type":"input_text","text":...}]}}`
|
|
166
|
+
record — the message lives under `payload`, not at the top level of the
|
|
167
|
+
line or under a `message` key, and a `role: "developer"` record on the same
|
|
168
|
+
shape is not a human message. The 6.13.0-beta.1 adapter read neither of the
|
|
169
|
+
two shapes it tried against this real one, so R9 never exempted a turn on
|
|
170
|
+
Codex; fixed for 6.13.0-beta.2. A failed parse (or an unrecognized record
|
|
171
|
+
shape) still leaves the human side empty rather than throwing — detection
|
|
172
|
+
still runs on the assistant's message, only the R9 exemption for that one
|
|
173
|
+
turn may be missed.
|
|
174
|
+
|
|
175
|
+
Cursor was evaluated and is not supported: whether its stop hook can actually
|
|
176
|
+
block a turn in the way this standard requires was unresolved as of this
|
|
177
|
+
writing, and shipping an adapter against an unverified contract would repeat
|
|
178
|
+
the exact failure R3 exists to prevent — an enforcement mechanism nobody has
|
|
179
|
+
confirmed enforces anything.
|
|
180
|
+
|
|
181
|
+
On any harness not in the table above, the check is inactive — the same
|
|
182
|
+
silent-by-default failure as an unsupported language (R8). `uds init
|
|
183
|
+
--with-hooks` reports which harnesses it wired a hook into; it does not
|
|
184
|
+
enumerate the rest here, because that list is a citation waiting to go stale
|
|
185
|
+
the next time a harness is added or dropped.
|
|
186
|
+
|
|
187
|
+
---
|
|
188
|
+
|
|
149
189
|
## What the detector matches
|
|
150
190
|
|
|
151
191
|
The shape is: **a first-person future marker, then an action verb, in the same
|
|
@@ -194,3 +234,5 @@ only because a corpus existed; the two that shipped were the ones no case covere
|
|
|
194
234
|
- [ ] The check recognises its own block message and does not read it as the human's
|
|
195
235
|
- [ ] The check recognises the itemized blocker ending R2 defines, and does not block it
|
|
196
236
|
- [ ] The attribution search excludes the check's own headings and scaffolding
|
|
237
|
+
- [ ] Each supported harness's block contract (config file, event, output shape) is verified against that harness's own docs, not assumed from another harness
|
|
238
|
+
- [ ] The installer only writes a harness's hook config for a harness the adopter selected
|
|
@@ -0,0 +1,143 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* UDS Hook: Turn Completion Integrity — Codex adapter
|
|
4
|
+
*
|
|
5
|
+
* Runs at turn end. Blocks when the agent's final message states a first-person
|
|
6
|
+
* commitment to a next action that the turn then ended without taking.
|
|
7
|
+
*
|
|
8
|
+
* Contract (Codex Stop hook — https://developers.openai.com/codex/hooks,
|
|
9
|
+
* fetched 2026-09-25, cross-checked against the same content served at
|
|
10
|
+
* https://learn.chatgpt.com/docs/hooks):
|
|
11
|
+
* config — <repo>/.codex/hooks.json (or ~/.codex/hooks.json), a dedicated
|
|
12
|
+
* file, NOT config.toml's [hooks] table. Codex runs matching
|
|
13
|
+
* hooks from every location that defines them, so this adapter
|
|
14
|
+
* deliberately touches only hooks.json.
|
|
15
|
+
* stdin — JSON with session_id, transcript_path (string|null), cwd,
|
|
16
|
+
* hook_event_name, turn_id, stop_hook_active, permission_mode,
|
|
17
|
+
* and last_assistant_message (string|null) — the agent's final
|
|
18
|
+
* text for this turn, given directly, no transcript parse needed.
|
|
19
|
+
* output — "Stop expects JSON on stdout when it exits 0. Plain text output
|
|
20
|
+
* is invalid for this event." Block: {"decision":"block","reason":
|
|
21
|
+
* "..."}. Allow: any other JSON (this adapter always writes "{}"
|
|
22
|
+
* so stdout is valid JSON on every path, including R5 failures).
|
|
23
|
+
*
|
|
24
|
+
* R9 (exempt a human-directed stop) is best-effort here. Codex's Stop payload
|
|
25
|
+
* gives the assistant's last message directly but not the human's; this
|
|
26
|
+
* adapter reads transcript_path for it (rollout.jsonl), tolerantly. The real
|
|
27
|
+
* record shape has been confirmed against a codex-cli 0.156.1 install
|
|
28
|
+
* (2026-09-26; see extractMessage() below) — 6.13.0-beta.1 tried two guessed
|
|
29
|
+
* shapes that never matched it, so R9 never actually exempted a Codex turn.
|
|
30
|
+
* If the read still fails, or an unrecognized record type is seen, the user
|
|
31
|
+
* side is treated as empty — which means R9 cannot exempt that turn, not
|
|
32
|
+
* that the check goes silent (last_assistant_message still drives
|
|
33
|
+
* detection). See core/turn-completion-integrity.md's Codex section.
|
|
34
|
+
*
|
|
35
|
+
* Judgement (packs, cooldown, rolling window, self-echo) lives in
|
|
36
|
+
* turn-completion/engine.mjs and is shared with every other adapter; this
|
|
37
|
+
* file only reads Codex's stdin shape and writes Codex's output shape.
|
|
38
|
+
*
|
|
39
|
+
* Usage: node check-turn-completion-codex.mjs (reads stdin)
|
|
40
|
+
* node check-turn-completion-codex.mjs --self-test
|
|
41
|
+
* node check-turn-completion-codex.mjs --languages
|
|
42
|
+
*
|
|
43
|
+
* @see core/turn-completion-integrity.md
|
|
44
|
+
*/
|
|
45
|
+
import { readFileSync } from 'node:fs';
|
|
46
|
+
import { decide, isSelfEcho, runSelfTest, printLanguages } from './turn-completion/engine.mjs';
|
|
47
|
+
|
|
48
|
+
/**
|
|
49
|
+
* A user-message record's `{ role, content }` payload, whichever of the
|
|
50
|
+
* plausible JSONL shapes `ev` turns out to be.
|
|
51
|
+
*
|
|
52
|
+
* 🔴 2026-09-26: verified against a real codex-cli 0.156.1
|
|
53
|
+
* `~/.codex/sessions/**\/*.jsonl` transcript. The actual shape is
|
|
54
|
+
* `{"type":"response_item","payload":{"type":"message","role":"user",
|
|
55
|
+
* "content":[{"type":"input_text","text":"…"}]}}` — the message lives under
|
|
56
|
+
* `ev.payload`, not `ev.message` and not `ev` itself. The two fallback shapes
|
|
57
|
+
* below were this file's original best-effort guesses (Claude-Code-style
|
|
58
|
+
* nested `message`, and a flat record); neither matches what Codex actually
|
|
59
|
+
* writes, so `bestEffortLastUserMessage` always returned '' and R9 (exempting
|
|
60
|
+
* a human-directed stop) never fired on Codex. They stay as fallbacks in case
|
|
61
|
+
* a different Codex version or record type uses one of them, but `ev.payload`
|
|
62
|
+
* is tried first since it is the confirmed real shape.
|
|
63
|
+
*
|
|
64
|
+
* Other record types seen in a real rollout (`event_msg`, `token_usage_record`,
|
|
65
|
+
* `session_meta`, `world_state`, and `response_item` payloads of type
|
|
66
|
+
* `reasoning`/`custom_tool_call`/…) do not have `role: "user"` on the object
|
|
67
|
+
* this function returns, so they fall through untouched. A `role: "developer"`
|
|
68
|
+
* message is deliberately NOT treated as a user message — only
|
|
69
|
+
* `payload.type === "message" && payload.role === "user"` counts.
|
|
70
|
+
*/
|
|
71
|
+
function extractMessage(ev) {
|
|
72
|
+
if (ev && ev.payload && ev.payload.type === 'message') return ev.payload;
|
|
73
|
+
if (ev && ev.message && typeof ev.message === 'object') return ev.message;
|
|
74
|
+
return ev;
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
/**
|
|
78
|
+
* Best-effort extraction of the human's last message from a Codex transcript.
|
|
79
|
+
* Tolerant of several plausible JSONL shapes; any failure returns '', which
|
|
80
|
+
* is the same as "cannot tell" (R5) — it does not stop last_assistant_message
|
|
81
|
+
* from still being checked.
|
|
82
|
+
*/
|
|
83
|
+
function bestEffortLastUserMessage(transcriptPath) {
|
|
84
|
+
if (typeof transcriptPath !== 'string' || !transcriptPath) return '';
|
|
85
|
+
try {
|
|
86
|
+
let user = '';
|
|
87
|
+
for (const line of readFileSync(transcriptPath, 'utf8').split('\n')) {
|
|
88
|
+
if (!line.trim()) continue;
|
|
89
|
+
let ev;
|
|
90
|
+
try { ev = JSON.parse(line); } catch { continue; }
|
|
91
|
+
const msg = extractMessage(ev);
|
|
92
|
+
if (!msg || msg.role !== 'user') continue;
|
|
93
|
+
const c = msg.content;
|
|
94
|
+
const text = typeof c === 'string'
|
|
95
|
+
? c
|
|
96
|
+
: Array.isArray(c)
|
|
97
|
+
? c.filter((p) => p && (p.type === 'text' || p.type === 'input_text' || typeof p.text === 'string')).map((p) => p.text || '').join('\n')
|
|
98
|
+
: '';
|
|
99
|
+
if (text && text.trim() && !isSelfEcho(text)) user = text;
|
|
100
|
+
}
|
|
101
|
+
return user;
|
|
102
|
+
} catch {
|
|
103
|
+
return '';
|
|
104
|
+
}
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
async function main() {
|
|
108
|
+
let out = {};
|
|
109
|
+
try {
|
|
110
|
+
const raw = readFileSync(0, 'utf8');
|
|
111
|
+
const data = JSON.parse(raw);
|
|
112
|
+
if (data && typeof data === 'object' && data.stop_hook_active !== true) {
|
|
113
|
+
const assistantText = typeof data.last_assistant_message === 'string' ? data.last_assistant_message : '';
|
|
114
|
+
const userText = bestEffortLastUserMessage(data.transcript_path);
|
|
115
|
+
const verdict = await decide({
|
|
116
|
+
sessionId: data.session_id,
|
|
117
|
+
assistantText,
|
|
118
|
+
userText,
|
|
119
|
+
});
|
|
120
|
+
if (verdict.fire) out = { decision: 'block', reason: verdict.reason };
|
|
121
|
+
}
|
|
122
|
+
} catch {
|
|
123
|
+
/* R5: fail open — stdout stays valid JSON, turn ends */
|
|
124
|
+
}
|
|
125
|
+
process.stdout.write(JSON.stringify(out));
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
const arg = process.argv[2];
|
|
129
|
+
if (arg === '--self-test') {
|
|
130
|
+
process.exit((await runSelfTest('turn-completion-codex')) ? 0 : 1);
|
|
131
|
+
} else if (arg === '--languages') {
|
|
132
|
+
await printLanguages();
|
|
133
|
+
} else {
|
|
134
|
+
try {
|
|
135
|
+
await main();
|
|
136
|
+
} catch {
|
|
137
|
+
// R5, belt and braces: main() already guards its own body, but a Codex
|
|
138
|
+
// Stop hook must exit 0 with valid JSON on stdout no matter what, and an
|
|
139
|
+
// uncaught throw here would instead crash with a non-zero exit and a
|
|
140
|
+
// stack trace on stderr.
|
|
141
|
+
process.stdout.write('{}');
|
|
142
|
+
}
|
|
143
|
+
}
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* UDS Hook: Turn Completion Integrity — Gemini CLI adapter
|
|
4
|
+
*
|
|
5
|
+
* Runs once per turn, after the model's final response. Blocks when that
|
|
6
|
+
* response states a first-person commitment to a next action that the turn
|
|
7
|
+
* then ended without taking.
|
|
8
|
+
*
|
|
9
|
+
* Contract (Gemini CLI AfterAgent hook — https://geminicli.com/docs/hooks/reference/
|
|
10
|
+
* and https://geminicli.com/docs/hooks/, fetched 2026-09-25):
|
|
11
|
+
* config — .gemini/settings.json (project) / ~/.gemini/settings.json
|
|
12
|
+
* (user) / /etc/gemini-cli/settings.json (system), under a
|
|
13
|
+
* "hooks" key shared with the rest of Gemini CLI's settings.
|
|
14
|
+
* Project settings take precedence over user, which take
|
|
15
|
+
* precedence over system. AfterAgent does not use matchers
|
|
16
|
+
* (matchers apply only to Tool hooks).
|
|
17
|
+
* stdin — JSON with session_id, transcript_path, cwd, hook_event_name,
|
|
18
|
+
* timestamp, prompt, prompt_response, stop_hook_active.
|
|
19
|
+
* `prompt` is the human message that started THIS turn and
|
|
20
|
+
* `prompt_response` is the agent's final text for it — both given
|
|
21
|
+
* directly, so R9 (exempt a human-directed stop) needs no
|
|
22
|
+
* transcript parsing here, unlike the Codex adapter.
|
|
23
|
+
* output — exit codes are shared across every Gemini CLI hook type: "0:
|
|
24
|
+
* stdout is parsed as JSON. Preferred for all logic. 2: System
|
|
25
|
+
* Block, stderr is the rejection reason." AfterAgent's own JSON
|
|
26
|
+
* shape: {"decision":"deny","reason":"..."} rejects the response
|
|
27
|
+
* and forces a retry, with reason "sent to the agent as a new
|
|
28
|
+
* prompt". This adapter uses the preferred exit-0 + JSON path,
|
|
29
|
+
* not exit code 2, and always writes valid JSON on stdout
|
|
30
|
+
* (including on every R5 failure path) so exit 0 is never
|
|
31
|
+
* ambiguous with "nothing to say".
|
|
32
|
+
*
|
|
33
|
+
* Judgement (packs, cooldown, rolling window, self-echo) lives in
|
|
34
|
+
* turn-completion/engine.mjs and is shared with every other adapter; this
|
|
35
|
+
* file only reads Gemini CLI's stdin shape and writes Gemini CLI's output
|
|
36
|
+
* shape.
|
|
37
|
+
*
|
|
38
|
+
* Usage: node check-turn-completion-gemini.mjs (reads stdin)
|
|
39
|
+
* node check-turn-completion-gemini.mjs --self-test
|
|
40
|
+
* node check-turn-completion-gemini.mjs --languages
|
|
41
|
+
*
|
|
42
|
+
* @see core/turn-completion-integrity.md
|
|
43
|
+
*/
|
|
44
|
+
import { readFileSync } from 'node:fs';
|
|
45
|
+
import { decide, runSelfTest, printLanguages } from './turn-completion/engine.mjs';
|
|
46
|
+
|
|
47
|
+
async function main() {
|
|
48
|
+
let out = {};
|
|
49
|
+
try {
|
|
50
|
+
const raw = readFileSync(0, 'utf8');
|
|
51
|
+
const data = JSON.parse(raw);
|
|
52
|
+
if (data && typeof data === 'object' && data.stop_hook_active !== true) {
|
|
53
|
+
const assistantText = typeof data.prompt_response === 'string' ? data.prompt_response : '';
|
|
54
|
+
const userText = typeof data.prompt === 'string' ? data.prompt : '';
|
|
55
|
+
const verdict = await decide({
|
|
56
|
+
sessionId: data.session_id,
|
|
57
|
+
assistantText,
|
|
58
|
+
userText,
|
|
59
|
+
});
|
|
60
|
+
if (verdict.fire) out = { decision: 'deny', reason: verdict.reason };
|
|
61
|
+
}
|
|
62
|
+
} catch {
|
|
63
|
+
/* R5: fail open — stdout stays valid JSON, turn ends */
|
|
64
|
+
}
|
|
65
|
+
process.stdout.write(JSON.stringify(out));
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
const arg = process.argv[2];
|
|
69
|
+
if (arg === '--self-test') {
|
|
70
|
+
process.exit((await runSelfTest('turn-completion-gemini')) ? 0 : 1);
|
|
71
|
+
} else if (arg === '--languages') {
|
|
72
|
+
await printLanguages();
|
|
73
|
+
} else {
|
|
74
|
+
try {
|
|
75
|
+
await main();
|
|
76
|
+
} catch {
|
|
77
|
+
// R5, belt and braces — see check-turn-completion-codex.mjs for why this
|
|
78
|
+
// outer guard exists alongside main()'s own.
|
|
79
|
+
process.stdout.write('{}');
|
|
80
|
+
}
|
|
81
|
+
}
|