universal-dev-standards 6.8.0 → 6.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/uds.js +12 -2
- package/bundled/ai/standards/acceptance-criteria-traceability.ai.yaml +14 -2
- package/bundled/ai/standards/adr-standards.ai.yaml +14 -2
- package/bundled/ai/standards/code-review.ai.yaml +13 -3
- package/bundled/ai/standards/commit-message.ai.yaml +8 -4
- package/bundled/ai/standards/deferred-item-exit.ai.yaml +225 -0
- package/bundled/ai/standards/feature-discovery-standards.ai.yaml +14 -2
- package/bundled/ai/standards/governance-layer.ai.yaml +128 -2
- package/bundled/ai/standards/logging.ai.yaml +2 -2
- package/bundled/ai/standards/retrospective-standards.ai.yaml +14 -2
- package/bundled/ai/standards/reverse-engineering-standards.ai.yaml +73 -2
- package/bundled/ai/standards/security-standards.ai.yaml +2 -2
- package/bundled/ai/standards/spec-driven-development.ai.yaml +14 -2
- package/bundled/ai/standards/tech-debt-standards.ai.yaml +87 -3
- package/bundled/ai/standards/turn-completion-integrity.ai.yaml +131 -0
- package/bundled/core/acceptance-criteria-traceability.md +5 -2
- package/bundled/core/adr-standards.md +26 -2
- package/bundled/core/code-review-checklist.md +5 -2
- package/bundled/core/context-aware-loading.md +1 -1
- package/bundled/core/deferred-item-exit.md +254 -0
- package/bundled/core/feature-discovery-standards.md +5 -1
- package/bundled/core/governance-layer.md +114 -2
- package/bundled/core/retrospective-standards.md +4 -2
- package/bundled/core/reverse-engineering-standards.md +81 -2
- package/bundled/core/spec-driven-development.md +8 -2
- package/bundled/core/tech-debt-standards.md +67 -8
- package/bundled/core/turn-completion-integrity.md +196 -0
- package/bundled/hooks/check-dangerous-cmd.mjs +60 -0
- package/bundled/hooks/check-logging-standard.mjs +59 -0
- package/bundled/hooks/check-turn-completion.mjs +233 -0
- package/bundled/hooks/inject-standards.mjs +183 -0
- package/bundled/hooks/telemetry-wrapper.mjs +77 -0
- package/bundled/hooks/turn-completion/detect.mjs +99 -0
- package/bundled/hooks/turn-completion/locales/en.mjs +159 -0
- package/bundled/hooks/turn-completion/locales/zh-TW.mjs +166 -0
- package/bundled/hooks/validate-commit-msg.mjs +104 -0
- package/bundled/locales/zh-CN/CHANGELOG.md +47 -3
- package/bundled/locales/zh-CN/CLAUDE.md +1 -1
- package/bundled/locales/zh-CN/README.md +2 -2
- package/bundled/locales/zh-CN/SECURITY.md +1 -1
- package/bundled/locales/zh-CN/core/adr-standards.md +1 -1
- package/bundled/locales/zh-CN/core/governance-layer.md +118 -6
- package/bundled/locales/zh-CN/core/retrospective-standards.md +1 -1
- package/bundled/locales/zh-CN/core/tech-debt-standards.md +71 -4
- package/bundled/locales/zh-CN/core/turn-completion-integrity.md +190 -0
- package/bundled/locales/zh-CN/docs/CHEATSHEET.md +8 -1
- package/bundled/locales/zh-CN/docs/CLI-INIT-OPTIONS.md +29 -68
- package/bundled/locales/zh-CN/docs/FEATURE-REFERENCE.md +25 -15
- package/bundled/locales/zh-CN/docs/USAGE-MODES-COMPARISON.md +1 -2
- package/bundled/locales/zh-CN/integrations/google-antigravity/{INSTRUCTIONS.md → AGENTS.md} +1 -1
- package/bundled/locales/zh-CN/integrations/google-antigravity/README.md +3 -3
- package/bundled/locales/zh-CN/skills/atdd-assistant/SKILL.md +2 -0
- package/bundled/locales/zh-CN/skills/bdd-assistant/SKILL.md +2 -0
- package/bundled/locales/zh-CN/skills/brainstorm-assistant/SKILL.md +22 -12
- package/bundled/locales/zh-CN/skills/brainstorm-assistant/guide.md +12 -9
- package/bundled/locales/zh-CN/skills/code-review-assistant/SKILL.md +1 -0
- package/bundled/locales/zh-CN/skills/commands/brainstorm.md +17 -13
- package/bundled/locales/zh-CN/skills/commands/config.md +0 -1
- package/bundled/locales/zh-CN/skills/commands/init.md +1 -2
- package/bundled/locales/zh-CN/skills/commit-standards/SKILL.md +2 -0
- package/bundled/locales/zh-CN/skills/contract-test-assistant/SKILL.md +1 -0
- package/bundled/locales/zh-CN/skills/dev-methodology/SKILL.md +2 -0
- package/bundled/locales/zh-CN/skills/observability-assistant/SKILL.md +1 -0
- package/bundled/locales/zh-CN/skills/project-structure-guide/SKILL.md +1 -0
- package/bundled/locales/zh-CN/skills/release-standards/SKILL.md +3 -0
- package/bundled/locales/zh-CN/skills/requirement-assistant/SKILL.md +2 -0
- package/bundled/locales/zh-CN/skills/reverse-engineer/SKILL.md +3 -0
- package/bundled/locales/zh-CN/skills/runbook-assistant/SKILL.md +1 -0
- package/bundled/locales/zh-CN/skills/slo-assistant/SKILL.md +1 -0
- package/bundled/locales/zh-CN/skills/tdd-assistant/SKILL.md +2 -0
- package/bundled/locales/zh-TW/CHANGELOG.md +47 -3
- package/bundled/locales/zh-TW/CLAUDE.md +1 -1
- package/bundled/locales/zh-TW/README.md +2 -2
- package/bundled/locales/zh-TW/SECURITY.md +1 -1
- package/bundled/locales/zh-TW/core/acceptance-criteria-traceability.md +2 -0
- package/bundled/locales/zh-TW/core/adr-standards.md +26 -5
- package/bundled/locales/zh-TW/core/code-review-checklist.md +2 -0
- package/bundled/locales/zh-TW/core/container-image-standards.md +2 -2
- package/bundled/locales/zh-TW/core/contract-testing-standards.md +2 -2
- package/bundled/locales/zh-TW/core/cross-flow-regression.md +8 -7
- package/bundled/locales/zh-TW/core/data-contract.md +2 -2
- package/bundled/locales/zh-TW/core/data-migration-testing.md +2 -2
- package/bundled/locales/zh-TW/core/data-pipeline.md +2 -2
- package/bundled/locales/zh-TW/core/deferred-item-exit.md +251 -0
- package/bundled/locales/zh-TW/core/documentation-writing-standards.md +228 -3
- package/bundled/locales/zh-TW/core/full-coverage-testing.md +15 -2
- package/bundled/locales/zh-TW/core/governance-layer.md +118 -5
- package/bundled/locales/zh-TW/core/iac-design-principles.md +2 -2
- package/bundled/locales/zh-TW/core/incident-response.md +2 -2
- package/bundled/locales/zh-TW/core/model-provenance.md +4 -2
- package/bundled/locales/zh-TW/core/pii-classification.md +42 -6
- package/bundled/locales/zh-TW/core/prd-standards.md +4 -2
- package/bundled/locales/zh-TW/core/product-metrics-standards.md +4 -2
- package/bundled/locales/zh-TW/core/release-readiness-gate.md +2 -2
- package/bundled/locales/zh-TW/core/resource-cost-boundary.md +2 -2
- package/bundled/locales/zh-TW/core/retrospective-standards.md +5 -3
- package/bundled/locales/zh-TW/core/reverse-engineering-standards.md +83 -5
- package/bundled/locales/zh-TW/core/runbook.md +2 -2
- package/bundled/locales/zh-TW/core/schema-evolution.md +2 -2
- package/bundled/locales/zh-TW/core/secret-management-standards.md +2 -2
- package/bundled/locales/zh-TW/core/slo-sli.md +2 -2
- package/bundled/locales/zh-TW/core/spec-driven-development.md +2 -0
- package/bundled/locales/zh-TW/core/tech-debt-standards.md +71 -4
- package/bundled/locales/zh-TW/core/turn-completion-integrity.md +190 -0
- package/bundled/locales/zh-TW/core/user-journey-testing.md +2 -2
- package/bundled/locales/zh-TW/core/user-story-mapping.md +2 -2
- package/bundled/locales/zh-TW/core/verification-oracle.md +2 -2
- package/bundled/locales/zh-TW/docs/CHEATSHEET.md +8 -1
- package/bundled/locales/zh-TW/docs/CLI-INIT-OPTIONS.md +29 -68
- package/bundled/locales/zh-TW/docs/FEATURE-REFERENCE.md +25 -15
- package/bundled/locales/zh-TW/docs/USAGE-MODES-COMPARISON.md +1 -2
- package/bundled/locales/zh-TW/integrations/google-antigravity/{INSTRUCTIONS.md → AGENTS.md} +1 -1
- package/bundled/locales/zh-TW/integrations/google-antigravity/README.md +3 -3
- package/bundled/locales/zh-TW/skills/adr-assistant/SKILL.md +1 -1
- package/bundled/locales/zh-TW/skills/atdd-assistant/SKILL.md +2 -0
- package/bundled/locales/zh-TW/skills/bdd-assistant/SKILL.md +2 -0
- package/bundled/locales/zh-TW/skills/brainstorm-assistant/SKILL.md +22 -12
- package/bundled/locales/zh-TW/skills/brainstorm-assistant/guide.md +12 -9
- package/bundled/locales/zh-TW/skills/code-review-assistant/SKILL.md +1 -0
- package/bundled/locales/zh-TW/skills/commands/brainstorm.md +17 -13
- package/bundled/locales/zh-TW/skills/commands/config.md +0 -1
- package/bundled/locales/zh-TW/skills/commands/init.md +1 -2
- package/bundled/locales/zh-TW/skills/commit-standards/SKILL.md +2 -0
- package/bundled/locales/zh-TW/skills/contract-test-assistant/SKILL.md +2 -1
- package/bundled/locales/zh-TW/skills/dev-methodology/SKILL.md +2 -0
- package/bundled/locales/zh-TW/skills/dev-workflow-guide/SKILL.md +1 -1
- package/bundled/locales/zh-TW/skills/knowledge-graph/guide.md +2 -2
- package/bundled/locales/zh-TW/skills/migration-assistant/SKILL.md +1 -1
- package/bundled/locales/zh-TW/skills/observability-assistant/SKILL.md +1 -0
- package/bundled/locales/zh-TW/skills/project-discovery/SKILL.md +1 -0
- package/bundled/locales/zh-TW/skills/project-structure-guide/SKILL.md +1 -0
- package/bundled/locales/zh-TW/skills/release-standards/SKILL.md +3 -0
- package/bundled/locales/zh-TW/skills/requirement-assistant/SKILL.md +2 -0
- package/bundled/locales/zh-TW/skills/reverse-engineer/SKILL.md +3 -0
- package/bundled/locales/zh-TW/skills/runbook-assistant/SKILL.md +1 -0
- package/bundled/locales/zh-TW/skills/slo-assistant/SKILL.md +1 -0
- package/bundled/locales/zh-TW/skills/tdd-assistant/SKILL.md +2 -0
- package/bundled/skills/atdd-assistant/SKILL.md +2 -0
- package/bundled/skills/bdd-assistant/SKILL.md +2 -0
- package/bundled/skills/brainstorm-assistant/SKILL.md +31 -13
- package/bundled/skills/brainstorm-assistant/guide.md +9 -6
- package/bundled/skills/code-review-assistant/SKILL.md +1 -0
- package/bundled/skills/commands/brainstorm.md +12 -9
- package/bundled/skills/commands/config.md +0 -1
- package/bundled/skills/commands/init.md +2 -3
- package/bundled/skills/commit-standards/SKILL.md +2 -0
- package/bundled/skills/contract-test-assistant/SKILL.md +1 -0
- package/bundled/skills/dev-methodology/SKILL.md +4 -0
- package/bundled/skills/observability-assistant/SKILL.md +1 -0
- package/bundled/skills/project-discovery/SKILL.md +1 -0
- package/bundled/skills/project-structure-guide/SKILL.md +1 -0
- package/bundled/skills/release-standards/SKILL.md +3 -0
- package/bundled/skills/requirement-assistant/SKILL.md +2 -0
- package/bundled/skills/reverse-engineer/SKILL.md +3 -0
- package/bundled/skills/runbook-assistant/SKILL.md +1 -0
- package/bundled/skills/slo-assistant/SKILL.md +1 -0
- package/bundled/skills/tdd-assistant/SKILL.md +2 -0
- package/bundled/templates/.ai-context.yaml.template +194 -0
- package/bundled/templates/CLAUDE.md.template +145 -0
- package/bundled/templates/DESIGN.md +237 -0
- package/bundled/templates/SKILL-BRIEF-TEMPLATE.md +57 -0
- package/bundled/templates/SKILL-CANDIDATES.md +39 -0
- package/bundled/templates/gates/check-error-exit.mjs +309 -0
- package/bundled/templates/mcp-config.json +10 -0
- package/bundled/templates/methodology-template.yaml +209 -0
- package/bundled/templates/migration-template.md +408 -0
- package/bundled/templates/requirement-checklist.md +410 -0
- package/bundled/templates/requirement-document-template.md +591 -0
- package/bundled/templates/requirement-template.md +881 -0
- package/bundled/templates/reverse-spec-template.md +409 -0
- package/bundled/templates/test-case-template.md +74 -0
- package/bundled/templates/test-plan-template.md +74 -0
- package/package.json +7 -5
- package/src/commands/audit.js +82 -0
- package/src/commands/check.js +66 -10
- package/src/commands/init.js +161 -16
- package/src/commands/update.js +286 -14
- package/src/compilers/claude-code-compiler.js +4 -1
- package/src/config/ai-agent-paths.js +62 -17
- package/src/core/constants.js +42 -11
- package/src/core/manifest.js +201 -3
- package/src/core/paths.js +2 -2
- package/src/i18n/messages.js +6 -29
- package/src/installers/hooks-installer.js +167 -75
- package/src/installers/integration-installer.js +9 -5
- package/src/prompts/init.js +14 -14
- package/src/utils/detector.js +21 -1
- package/src/utils/effect-boundary.js +1093 -0
- package/src/utils/hasher.js +166 -1
- package/src/utils/hook-stats.js +1 -1
- package/src/utils/integration-generator.js +79 -1
- package/src/utils/reference-sync.js +4 -1
- package/src/utils/yaml-generator.js +51 -9
- package/standards-registry.json +31 -8
|
@@ -2,8 +2,8 @@
|
|
|
2
2
|
|
|
3
3
|
> **Language**: English | [繁體中文](../locales/zh-TW/core/tech-debt-standards.md)
|
|
4
4
|
|
|
5
|
-
**Version**: 1.
|
|
6
|
-
**Last Updated**: 2026-
|
|
5
|
+
**Version**: 1.1.0
|
|
6
|
+
**Last Updated**: 2026-08-20
|
|
7
7
|
**Applicability**: All software projects
|
|
8
8
|
**Scope**: universal
|
|
9
9
|
**Industry Standards**: Martin Fowler's Technical Debt Quadrant, Ward Cunningham's Debt Metaphor
|
|
@@ -65,13 +65,70 @@ Every technical debt item must be recorded in a registry with the following 11 f
|
|
|
65
65
|
|
|
66
66
|
### Registry Storage Options
|
|
67
67
|
|
|
68
|
-
The registry can be stored in
|
|
68
|
+
The registry can be stored in any of the following locations. Each option is legal **only if** it also satisfies the condition in the right-hand column — see [Overdue Handling](#overdue-handling) below.
|
|
69
69
|
|
|
70
|
-
| Option | Best For | Format |
|
|
71
|
-
|
|
72
|
-
| `docs/tech-debt-registry.md` | Small teams, simple tracking | Markdown table |
|
|
73
|
-
| Issue tracker (GitHub Issues, Jira) | Larger teams, workflow integration | Tagged issues with `tech-debt` label |
|
|
74
|
-
| Dedicated spreadsheet | Non-technical stakeholders | CSV/Excel with the 11 fields |
|
|
70
|
+
| Option | Best For | Format | Overdue-check condition |
|
|
71
|
+
|--------|----------|--------|-------------------------|
|
|
72
|
+
| `docs/tech-debt-registry.md` | Small teams, simple tracking | Markdown table | A repository check parses the table and fails on rows past their Target Resolution Date — **or** the file carries an Unattended Declaration |
|
|
73
|
+
| Issue tracker (GitHub Issues, Jira) | Larger teams, workflow integration | Tagged issues with `tech-debt` label | A scheduled query on the due-date field that actually **runs** and raises; a saved filter nobody opens is not a check — **or** an Unattended Declaration |
|
|
74
|
+
| Dedicated spreadsheet | Non-technical stakeholders | CSV/Excel with the 11 fields | Only if the sheet is exported, on a stated cadence, to a location a check reads — **or** an Unattended Declaration |
|
|
75
|
+
|
|
76
|
+
> **Why that column had to be added.** A registry no program can read satisfies every other requirement in this section: all 11 fields present, an Owner named, a Target Resolution Date set. Nothing about it is wrong — until the date passes, and then nothing happens. The spreadsheet option is deliberately **not** removed; a spreadsheet is often the only format the people who fund the work will open. What is no longer legal is for any storage option to be a place where dates go to expire unobserved.
|
|
77
|
+
|
|
78
|
+
### Overdue Handling
|
|
79
|
+
|
|
80
|
+
A **Target Resolution Date** that passes with no consequence is indistinguishable from having no date at all. This subsection defines what "the date arrived" means.
|
|
81
|
+
|
|
82
|
+
#### The three legal dispositions
|
|
83
|
+
|
|
84
|
+
When an item reaches its Target Resolution Date, exactly one of three things must happen, and it must leave a trace:
|
|
85
|
+
|
|
86
|
+
| Disposition | Meaning | Required trace |
|
|
87
|
+
|-------------|---------|----------------|
|
|
88
|
+
| **Resolve** | Do the work | Registry entry closed + a `Tech-Debt: TD-NNN resolved` commit footer (see [Commit Marking](#6-commit-marking)) |
|
|
89
|
+
| **Withdraw** | Decide not to do it, and delete the record | Entry marked withdrawn, with a reason and a date. It stops being counted as debt because it no longer is any |
|
|
90
|
+
| **Extend** | Set a new Target Resolution Date | The new date **and** a written reason, recorded together; earlier dates stay visible (append, do not overwrite) |
|
|
91
|
+
|
|
92
|
+
There is no fourth disposition. "Still open, date passed, nobody looked" is not a state this standard permits — it is the failure this subsection exists to name.
|
|
93
|
+
|
|
94
|
+
**Rule TD-EXP-001 (Required)** — An item past its Target Resolution Date with none of the three dispositions applied makes the registry non-compliant. Overdue is a finding, not a neutral background condition.
|
|
95
|
+
|
|
96
|
+
**Rule TD-EXP-002 (Required, prohibition)** — Expiry MUST NOT be implemented as **automatic extension** or **automatic closure**.
|
|
97
|
+
|
|
98
|
+
- Automatic extension makes the date unfalsifiable: it can never be missed, so it never measures anything.
|
|
99
|
+
- Automatic closure deletes the record without anyone having decided to.
|
|
100
|
+
|
|
101
|
+
Both stop the clock, and neither leaves a trace that it did. A date whose only consumer is a job that pushes it forward is decoration.
|
|
102
|
+
|
|
103
|
+
**Rule TD-EXP-003 (Required)** — Extending requires the reason to be recorded next to the new date. Changing only the date is not an extension; it is an erasure with a timestamp on it.
|
|
104
|
+
|
|
105
|
+
**Rule TD-EXP-004 (Recommended)** — An item extended three or more times with no progress recorded between extensions should be re-triaged as a **Withdraw** candidate. Repeated extension is evidence that nobody intends to do the work; recording that honestly is more useful to the next reader than a fourth date.
|
|
106
|
+
|
|
107
|
+
#### Unattended Declaration
|
|
108
|
+
|
|
109
|
+
Not every team can run a check. **A registry with no automated overdue check is still compliant — but only if it says so.**
|
|
110
|
+
|
|
111
|
+
An **Unattended Declaration** is a visible, dated statement inside the registry itself:
|
|
112
|
+
|
|
113
|
+
```
|
|
114
|
+
Unattended — no automated check reads this registry for overdue items.
|
|
115
|
+
Reviewed manually by <owner> on a <cadence> cadence. Last reviewed: <date>.
|
|
116
|
+
```
|
|
117
|
+
|
|
118
|
+
**Rule TD-EXP-005 (Required)** — A registry MUST be in exactly one of two states, and MUST make which one visible to anyone reading the registry:
|
|
119
|
+
|
|
120
|
+
1. **Checked** — a named, runnable check reads the registry and fails when an item is past its Target Resolution Date.
|
|
121
|
+
2. **Unattended** — no such check exists, an Unattended Declaration is present, and it names an owner, a review cadence, and the date of the last review.
|
|
122
|
+
|
|
123
|
+
A registry in neither state is non-compliant. The declaration is not paperwork: it is the whole difference between a reader who knows nothing is watching and a reader who assumes something is.
|
|
124
|
+
|
|
125
|
+
**Rule TD-EXP-006 (Required)** — The "Last reviewed" date in an Unattended Declaration is itself subject to the cadence that declaration states. A declaration whose last review is older than its own cadence is an overdue item in its own right, and the three dispositions apply to it.
|
|
126
|
+
|
|
127
|
+
> **This standard ships no checker, and that is a known cost — stated here rather than left to be discovered.**
|
|
128
|
+
>
|
|
129
|
+
> UDS defines the requirement; it does not provide a program that enforces it. Adopters who want the **Checked** state must write the check themselves, and most teams will not — which is exactly the failure mode described above, now applying to this section as well.
|
|
130
|
+
>
|
|
131
|
+
> That is why the minimum bar here is **not** "build a checker". It is **"never let a registry be silently unattended."** Declaring `Unattended` costs one paragraph and is fully compliant. Being unattended *without* declaring it is the one outcome this standard actually forbids, because it is the only one that misleads the reader about whether anything is watching.
|
|
75
132
|
|
|
76
133
|
---
|
|
77
134
|
|
|
@@ -200,6 +257,7 @@ Tech-Debt: TD-042 resolved
|
|
|
200
257
|
- [Refactoring Standards](refactoring-standards.md) — Techniques for resolving code debt
|
|
201
258
|
- [Testing Standards](testing-standards.md) — Addressing test debt
|
|
202
259
|
- [Commit Message Guide](commit-message-guide.md) — Commit format including debt markers
|
|
260
|
+
- [Governance Layer](governance-layer.md) — Standard #0; applies the same clock to pending decisions and accepted risks, and defines aggregate-reporting and freshness-metric rules
|
|
203
261
|
|
|
204
262
|
---
|
|
205
263
|
|
|
@@ -207,6 +265,7 @@ Tech-Debt: TD-042 resolved
|
|
|
207
265
|
|
|
208
266
|
| Version | Date | Changes |
|
|
209
267
|
|---------|------|---------|
|
|
268
|
+
| 1.1.0 | 2026-08-20 | Added: Overdue Handling — three legal dispositions on expiry (resolve / withdraw / extend-with-reason), TD-EXP-001..006, prohibition on automatic extension and automatic closure, Unattended Declaration as the compliant alternative to a checker; Registry Storage Options gains an overdue-check condition per option (a registry no program can read is legal only when declared unattended) |
|
|
210
269
|
| 1.0.0 | 2026-03-31 | Initial release: 6 debt types, registry template, budget allocation, prioritization matrix, quantitative metrics, commit marking |
|
|
211
270
|
|
|
212
271
|
---
|
|
@@ -0,0 +1,196 @@
|
|
|
1
|
+
# Turn Completion Integrity
|
|
2
|
+
|
|
3
|
+
> **Language**: English | [繁體中文](../locales/zh-TW/core/turn-completion-integrity.md)
|
|
4
|
+
|
|
5
|
+
**Version**: 1.3.0
|
|
6
|
+
**Last Updated**: 2026-09-08
|
|
7
|
+
**Applicability**: Any harness where an agent ends a turn and hands control back to a human
|
|
8
|
+
**Scope**: universal
|
|
9
|
+
**Industry Standards**: none claimed — derived from observed failures, see Evidence
|
|
10
|
+
|
|
11
|
+
---
|
|
12
|
+
|
|
13
|
+
## Purpose
|
|
14
|
+
|
|
15
|
+
An agent writes *"I'll do X next"* and then ends the turn without doing X.
|
|
16
|
+
|
|
17
|
+
The human reads a sentence describing work that was never done. Nothing errors, no
|
|
18
|
+
check fails, and the transcript reads like progress. The next turn starts from a
|
|
19
|
+
state the human believes is further along than it is.
|
|
20
|
+
|
|
21
|
+
This is not a knowledge failure — the agent named the correct next step. It is a
|
|
22
|
+
**closure** failure: the turn ended at the point where the work should have started.
|
|
23
|
+
|
|
24
|
+
---
|
|
25
|
+
|
|
26
|
+
## The Rule
|
|
27
|
+
|
|
28
|
+
**R1.** If the agent's final message states a first-person commitment to a next
|
|
29
|
+
action, the turn MUST NOT end until that action is taken or the agent states what
|
|
30
|
+
blocks it.
|
|
31
|
+
|
|
32
|
+
**R2.** "What blocks it" means naming the person or input each remaining item waits
|
|
33
|
+
on. *"The main parts are done"* is not a blocker statement; *"the deploy waits on
|
|
34
|
+
your key, the doc waits on nothing and I am doing it now"* is.
|
|
35
|
+
|
|
36
|
+
**R3.** A rule written only in the agent's instructions is not enforcement. R1 MUST
|
|
37
|
+
be evaluated **at the moment the turn ends**, by something that is not the agent.
|
|
38
|
+
|
|
39
|
+
---
|
|
40
|
+
|
|
41
|
+
## Why R3 exists
|
|
42
|
+
|
|
43
|
+
The instruction form of this rule was written into a project's agent instructions and
|
|
44
|
+
violated inside the same session that read it. It was then written into a published
|
|
45
|
+
standard, marked optional, with no checker reading it; the same violation recurred
|
|
46
|
+
seven days later.
|
|
47
|
+
|
|
48
|
+
An instruction is an input. The failure happens at a decision point — *"I am about to
|
|
49
|
+
end this turn"* — and an input that is not consulted at that point is not in effect.
|
|
50
|
+
This is the general shape: **a documented requirement with no enforcement at its
|
|
51
|
+
decision point is a suggestion, and its compliance is unmeasured, not high.**
|
|
52
|
+
|
|
53
|
+
---
|
|
54
|
+
|
|
55
|
+
## What the check MUST NOT do
|
|
56
|
+
|
|
57
|
+
**R4. Do not block on work that is always outstanding.** A repository always has open
|
|
58
|
+
TODOs, a backlog always has items. A gate that is true on every turn is turned off,
|
|
59
|
+
and then it protects nothing. The check fires **only** on a commitment the agent made
|
|
60
|
+
in that message.
|
|
61
|
+
|
|
62
|
+
**R5. Fail open, always.** Any error — unreadable transcript, missing field, malformed
|
|
63
|
+
JSON, unexpected schema — MUST let the turn end. A hook that can trap a session is
|
|
64
|
+
worse than no hook, because the human's only recovery is to disable it, and they will
|
|
65
|
+
disable it permanently.
|
|
66
|
+
|
|
67
|
+
**R6. Bound the blocking.** The check MUST enforce
|
|
68
|
+
(a) a cooldown between blocks, and
|
|
69
|
+
(b) a cap on blocks within a **rolling time window**.
|
|
70
|
+
|
|
71
|
+
A cap counted per session with no reset is not a safety limit — it is an off switch on
|
|
72
|
+
a delay, and it disarms silently in exactly the long session where the rule matters
|
|
73
|
+
most.
|
|
74
|
+
|
|
75
|
+
---
|
|
76
|
+
|
|
77
|
+
## The human can end the turn, and the agent's words cannot say so
|
|
78
|
+
|
|
79
|
+
**R9.** The check MUST exempt a turn the human asked to end, and MUST determine
|
|
80
|
+
that from the human's own most recent message — not from the agent's.
|
|
81
|
+
|
|
82
|
+
This rule exists because its absence was measured, once, on the first real
|
|
83
|
+
firing after this standard shipped. The human wrote *"pause, I'm going home"*;
|
|
84
|
+
the agent acknowledged and listed what it would resume; the check read an
|
|
85
|
+
abandoned commitment and blocked.
|
|
86
|
+
|
|
87
|
+
The detector was not wrong about the pattern. **A turn ending by instruction and
|
|
88
|
+
a turn ending on an abandoned commitment produce the same words from the agent**,
|
|
89
|
+
because in both cases the agent names work it is not doing now. Nothing in the
|
|
90
|
+
final message separates them, so a check that reads only that message cannot.
|
|
91
|
+
|
|
92
|
+
**R10.** The stop-request pattern MUST be narrow, and its corpus MUST include
|
|
93
|
+
work instructions that merely contain a stopping word. A false exemption is not
|
|
94
|
+
one missed block: it silences the check for the rest of the session. During
|
|
95
|
+
implementation the pattern matched *"stop using the hardcoded list and walk the
|
|
96
|
+
registry instead"* — an instruction to do work, read as an instruction to stop.
|
|
97
|
+
|
|
98
|
+
**R11.** The check's own block message re-enters the transcript as a human turn.
|
|
99
|
+
It MUST be able to recognise its own output and skip it when looking for the
|
|
100
|
+
human's last message.
|
|
101
|
+
|
|
102
|
+
Without this, R9 works exactly once. The block message becomes "the human's most
|
|
103
|
+
recent message" on the next run, the real instruction to stop is hidden behind
|
|
104
|
+
it, and the exemption disappears — silently, and only in the situation it was
|
|
105
|
+
built for.
|
|
106
|
+
|
|
107
|
+
This is the same family as a detector matching the prose that documents it,
|
|
108
|
+
one step further along: **the check reads its own output.** Carry a fixed marker
|
|
109
|
+
string in the block message and skip any human turn containing it.
|
|
110
|
+
|
|
111
|
+
**R12.** The check MUST be able to recognise the ending R2 defines. A check that
|
|
112
|
+
enforces R2 but cannot see an itemized blocker list blocks the one ending the
|
|
113
|
+
standard asks for — it punishes correct behaviour, and that is a faster way to
|
|
114
|
+
get uninstalled than missing a violation.
|
|
115
|
+
|
|
116
|
+
Recognition needs three things, and dropping any one of them was measured to
|
|
117
|
+
break it:
|
|
118
|
+
|
|
119
|
+
| Requirement | Why |
|
|
120
|
+
|---|---|
|
|
121
|
+
| Two or more list items | One item with an attribution is a sentence, not an itemization |
|
|
122
|
+
| A blocker attribution anywhere in the message, not per item | Real writing puts it in the preamble — *"the rest are all on you, itemized:"* — and requiring it per item missed the first real message it was tested against |
|
|
123
|
+
| No vague-completion phrase | *"the rest are done"* is the summary R2 rejects; a list that disposes of the remainder that way is a summary wearing a list's clothes |
|
|
124
|
+
|
|
125
|
+
🔴 **Exclude the check's own scaffolding from the attribution search.** The
|
|
126
|
+
heading *"what you need to decide"* contains the word *"you"*, so a list of work
|
|
127
|
+
already finished satisfied the attribution test by way of the heading above it.
|
|
128
|
+
The detector read its own structure — the same failure as R11, one layer down.
|
|
129
|
+
|
|
130
|
+
---
|
|
131
|
+
|
|
132
|
+
## Language coverage is a correctness property, not a translation task
|
|
133
|
+
|
|
134
|
+
The check reads prose written by the agent, so **its detector is language-specific**.
|
|
135
|
+
A detector carrying only one language's patterns, shipped to an adopter working in
|
|
136
|
+
another, produces a hook that is installed, runs on every turn, exits 0, and can never
|
|
137
|
+
fire. Every observable signal is identical to "the agent is behaving well".
|
|
138
|
+
|
|
139
|
+
**R7.** Each supported language MUST ship as a locale pack carrying **its own corpus**
|
|
140
|
+
of cases that must block and cases that must pass, and the pack's self-test MUST run
|
|
141
|
+
in CI. A locale pack whose corpus does not pass is not shipped.
|
|
142
|
+
|
|
143
|
+
**R8.** An adopter whose language has no pack MUST be told the check is inactive for
|
|
144
|
+
their language. Silence here reproduces the exact failure the standard exists to
|
|
145
|
+
prevent, one level up.
|
|
146
|
+
|
|
147
|
+
---
|
|
148
|
+
|
|
149
|
+
## What the detector matches
|
|
150
|
+
|
|
151
|
+
The shape is: **a first-person future marker, then an action verb, in the same
|
|
152
|
+
sentence**, minus three exclusions.
|
|
153
|
+
|
|
154
|
+
| Element | Purpose | Failure if omitted |
|
|
155
|
+
|---|---|---|
|
|
156
|
+
| First-person + future marker | Distinguishes a commitment from a description | Matches the user's words quoted back |
|
|
157
|
+
| Action verb | Distinguishes doing from reporting | *"I'll explain why"* counts as work |
|
|
158
|
+
| Same-sentence scope | Keeps the two halves related | A negated clause and a later real commitment merge into one non-match |
|
|
159
|
+
| Exclude reporting verbs | *"I'll say / note / mention"* is not work | The check fires on its own summaries |
|
|
160
|
+
| Exclude negation | *"I won't change that"* is a decision, not a commitment | A refusal reads as a promise |
|
|
161
|
+
| Exclude quoted and tabular text | Examples of the pattern are not instances of it | The check fires on its own documentation |
|
|
162
|
+
|
|
163
|
+
Each exclusion above was added because its absence produced a false block in testing.
|
|
164
|
+
An implementation that drops one will reproduce that block.
|
|
165
|
+
|
|
166
|
+
---
|
|
167
|
+
|
|
168
|
+
## Evidence
|
|
169
|
+
|
|
170
|
+
Observed over one project, 2026-08 to 2026-09:
|
|
171
|
+
|
|
172
|
+
| Observation | Count |
|
|
173
|
+
|---|---|
|
|
174
|
+
| Same complaint raised by the human about stated-but-undone next steps | 5 |
|
|
175
|
+
| Times the instruction form was present and violated in the same session | ≥1 |
|
|
176
|
+
| Detector versions defeated by the pattern in their own source or examples | 5 |
|
|
177
|
+
| Detector false positives caught by a corpus before shipping | 3 |
|
|
178
|
+
|
|
179
|
+
The last row is the argument for R7. Three of the five detector defects were caught
|
|
180
|
+
only because a corpus existed; the two that shipped were the ones no case covered.
|
|
181
|
+
|
|
182
|
+
---
|
|
183
|
+
|
|
184
|
+
## Checklist
|
|
185
|
+
|
|
186
|
+
- [ ] The check runs at turn end, not as an instruction to the agent
|
|
187
|
+
- [ ] Every failure path exits without blocking
|
|
188
|
+
- [ ] A cooldown and a rolling-window cap are both present
|
|
189
|
+
- [ ] Each shipped language has a corpus, and the corpus runs in CI
|
|
190
|
+
- [ ] Adopters in unsupported languages are told the check is inactive
|
|
191
|
+
- [ ] The check does not consult repository state (open TODOs, backlog)
|
|
192
|
+
- [ ] A turn the human asked to end is exempt, decided from the human's message
|
|
193
|
+
- [ ] The stop-request corpus includes work instructions containing a stopping word
|
|
194
|
+
- [ ] The check recognises its own block message and does not read it as the human's
|
|
195
|
+
- [ ] The check recognises the itemized blocker ending R2 defines, and does not block it
|
|
196
|
+
- [ ] The attribution search excludes the check's own headings and scaffolding
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* UDS Hook: Dangerous Command Detection
|
|
4
|
+
*
|
|
5
|
+
* Detects potentially destructive shell commands and blocks execution.
|
|
6
|
+
* Exit code: 0 = safe, 1 = dangerous
|
|
7
|
+
*
|
|
8
|
+
* Usage: echo "rm -rf /" | node check-dangerous-cmd.mjs
|
|
9
|
+
* or: node check-dangerous-cmd.mjs "rm -rf /"
|
|
10
|
+
*
|
|
11
|
+
* Performance target: < 500ms
|
|
12
|
+
*
|
|
13
|
+
* @see docs/specs/SPEC-HOOKS-001-core-standard-hooks.md (REQ-2)
|
|
14
|
+
*/
|
|
15
|
+
|
|
16
|
+
const DANGEROUS_PATTERNS = [
|
|
17
|
+
/rm\s+-rf\s+\//,
|
|
18
|
+
/mkfs\./,
|
|
19
|
+
/dd\s+if=.*of=\/dev\//,
|
|
20
|
+
/format\s+[a-zA-Z]:/,
|
|
21
|
+
/>\s*\/dev\/sda/,
|
|
22
|
+
/chmod\s+-R\s+777\s+\//,
|
|
23
|
+
/:()\{\s*:\|:&\s*\};:/, // fork bomb
|
|
24
|
+
];
|
|
25
|
+
|
|
26
|
+
/**
|
|
27
|
+
* Check if a command matches dangerous patterns.
|
|
28
|
+
* @param {string} cmd - The shell command to check
|
|
29
|
+
* @returns {boolean} true if dangerous
|
|
30
|
+
*/
|
|
31
|
+
export function isDangerousCommand(cmd) {
|
|
32
|
+
if (!cmd || typeof cmd !== 'string') return false;
|
|
33
|
+
return DANGEROUS_PATTERNS.some((p) => p.test(cmd));
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
// CLI mode
|
|
37
|
+
if (process.argv[1] && process.argv[1].endsWith('check-dangerous-cmd.mjs')) {
|
|
38
|
+
const input = process.argv[2];
|
|
39
|
+
if (input) {
|
|
40
|
+
if (isDangerousCommand(input)) {
|
|
41
|
+
console.error(`⚠️ Dangerous command detected: "${input}"`);
|
|
42
|
+
process.exit(1);
|
|
43
|
+
} else {
|
|
44
|
+
process.exit(0);
|
|
45
|
+
}
|
|
46
|
+
} else {
|
|
47
|
+
let data = '';
|
|
48
|
+
process.stdin.setEncoding('utf-8');
|
|
49
|
+
process.stdin.on('data', (chunk) => { data += chunk; });
|
|
50
|
+
process.stdin.on('end', () => {
|
|
51
|
+
const cmd = data.trim();
|
|
52
|
+
if (isDangerousCommand(cmd)) {
|
|
53
|
+
console.error(`⚠️ Dangerous command detected: "${cmd}"`);
|
|
54
|
+
process.exit(1);
|
|
55
|
+
} else {
|
|
56
|
+
process.exit(0);
|
|
57
|
+
}
|
|
58
|
+
});
|
|
59
|
+
}
|
|
60
|
+
}
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* UDS Hook: Structured Logging Check
|
|
4
|
+
*
|
|
5
|
+
* Checks code for unstructured logging calls (console.log, console.warn, etc.)
|
|
6
|
+
* and suggests using structured logging instead.
|
|
7
|
+
* Exit code: 0 = pass, 1 = unstructured logging found
|
|
8
|
+
*
|
|
9
|
+
* Usage: echo 'console.log("debug")' | node check-logging-standard.mjs
|
|
10
|
+
* or: node check-logging-standard.mjs 'console.log("debug")'
|
|
11
|
+
*
|
|
12
|
+
* Performance target: < 500ms
|
|
13
|
+
*
|
|
14
|
+
* @see docs/specs/SPEC-HOOKS-001-core-standard-hooks.md (REQ-3)
|
|
15
|
+
*/
|
|
16
|
+
|
|
17
|
+
const UNSTRUCTURED_PATTERNS = [
|
|
18
|
+
/console\.log\(/,
|
|
19
|
+
/console\.warn\(/,
|
|
20
|
+
/console\.error\(/,
|
|
21
|
+
/console\.info\(/,
|
|
22
|
+
/console\.debug\(/,
|
|
23
|
+
];
|
|
24
|
+
|
|
25
|
+
/**
|
|
26
|
+
* Check if code contains unstructured logging calls.
|
|
27
|
+
* @param {string} code - The code to check
|
|
28
|
+
* @returns {boolean} true if unstructured logging found
|
|
29
|
+
*/
|
|
30
|
+
export function hasUnstructuredLogging(code) {
|
|
31
|
+
if (!code || typeof code !== 'string') return false;
|
|
32
|
+
return UNSTRUCTURED_PATTERNS.some((p) => p.test(code));
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
// CLI mode
|
|
36
|
+
if (process.argv[1] && process.argv[1].endsWith('check-logging-standard.mjs')) {
|
|
37
|
+
const input = process.argv[2];
|
|
38
|
+
if (input) {
|
|
39
|
+
if (hasUnstructuredLogging(input)) {
|
|
40
|
+
console.error('⚠️ Unstructured logging detected. Use structured logging (e.g., JSON logger) instead.');
|
|
41
|
+
process.exit(1);
|
|
42
|
+
} else {
|
|
43
|
+
process.exit(0);
|
|
44
|
+
}
|
|
45
|
+
} else {
|
|
46
|
+
let data = '';
|
|
47
|
+
process.stdin.setEncoding('utf-8');
|
|
48
|
+
process.stdin.on('data', (chunk) => { data += chunk; });
|
|
49
|
+
process.stdin.on('end', () => {
|
|
50
|
+
const code = data.trim();
|
|
51
|
+
if (hasUnstructuredLogging(code)) {
|
|
52
|
+
console.error('⚠️ Unstructured logging detected. Use structured logging (e.g., JSON logger) instead.');
|
|
53
|
+
process.exit(1);
|
|
54
|
+
} else {
|
|
55
|
+
process.exit(0);
|
|
56
|
+
}
|
|
57
|
+
});
|
|
58
|
+
}
|
|
59
|
+
}
|
|
@@ -0,0 +1,233 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* UDS Hook: Turn Completion Integrity
|
|
4
|
+
*
|
|
5
|
+
* Runs at turn end. Blocks when the agent's final message states a first-person
|
|
6
|
+
* commitment to a next action that the turn then ended without taking.
|
|
7
|
+
*
|
|
8
|
+
* Contract (Claude Code Stop hook):
|
|
9
|
+
* stdin — JSON with session_id, transcript_path, stop_hook_active
|
|
10
|
+
* block — print {"decision":"block","reason":"..."} on stdout, exit 0
|
|
11
|
+
* allow — print nothing, exit 0
|
|
12
|
+
*
|
|
13
|
+
* Every failure path allows. See core/turn-completion-integrity.md R5: a hook
|
|
14
|
+
* that can trap a session is worse than none, because the only recovery a human
|
|
15
|
+
* has is to disable it, and they will disable it permanently.
|
|
16
|
+
*
|
|
17
|
+
* Usage: node check-turn-completion.mjs (reads stdin)
|
|
18
|
+
* node check-turn-completion.mjs --self-test
|
|
19
|
+
* node check-turn-completion.mjs --languages
|
|
20
|
+
*
|
|
21
|
+
* @see docs/specs/SPEC-HOOKS-001-core-standard-hooks.md
|
|
22
|
+
* @see core/turn-completion-integrity.md
|
|
23
|
+
*/
|
|
24
|
+
import { readFileSync, mkdirSync, writeFileSync, existsSync } from 'node:fs';
|
|
25
|
+
import { homedir } from 'node:os';
|
|
26
|
+
import { join, dirname } from 'node:path';
|
|
27
|
+
import { fileURLToPath } from 'node:url';
|
|
28
|
+
import { detectCommitment, userAskedToStop } from './turn-completion/detect.mjs';
|
|
29
|
+
|
|
30
|
+
export const VERSION = '1.1.0';
|
|
31
|
+
|
|
32
|
+
const COOLDOWN_SEC = 120;
|
|
33
|
+
// A cap counted per session with no reset is an off switch on a delay: it
|
|
34
|
+
// disarms silently in exactly the long session where the rule matters most.
|
|
35
|
+
// A rolling window bounds runaway loops just as well and recovers by itself.
|
|
36
|
+
const MAX_BLOCKS_PER_WINDOW = 6;
|
|
37
|
+
const WINDOW_SEC = 3600;
|
|
38
|
+
|
|
39
|
+
const STATE_DIR = join(homedir(), '.uds', 'turn-completion');
|
|
40
|
+
const HERE = dirname(fileURLToPath(import.meta.url));
|
|
41
|
+
|
|
42
|
+
/** Locales this hook ships. The self-test fails if any of them will not load. */
|
|
43
|
+
export const SHIPPED_LOCALES = ['en', 'zh-TW'];
|
|
44
|
+
|
|
45
|
+
/**
|
|
46
|
+
* Load every shipped locale pack.
|
|
47
|
+
*
|
|
48
|
+
* At runtime a pack that fails to load is skipped (R5: a broken pack must not
|
|
49
|
+
* stop the turn from ending). 🔴 But that swallow is also how a shipping
|
|
50
|
+
* mistake hides: with zero packs loaded the hook runs, exits 0, and can never
|
|
51
|
+
* fire — the same shape as good behaviour. So `failed` is returned rather than
|
|
52
|
+
* discarded, and the self-test treats a non-empty `failed` as a failure.
|
|
53
|
+
* Caught while renaming the packs to .mjs: the loader still asked for .js.
|
|
54
|
+
*/
|
|
55
|
+
export async function loadPacks() {
|
|
56
|
+
const packs = [];
|
|
57
|
+
const failed = [];
|
|
58
|
+
for (const id of SHIPPED_LOCALES) {
|
|
59
|
+
try {
|
|
60
|
+
packs.push(await import(join(HERE, 'turn-completion', 'locales', `${id}.mjs`)));
|
|
61
|
+
} catch (e) {
|
|
62
|
+
failed.push({ id, why: String((e && e.message) || e) });
|
|
63
|
+
}
|
|
64
|
+
}
|
|
65
|
+
return { packs, failed };
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
/**
|
|
69
|
+
* 🔴 This hook's own block message enters the transcript as a user turn. On the
|
|
70
|
+
* next run it would therefore be read as "the human's last message", hiding the
|
|
71
|
+
* real one ("pause, I'm going home") and silently voiding the R9 exemption.
|
|
72
|
+
*
|
|
73
|
+
* Same family as the detector matching the prose that documents it — except
|
|
74
|
+
* here it reads its own output. The reason string carries this line so it can
|
|
75
|
+
* recognise itself.
|
|
76
|
+
*/
|
|
77
|
+
export const SELF_ECHO = 'UDS standard turn-completion-integrity (R1)';
|
|
78
|
+
|
|
79
|
+
/** Plain text of a transcript message, or '' if it carries none. */
|
|
80
|
+
function textOf(msg) {
|
|
81
|
+
const c = msg && msg.content;
|
|
82
|
+
if (typeof c === 'string') return c;
|
|
83
|
+
if (!Array.isArray(c)) return '';
|
|
84
|
+
// Tool results also arrive with role "user"; only text blocks are prose.
|
|
85
|
+
const parts = c.filter((p) => p && p.type === 'text').map((p) => p.text || '');
|
|
86
|
+
return parts.some((p) => p.trim()) ? parts.join('\n') : '';
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
/**
|
|
90
|
+
* Last assistant message and last human message in a Claude Code JSONL
|
|
91
|
+
* transcript. The human's message is needed because a turn that ends by their
|
|
92
|
+
* instruction looks, from the agent's words alone, exactly like one that ends
|
|
93
|
+
* on an abandoned commitment.
|
|
94
|
+
*/
|
|
95
|
+
export function lastMessages(transcriptPath) {
|
|
96
|
+
let assistant = '';
|
|
97
|
+
let user = '';
|
|
98
|
+
for (const line of readFileSync(transcriptPath, 'utf8').split('\n')) {
|
|
99
|
+
if (!line.trim()) continue;
|
|
100
|
+
let ev;
|
|
101
|
+
try { ev = JSON.parse(line); } catch { continue; }
|
|
102
|
+
const msg = ev && ev.message;
|
|
103
|
+
if (!msg) continue;
|
|
104
|
+
const t = textOf(msg);
|
|
105
|
+
if (!t) continue;
|
|
106
|
+
if (msg.role === 'assistant') assistant = t;
|
|
107
|
+
else if (msg.role === 'user' && !t.includes(SELF_ECHO)) user = t;
|
|
108
|
+
}
|
|
109
|
+
return { assistant, user };
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
function readState(path) {
|
|
113
|
+
try { return JSON.parse(readFileSync(path, 'utf8')); } catch { return {}; }
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
function reason(packId, sentence) {
|
|
117
|
+
return [
|
|
118
|
+
'Your last message stated a next action, and then the turn ended without taking it.',
|
|
119
|
+
'',
|
|
120
|
+
`Detected by the ${packId} pack, in: "${sentence}"`,
|
|
121
|
+
'',
|
|
122
|
+
'UDS standard turn-completion-integrity (R1): a stated next action is not optional.',
|
|
123
|
+
'',
|
|
124
|
+
'Do one of these now:',
|
|
125
|
+
' (a) take the action you just said you would take;',
|
|
126
|
+
' (b) if it is actually blocked, say what blocks it and what you need, then do',
|
|
127
|
+
' the next item that is not blocked;',
|
|
128
|
+
' (c) if everything is done or blocked, list every remaining item WITH the',
|
|
129
|
+
' person or input it waits on. That itemized list is the only shape of',
|
|
130
|
+
' ending that R2 accepts — "the main parts are done" is not one.',
|
|
131
|
+
].join('\n');
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
async function main() {
|
|
135
|
+
let raw = '';
|
|
136
|
+
try {
|
|
137
|
+
raw = readFileSync(0, 'utf8');
|
|
138
|
+
} catch {
|
|
139
|
+
return;
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
let data;
|
|
143
|
+
try { data = JSON.parse(raw); } catch { return; }
|
|
144
|
+
if (!data || typeof data !== 'object') return;
|
|
145
|
+
if (data.stop_hook_active === true) return;
|
|
146
|
+
|
|
147
|
+
const tp = data.transcript_path;
|
|
148
|
+
if (!tp || !existsSync(tp)) return; // cannot tell is not the same as should block
|
|
149
|
+
|
|
150
|
+
const sid = String(data.session_id || 'unknown');
|
|
151
|
+
const statePath = join(STATE_DIR, `${sid}.json`);
|
|
152
|
+
const st = readState(statePath);
|
|
153
|
+
const now = Date.now() / 1000;
|
|
154
|
+
|
|
155
|
+
const stamps = (Array.isArray(st.stamps) ? st.stamps : [])
|
|
156
|
+
.filter((t) => typeof t === 'number' && now - t < WINDOW_SEC);
|
|
157
|
+
if (stamps.length >= MAX_BLOCKS_PER_WINDOW) return;
|
|
158
|
+
if (now - (st.last || 0) < COOLDOWN_SEC) return;
|
|
159
|
+
|
|
160
|
+
let msgs;
|
|
161
|
+
try { msgs = lastMessages(tp); } catch { return; }
|
|
162
|
+
if (!msgs.assistant.trim()) return;
|
|
163
|
+
|
|
164
|
+
const { packs } = await loadPacks();
|
|
165
|
+
if (packs.length === 0) return;
|
|
166
|
+
|
|
167
|
+
// The human asked for the turn to end. That is a legitimate ending, and the
|
|
168
|
+
// agent's own words cannot distinguish it from an abandoned commitment.
|
|
169
|
+
if (userAskedToStop(msgs.user, packs)) return;
|
|
170
|
+
|
|
171
|
+
const hit = detectCommitment(msgs.assistant, packs);
|
|
172
|
+
if (!hit.fired) return;
|
|
173
|
+
|
|
174
|
+
stamps.push(now);
|
|
175
|
+
try {
|
|
176
|
+
mkdirSync(STATE_DIR, { recursive: true });
|
|
177
|
+
writeFileSync(statePath, JSON.stringify({ stamps, last: now }));
|
|
178
|
+
} catch {
|
|
179
|
+
/* state is an optimisation; failing to write it must not change the verdict */
|
|
180
|
+
}
|
|
181
|
+
|
|
182
|
+
process.stdout.write(JSON.stringify({
|
|
183
|
+
decision: 'block',
|
|
184
|
+
reason: reason(hit.packId, hit.sentence),
|
|
185
|
+
}));
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
async function selfTest() {
|
|
189
|
+
const { packs, failed } = await loadPacks();
|
|
190
|
+
let ok = failed.length === 0 && packs.length === SHIPPED_LOCALES.length;
|
|
191
|
+
console.log(`[turn-completion] v${VERSION} — packs: ${packs.map((p) => p.id).join(', ')}`
|
|
192
|
+
+ ` (${packs.length}/${SHIPPED_LOCALES.length} shipped locales)`);
|
|
193
|
+
for (const f of failed) console.log(` x locale ${f.id} failed to load — ${f.why}`);
|
|
194
|
+
|
|
195
|
+
for (const pack of packs) {
|
|
196
|
+
for (const [want, label, text] of pack.corpus) {
|
|
197
|
+
// Run against ALL packs, which is the shipped configuration. A pack that
|
|
198
|
+
// is correct alone and wrong beside another is not correct.
|
|
199
|
+
const got = detectCommitment(text, packs).fired;
|
|
200
|
+
const good = got === want;
|
|
201
|
+
ok &&= good;
|
|
202
|
+
console.log(` ${good ? 'OK ' : 'x '} [${pack.id}] ${want ? 'must block' : 'must pass'} — ${label}`
|
|
203
|
+
+ (good ? '' : ` (got: ${got ? 'block' : 'pass'})`));
|
|
204
|
+
}
|
|
205
|
+
for (const [want, label, text] of pack.stopCorpus || []) {
|
|
206
|
+
const got = userAskedToStop(text, packs);
|
|
207
|
+
const good = got === want;
|
|
208
|
+
ok &&= good;
|
|
209
|
+
console.log(` ${good ? 'OK ' : 'x '} [${pack.id}] user ${want ? 'IS' : 'is NOT'} asking to stop — ${label}`
|
|
210
|
+
+ (good ? '' : ` (got: ${got ? 'exempt' : 'not exempt'})`));
|
|
211
|
+
}
|
|
212
|
+
}
|
|
213
|
+
// The block message must contain the marker, or the hook cannot tell its own
|
|
214
|
+
// output from the human's next instruction.
|
|
215
|
+
const selfRecognised = reason('en', 'x').includes(SELF_ECHO);
|
|
216
|
+
ok &&= selfRecognised;
|
|
217
|
+
console.log(` ${selfRecognised ? 'OK ' : 'x '} block message carries the self-echo marker`);
|
|
218
|
+
|
|
219
|
+
console.log(`[turn-completion] self-test ${ok ? 'passed' : 'FAILED'}`);
|
|
220
|
+
process.exit(ok ? 0 : 1);
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
const arg = process.argv[2];
|
|
224
|
+
if (arg === '--self-test') {
|
|
225
|
+
await selfTest();
|
|
226
|
+
} else if (arg === '--languages') {
|
|
227
|
+
const { packs } = await loadPacks();
|
|
228
|
+
console.log(packs.map((p) => `${p.id} (${p.label})`).join('\n'));
|
|
229
|
+
console.log('\nThis check reads prose. If you work in a language not listed above,'
|
|
230
|
+
+ '\nit is installed and running but cannot fire. See turn-completion-integrity R8.');
|
|
231
|
+
} else {
|
|
232
|
+
try { await main(); } catch { /* R5 */ }
|
|
233
|
+
}
|