gentle-pi 2.5.0 → 2.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (189) hide show
  1. package/README.md +139 -30
  2. package/assets/agents/gentle-ai-worker.md +4 -0
  3. package/assets/agents/jd-fix-agent.md +18 -0
  4. package/assets/agents/jd-judge-a.md +1 -1
  5. package/assets/agents/jd-judge-b.md +1 -1
  6. package/assets/agents/sdd-apply.md +7 -5
  7. package/assets/agents/sdd-archive.md +5 -3
  8. package/assets/agents/sdd-design.md +4 -0
  9. package/assets/agents/sdd-explore.md +4 -0
  10. package/assets/agents/sdd-init.md +4 -0
  11. package/assets/agents/sdd-onboard.md +4 -0
  12. package/assets/agents/sdd-proposal.md +4 -0
  13. package/assets/agents/sdd-remediate.md +37 -0
  14. package/assets/agents/sdd-research.md +26 -3
  15. package/assets/agents/sdd-spec.md +4 -0
  16. package/assets/agents/sdd-status.md +9 -75
  17. package/assets/agents/sdd-sync.md +4 -0
  18. package/assets/agents/sdd-tasks.md +4 -0
  19. package/assets/agents/sdd-verify.md +5 -3
  20. package/assets/chains/sdd-full.chain.md +4 -0
  21. package/assets/chains/sdd-plan.chain.md +4 -0
  22. package/assets/chains/sdd-verify.chain.md +4 -0
  23. package/assets/migrations/managed-assets-v2.5.0.json +7 -0
  24. package/assets/orchestrator-delegation.md +21 -3
  25. package/assets/sdd-orchestrator-workflow.md +54 -21
  26. package/assets/support/sdd-status-contract.md +34 -90
  27. package/contracts/telemetry/runtime-aggregate-v1.schema.json +65 -0
  28. package/docs/telemetry.md +57 -1
  29. package/docs/windows-startup-console-visibility.md +18 -0
  30. package/extensions/ask-user-choice.ts +143 -15
  31. package/extensions/codegraph-tools.ts +1 -0
  32. package/extensions/gentle-agents.ts +799 -50
  33. package/extensions/gentle-ai.ts +2033 -322
  34. package/extensions/gentle-shell.ts +145 -42
  35. package/extensions/gentle-todo.ts +47 -12
  36. package/extensions/quiet-tools.ts +1 -0
  37. package/extensions/runtime-metrics.ts +130 -0
  38. package/extensions/sdd-init.ts +2 -2
  39. package/extensions/startup-banner.ts +52 -75
  40. package/lib/agent-profiles.ts +550 -0
  41. package/lib/agents-completion-delivery.ts +72 -0
  42. package/lib/agents-config.ts +7 -10
  43. package/lib/agents-history.ts +9 -1
  44. package/lib/agents-messaging.ts +187 -0
  45. package/lib/agents-protocol.ts +77 -5
  46. package/lib/agents-runner.ts +548 -26
  47. package/lib/agents-thread-view.ts +57 -0
  48. package/lib/agents-view-layout.ts +40 -0
  49. package/lib/agents-view.ts +548 -191
  50. package/lib/agents-widget.ts +33 -14
  51. package/lib/gentle-ai-binary.ts +3 -1
  52. package/lib/gentle-ai-renderer.ts +8 -6
  53. package/lib/native-review-cli.ts +283 -1
  54. package/lib/orchestrator-presence.ts +337 -0
  55. package/lib/profiles-orchestrator.ts +203 -0
  56. package/lib/review-candidate-view-owner.ts +296 -46
  57. package/lib/review-candidate-view.ts +30 -20
  58. package/lib/review-consent-component.ts +247 -0
  59. package/lib/review-consent-ui.ts +53 -8
  60. package/lib/review-host-relay.ts +28 -0
  61. package/lib/review-integration-v2.ts +187 -5
  62. package/lib/review-last-event-controller.ts +7 -4
  63. package/lib/review-reminder-receipt.ts +74 -0
  64. package/lib/review-session-standing-permission.ts +27 -6
  65. package/lib/runtime-metrics-children.ts +199 -0
  66. package/lib/runtime-metrics-delivery.ts +68 -0
  67. package/lib/runtime-metrics-native.ts +166 -0
  68. package/lib/runtime-metrics-pi-identity.ts +113 -0
  69. package/lib/runtime-metrics-policy.ts +51 -0
  70. package/lib/runtime-metrics.ts +255 -0
  71. package/lib/sdd-preflight.ts +362 -81
  72. package/lib/sdd-research-capabilities.ts +228 -0
  73. package/lib/sdd-status.ts +29 -7
  74. package/lib/session-worktree-registry.ts +118 -0
  75. package/lib/shell-bar.ts +47 -1
  76. package/lib/shell-card.ts +1 -4
  77. package/lib/shell-changes-view.ts +362 -37
  78. package/lib/shell-changes.ts +81 -1
  79. package/lib/shell-prompt.ts +11 -15
  80. package/lib/shell-sidebar-banner.ts +11 -0
  81. package/lib/shell-sidebar-layout.ts +213 -0
  82. package/lib/shell-sidebar.ts +41 -0
  83. package/lib/shell-todo.ts +28 -11
  84. package/lib/telemetry-trigger.ts +2 -0
  85. package/package.json +6 -3
  86. package/runtime/gentle-ai-binary.mjs +3 -1
  87. package/runtime/native-review-cli.mjs +283 -1
  88. package/runtime/review-integration-v2.mjs +187 -5
  89. package/runtime/telemetry-trigger.mjs +2 -0
  90. package/scripts/build-runtime-modules.mjs +9 -1
  91. package/scripts/check-types.mjs +125 -0
  92. package/scripts/gentle-ai-installer.mjs +10 -10
  93. package/scripts/install-gentle-ai.mjs +12 -0
  94. package/scripts/install-tui-mode-setting.mjs +114 -0
  95. package/scripts/test-packed-runner.mjs +16 -2
  96. package/scripts/types-baseline.json +99 -0
  97. package/scripts/verify-package-files.mjs +4 -2
  98. package/skills/_shared/review-ledger-contract.md +17 -1
  99. package/skills/issue-creation/SKILL.md +3 -3
  100. package/skills/judgment-day/SKILL.md +17 -3
  101. package/skills/judgment-day/references/prompts-and-formats.md +14 -3
  102. package/tests/agent-profiles.test.ts +722 -0
  103. package/tests/agents-completion-delivery.test.ts +94 -0
  104. package/tests/agents-config.test.ts +62 -0
  105. package/tests/agents-fake-child.ts +15 -1
  106. package/tests/agents-grouping.test.ts +179 -0
  107. package/tests/agents-integration.test.ts +100 -0
  108. package/tests/agents-messaging.test.ts +94 -0
  109. package/tests/agents-protocol.test.ts +45 -0
  110. package/tests/agents-queries.test.ts +190 -0
  111. package/tests/agents-responsive.test.ts +43 -0
  112. package/tests/agents-runner.test.ts +571 -14
  113. package/tests/agents-thread-view.test.ts +45 -0
  114. package/tests/agents-view.test.ts +476 -65
  115. package/tests/agents-widget.test.ts +31 -1
  116. package/tests/artifact-language.test.ts +25 -2
  117. package/tests/ask-user-choice.test.ts +169 -3
  118. package/tests/asset-installation-runtime.test.ts +108 -0
  119. package/tests/autonomous-guard.test.ts +116 -1
  120. package/tests/codegraph-tools.test.ts +2 -1
  121. package/tests/delegated-key-learnings-contract.test.ts +1 -1
  122. package/tests/devbinary/native-review-parity.devtest.ts +2 -0
  123. package/tests/feature-request-form.test.ts +67 -0
  124. package/tests/fixtures/agents-messaging-child.mjs +5 -0
  125. package/tests/fixtures/runtime-metrics-native-batches.json +6 -0
  126. package/tests/gentle-agents.test.ts +1460 -33
  127. package/tests/gentle-ai-binary.test.ts +7 -2
  128. package/tests/gentle-ai-installer.test.ts +47 -47
  129. package/tests/gentle-ai-renderer.test.ts +38 -0
  130. package/tests/gentle-ai.test.ts +944 -4
  131. package/tests/gentle-shell.test.ts +303 -12
  132. package/tests/gentle-todo.test.ts +54 -10
  133. package/tests/install-tui-mode-setting.test.ts +324 -0
  134. package/tests/issue-creation-skill.test.ts +22 -0
  135. package/tests/model-routing-authority.test.ts +12 -0
  136. package/tests/native-review-capability-contract.test.ts +12 -1
  137. package/tests/native-review-cli.test.ts +277 -3
  138. package/tests/native-review-parity.test.ts +14 -7
  139. package/tests/native-sdd-attempt-authority.test.ts +7 -2
  140. package/tests/orchestrator-presence.test.ts +389 -0
  141. package/tests/package-manifest.test.ts +232 -7
  142. package/tests/profiles-orchestrator.test.ts +208 -0
  143. package/tests/quiet-tool-rendering.test.ts +1 -0
  144. package/tests/rdd-aware-verification-contract.test.ts +10 -0
  145. package/tests/review-agent-end-preflight.test.ts +332 -24
  146. package/tests/review-candidate-view.test.ts +304 -6
  147. package/tests/review-consent-ui.test.ts +352 -0
  148. package/tests/review-contract-prompt.test.ts +14 -0
  149. package/tests/review-controller-native-routing.test.ts +563 -3
  150. package/tests/review-controller.test.ts +1 -1
  151. package/tests/review-host-relay-restart-parity.test.ts +142 -1
  152. package/tests/review-host-relay-routing.test.ts +364 -4
  153. package/tests/review-host-relay.test.ts +29 -0
  154. package/tests/review-integration-v2-forward.test.ts +44 -0
  155. package/tests/review-integration-v2.test.ts +164 -0
  156. package/tests/review-last-event-closure.test.ts +105 -1
  157. package/tests/review-ledger-contract.test.ts +61 -6
  158. package/tests/review-reminder-receipt.test.ts +62 -0
  159. package/tests/review-session-standing-permission-controller.test.ts +52 -4
  160. package/tests/review-session-standing-permission.test.ts +30 -0
  161. package/tests/runtime-harness.mjs +447 -39
  162. package/tests/runtime-metrics-children.test.ts +206 -0
  163. package/tests/runtime-metrics-delivery.test.ts +85 -0
  164. package/tests/runtime-metrics-extension.test.ts +187 -0
  165. package/tests/runtime-metrics-native.test.ts +209 -0
  166. package/tests/runtime-metrics-pi-identity.test.ts +113 -0
  167. package/tests/runtime-metrics-policy.test.ts +62 -0
  168. package/tests/runtime-metrics.test.ts +184 -0
  169. package/tests/sdd-agent-tools.test.ts +10 -1
  170. package/tests/sdd-execution-routing-contract.test.ts +28 -0
  171. package/tests/sdd-managed-runtime-settlement.test.ts +331 -0
  172. package/tests/sdd-native-managed-uptake.test.ts +253 -0
  173. package/tests/sdd-planning-routing-contract.test.ts +45 -0
  174. package/tests/sdd-preflight.test.ts +252 -8
  175. package/tests/sdd-research-capabilities.test.ts +256 -0
  176. package/tests/sdd-research-live.test.ts +241 -0
  177. package/tests/sdd-selection-transport.test.ts +504 -0
  178. package/tests/sdd-status.test.ts +51 -0
  179. package/tests/session-worktree-registry.test.ts +135 -0
  180. package/tests/shell-card.test.ts +24 -3
  181. package/tests/shell-changes-view.test.ts +471 -8
  182. package/tests/shell-changes.test.ts +168 -0
  183. package/tests/shell-prompt.test.ts +28 -6
  184. package/tests/shell-sidebar-banner.test.ts +23 -0
  185. package/tests/shell-sidebar-layout.test.ts +387 -0
  186. package/tests/shell-sidebar.test.ts +50 -0
  187. package/tests/shell-todo.test.ts +100 -11
  188. package/tests/startup-banner.test.ts +126 -0
  189. package/tests/telemetry-trigger.test.ts +3 -1
@@ -0,0 +1,209 @@
1
+ import assert from "node:assert/strict";
2
+ import { EventEmitter } from "node:events";
3
+ import { PassThrough } from "node:stream";
4
+ import { test } from "node:test";
5
+ import { readFileSync, realpathSync } from "node:fs";
6
+ import { setGentleAiDevBinaryEnvironmentForTesting } from "../lib/gentle-ai-binary.ts";
7
+ import * as native from "../lib/runtime-metrics-native.ts";
8
+ import { parseAgentClass, type RuntimeMetricBucket } from "../lib/runtime-metrics.ts";
9
+ import { createHash } from "node:crypto";
10
+ import schema from "../contracts/telemetry/runtime-aggregate-v1.schema.json" with { type: "json" };
11
+ import fixturePayloads from "./fixtures/runtime-metrics-native-batches.json" with { type: "json" };
12
+
13
+ function source(): RuntimeMetricBucket {
14
+ const token = () => ({ reported: 1, unavailable: 0, unsupported: 0, sum: 7 });
15
+ const duration = () => ({ measured: 0, unavailable: 1, unsupported: 0, sum: 0 });
16
+ return { hostAgent: "pi", agentClass: parseAgentClass("worker")!, executor: "worker", provider: "openai", modelFamily: "gpt",
17
+ selectedProvider: "openai", selectedModelId: "gpt-5.4", observedModelId: "private-alias", responseModelId: "gpt-5.4",
18
+ effort: "high", providerThinkingLevel: "low", error: "none", responses: 1,
19
+ tokens: { input: token(), output: token(), cacheRead: token(), cacheWrite: token(), reasoning: token(), totalTokens: token() },
20
+ responseHeadersMs: duration(), fullResponseMs: duration() };
21
+ }
22
+
23
+ test("production encoder emits exact one-shot fixture with source occurrence coverage", () => {
24
+ const payload = native.encodeNativeRuntimeEvent([source()]);
25
+ assert.equal(typeof payload, "string", "production encoder must be enabled");
26
+ const value = JSON.parse(payload!);
27
+ assert.deepEqual(Object.keys(value).sort(), [...schema.required].sort());
28
+ assert.deepEqual(Object.keys(value.rows[0]).sort(), [...schema.$defs.row.required].sort());
29
+ assert.equal(value.rows[0].launches, null);
30
+ assert.equal(value.rows[0].responses, 1);
31
+ assert.equal(value.rows[0].selected_effort, "high");
32
+ assert.equal(value.rows[0].effective_effort, "low");
33
+ assert.equal(value.rows[0].model_evidence, "response");
34
+ assert.deepEqual(value.rows[0].model, { provider: "openai", id: "gpt-5.4" });
35
+ assert.ok(!payload!.includes("private"));
36
+ assert.ok(!payload!.includes("batch_id") && !payload!.includes("delivery_id"));
37
+ assert.deepEqual(fixturePayloads.batches, [payload]);
38
+ assert.equal(fixturePayloads.schema_sha256, createHash("sha256").update(readFileSync(new URL("../contracts/telemetry/runtime-aggregate-v1.schema.json", import.meta.url))).digest("hex"));
39
+ });
40
+
41
+ test("child launch occurrence retains selection evidence separately from response evidence", () => {
42
+ const payload = native.encodeNativeRuntimeEvent([source()], [{ evidence: "launch_configuration", agentClass: parseAgentClass("worker")!,
43
+ selectedProvider: "openai", selectedModelId: "gpt-5.4", selectedEffort: "high", launches: 1 }]);
44
+ const rows = JSON.parse(payload!).rows;
45
+ assert.equal(rows.length, 2);
46
+ assert.equal(rows[0].model_evidence, "response");
47
+ assert.equal(rows[1].model_evidence, "selected");
48
+ assert.equal(rows[1].launches, 1);
49
+ assert.equal(rows[1].responses, null);
50
+ assert.equal(rows[1].selected_effort, "high");
51
+ assert.equal(rows[1].effective_effort, "unavailable");
52
+ for (const field of ["input_tokens", "output_tokens", "cache_read_tokens", "cache_creation_tokens", "reasoning_tokens", "total_tokens"])
53
+ assert.deepEqual(rows[1][field], { reported: 0, unavailable: 0, unsupported: 0, sum: 0 });
54
+ });
55
+
56
+ test("encoder bounds and invalid coverage discard whole events without splitting", () => {
57
+ assert.equal(native.encodeNativeRuntimeEvent([]), undefined);
58
+ assert.equal(native.encodeNativeRuntimeEvent(Array(33).fill(source())), undefined);
59
+ const invalid = source(); invalid.tokens.input.reported = 0;
60
+ assert.equal(native.encodeNativeRuntimeEvent([invalid]), undefined);
61
+ invalid.tokens.input.sum = -1;
62
+ assert.equal(native.encodeNativeRuntimeEvent([invalid]), undefined);
63
+ const valid = source();
64
+ assert.equal(typeof native.encodeNativeRuntimeEvent([valid]), "string");
65
+ assert.equal(native.encodeNativeRuntimeEvent(Array(32).fill(valid)), undefined, "16 KiB bound applies even below row limit");
66
+ valid.fullResponseMs = { measured: 1, unavailable: 0, unsupported: 0, sum: 12.5 };
67
+ assert.deepEqual(JSON.parse(native.encodeNativeRuntimeEvent([valid])!).rows[0].duration,
68
+ { kind: "request", measured_count: 1, sum_ms: 12.5 });
69
+ valid.tokens.input = { reported: 0, unavailable: 0, unsupported: 1, sum: 0 };
70
+ assert.deepEqual(JSON.parse(native.encodeNativeRuntimeEvent([valid])!).rows[0].input_tokens, valid.tokens.input);
71
+ });
72
+
73
+ test("encoder filters custom identities and never promotes SDK model to response proof", () => {
74
+ const row = source(); row.provider = "custom"; row.responseModelId = "private-model";
75
+ row.agentClass = "private-agent" as any; row.effort = "private-effort" as any;
76
+ row.error = "authentication";
77
+ const encoded = JSON.parse(native.encodeNativeRuntimeEvent([row])!);
78
+ assert.deepEqual(encoded.rows[0].model, { provider: "custom", id: "custom" });
79
+ assert.equal(encoded.rows[0].agent_class, "unknown");
80
+ assert.equal(encoded.rows[0].selected_effort, "unavailable");
81
+ assert.equal(encoded.rows[0].error_category, "auth");
82
+ assert.ok(!JSON.stringify(encoded).includes("private"));
83
+ row.responseModelId = "unknown";
84
+ const selected = JSON.parse(native.encodeNativeRuntimeEvent([row])!);
85
+ assert.equal(selected.rows[0].model_evidence, "selected");
86
+ assert.deepEqual(selected.rows[0].model, { provider: "openai", id: "gpt-5.4" });
87
+ row.selectedModelId = "unknown";
88
+ const unknown = JSON.parse(native.encodeNativeRuntimeEvent([row])!);
89
+ assert.equal(unknown.rows[0].model_evidence, "unknown");
90
+ });
91
+
92
+ function fixture() {
93
+ const child = Object.assign(new EventEmitter(), {
94
+ stdin: new PassThrough(), stdout: new PassThrough(), stderr: new PassThrough(),
95
+ kill: () => { killed++; return true; }, unref() {},
96
+ });
97
+ let killed = 0;
98
+ const calls: unknown[][] = [];
99
+ let input = "";
100
+ child.stdin.on("data", chunk => { input += chunk; });
101
+ const deps = {
102
+ env: {}, resolve: () => "fake-native",
103
+ encode: () => '{"host":"pi","rows":[]}',
104
+ spawn: (...args: unknown[]) => { calls.push(args); return child as any; },
105
+ };
106
+ return { deps, child, calls, input: () => input, killed: () => killed,
107
+ close: (decision = "stored") => {
108
+ child.stdout.write(JSON.stringify({ schema: native.NATIVE_SEND_ACK_SCHEMA, decision }));
109
+ child.emit("close", 0, null);
110
+ } };
111
+ }
112
+
113
+ test("one native stdin send; busy drops without another invocation or policy probe", async () => {
114
+ const f = fixture();
115
+ const first = native.sendNativeRuntimeEvent([], "/fixture", f.deps);
116
+ assert.equal(f.calls.length, 1);
117
+ assert.deepEqual(f.calls[0][1], ["telemetry", "runtime", "send", "--json"]);
118
+ assert.equal(f.input(), '{"host":"pi","rows":[]}');
119
+ assert.equal(await native.sendNativeRuntimeEvent([], "/fixture", f.deps), "discarded");
120
+ f.close();
121
+ assert.equal(await first, "stored");
122
+ assert.equal(f.calls.length, 1);
123
+ });
124
+
125
+ test("production encoder reaches the fake one-shot child without a test encoding override", async () => {
126
+ const f = fixture();
127
+ const { encode: _encode, ...deps } = f.deps;
128
+ const first = native.sendNativeRuntimeEvent([source()], "/fixture", deps);
129
+ assert.equal(f.calls.length, 1);
130
+ assert.equal(f.input(), fixturePayloads.batches[0]);
131
+ f.close(); assert.equal(await first, "stored");
132
+ });
133
+
134
+ test("validated dev override reaches the one-shot child through the production resolver", async () => {
135
+ const file = realpathSync(process.execPath);
136
+ setGentleAiDevBinaryEnvironmentForTesting({ env: { GENTLE_PI_GENTLE_AI_DEV_BINARY: file }, home: "/unused" });
137
+ try {
138
+ const f = fixture();
139
+ const { resolve: _resolve, encode: _encode, ...deps } = f.deps;
140
+ const pending = native.sendNativeRuntimeEvent([source()], "/fixture", deps);
141
+ f.close();
142
+ assert.equal(await pending, "stored");
143
+ assert.equal(f.calls.length, 1);
144
+ assert.equal(f.calls[0][0], file);
145
+ assert.deepEqual(f.calls[0][1], ["telemetry", "runtime", "send", "--json"]);
146
+ assert.equal(f.input(), fixturePayloads.batches[0]);
147
+ } finally {
148
+ setGentleAiDevBinaryEnvironmentForTesting(undefined);
149
+ }
150
+ });
151
+
152
+ test("invalid dev overrides fail closed without spawning or falling back", async () => {
153
+ for (const file of ["relative-binary", realpathSync(new URL(".", import.meta.url))]) {
154
+ setGentleAiDevBinaryEnvironmentForTesting({ env: { GENTLE_PI_GENTLE_AI_DEV_BINARY: file }, home: "/unused" });
155
+ try {
156
+ const f = fixture();
157
+ const { resolve: _resolve, ...deps } = f.deps;
158
+ assert.equal(await native.sendNativeRuntimeEvent([], "/fixture", deps), "discarded");
159
+ assert.equal(f.calls.length, 0);
160
+ } finally {
161
+ setGentleAiDevBinaryEnvironmentForTesting(undefined);
162
+ }
163
+ }
164
+ });
165
+
166
+ test("opt-out and pre-cancel cause zero encoding, resolution or launch", async () => {
167
+ const f = fixture();
168
+ f.deps.resolve = () => { throw new Error("must not resolve"); };
169
+ f.deps.encode = () => { throw new Error("must not encode"); };
170
+ assert.equal(await native.sendNativeRuntimeEvent([], "/fixture", { ...f.deps, env: { DO_NOT_TRACK: "1" } }), "disabled");
171
+ assert.equal(await native.sendNativeRuntimeEvent([], "/fixture", { ...f.deps, signal: AbortSignal.abort() }), "discarded");
172
+ assert.equal(f.calls.length, 0);
173
+ });
174
+
175
+ test("cancel kills once and keeps slot occupied until actual close", async () => {
176
+ const f = fixture();
177
+ const abort = new AbortController();
178
+ const first = native.sendNativeRuntimeEvent([], "/fixture", { ...f.deps, signal: abort.signal });
179
+ abort.abort(); abort.abort();
180
+ assert.equal(f.killed(), 1);
181
+ assert.equal(await native.sendNativeRuntimeEvent([], "/fixture", f.deps), "discarded");
182
+ f.close();
183
+ assert.equal(await first, "discarded");
184
+ });
185
+
186
+ for (const decision of ["discarded", "disabled", "stored", "duplicate", "sent", "private error"]) {
187
+ test(`native result ${decision} never triggers retry`, async () => {
188
+ const f = fixture();
189
+ const first = native.sendNativeRuntimeEvent([], "/fixture", f.deps);
190
+ f.close(decision);
191
+ assert.equal(await first, ["stored", "duplicate", "disabled"].includes(decision) ? decision : "discarded");
192
+ assert.equal(f.calls.length, 1);
193
+ });
194
+ }
195
+
196
+ test("metrics receiver, attempt gate and send transport have no filesystem persistence surface", () => {
197
+ for (const path of ["../extensions/runtime-metrics.ts", "../lib/runtime-metrics-delivery.ts", "../lib/runtime-metrics-native.ts"]) {
198
+ const source = readFileSync(new URL(path, import.meta.url), "utf8");
199
+ assert.doesNotMatch(source, /node:fs|appendEntry\s*\(|writeFile|mkdir|createWriteStream/);
200
+ }
201
+ });
202
+
203
+ test("spawn failure silently discards", async () => {
204
+ const f = fixture();
205
+ let calls = 0;
206
+ assert.equal(await native.sendNativeRuntimeEvent([], "/fixture", { ...f.deps,
207
+ spawn: () => { calls++; throw new Error("private error"); } }), "discarded");
208
+ assert.equal(calls, 1);
209
+ });
@@ -0,0 +1,113 @@
1
+ import assert from "node:assert/strict";
2
+ import test from "node:test";
3
+ import { findPackageJSON } from "node:module";
4
+ import { OPENAI_CODEX_MODELS } from "@earendil-works/pi-ai/providers/openai-codex.models";
5
+ import { classifyPiCatalogName, createPiCatalogNameLookup, lookupPiCatalogName, type PiCatalogName } from "../lib/runtime-metrics-pi-identity.ts";
6
+ import { RuntimeMetrics, type FinalResponse } from "../lib/runtime-metrics.ts";
7
+
8
+ const model = Object.values(OPENAI_CODEX_MODELS)[0];
9
+ const name = { provider: model.provider, modelId: model.id };
10
+
11
+ test("a failed catalog load is retried lazily and later classifications recover", async () => {
12
+ let attempts = 0;
13
+ const publicName = Object.freeze({ classification: "catalog_public", modelId: name.modelId }) satisfies PiCatalogName;
14
+ const catalog = new Map([[JSON.stringify([name.provider, name.modelId]), publicName]]);
15
+ const lookup = createPiCatalogNameLookup(async () => {
16
+ attempts++;
17
+ if (attempts === 1) throw new Error("transient catalog failure");
18
+ return catalog;
19
+ });
20
+ await assert.rejects(lookup.lookup(name), /transient catalog failure/);
21
+ assert.deepEqual(lookup.classify(name), { classification: "unknown", modelId: "unknown" });
22
+ assert.strictEqual(await lookup.lookup(name), publicName);
23
+ assert.strictEqual(lookup.classify(name), publicName);
24
+ assert.equal(attempts, 2);
25
+ });
26
+
27
+ test("catalog loading attempts are capped per lookup instance", async () => {
28
+ let attempts = 0;
29
+ const lookup = createPiCatalogNameLookup(async () => {
30
+ attempts++;
31
+ throw new Error(`failure ${attempts}`);
32
+ });
33
+ for (let attempt = 0; attempt < 5; attempt++) await assert.rejects(lookup.lookup(name));
34
+ assert.equal(attempts, 3);
35
+ assert.deepEqual(lookup.classify(name), { classification: "unknown", modelId: "unknown" });
36
+ });
37
+
38
+ test("uninitialized synchronous classification fails closed", () => {
39
+ assert.deepEqual(classifyPiCatalogName(name), { classification: "unknown", modelId: "unknown" });
40
+ });
41
+
42
+ test("missing catalog name remains unknown; no route evidence is invented", async () => {
43
+ for (const input of [undefined, null, {}, { provider: model.provider }, { modelId: model.id }]) {
44
+ assert.deepEqual(await lookupPiCatalogName(input), { classification: "unknown", modelId: "unknown" });
45
+ }
46
+ });
47
+
48
+ test("custom aliases never export supplied strings", async () => {
49
+ for (const input of [{ ...name, modelId: "private-alias" }, { ...name, provider: "private-provider" }]) {
50
+ const result = await lookupPiCatalogName(input);
51
+ assert.deepEqual(result, { classification: "custom", modelId: "custom" });
52
+ assert.ok(Object.isFrozen(result));
53
+ }
54
+ });
55
+
56
+ test("caller-created public identities cannot bypass the catalog boundary", async () => {
57
+ for (const input of [{ modelId: "private-model", classification: "catalog_public" },
58
+ { ...name, modelId: "private-model", origin: "builtin" }]) {
59
+ assert.notEqual((await lookupPiCatalogName(input)).classification, "catalog_public");
60
+ }
61
+ });
62
+
63
+ test("malformed or oversized name metadata fails closed", async () => {
64
+ for (const patch of [{ modelId: "" }, { modelId: "x".repeat(129) }, { provider: 12 },
65
+ { modelId: null }, { provider: "x".repeat(33) }]) {
66
+ assert.deepEqual(await lookupPiCatalogName({ ...name, ...patch }), { classification: "unknown", modelId: "unknown" });
67
+ }
68
+ });
69
+
70
+ test("actual installed ESM catalog supplies immutable public names, not route claims", async () => {
71
+ const pi = import.meta.resolve("@earendil-works/pi-coding-agent");
72
+ assert.equal(findPackageJSON("@earendil-works/pi-ai", import.meta.url), findPackageJSON("@earendil-works/pi-ai", pi));
73
+ assert.ok(model);
74
+ const result = await lookupPiCatalogName(name);
75
+ assert.deepEqual(result, { classification: "catalog_public", modelId: model.id });
76
+ assert.ok(Object.isFrozen(result));
77
+ assert.strictEqual(await lookupPiCatalogName(name), result);
78
+ // Origin/endpoint assertions do not change privacy classification of a name.
79
+ // Neither a real catalog entry nor matching route strings establish dispatch.
80
+ for (const patch of [{ origin: "builtin" }, { origin: "unknown" }, { origin: "custom" },
81
+ { baseUrl: "https://private.invalid" }, { api: "custom-api" }]) {
82
+ assert.strictEqual(await lookupPiCatalogName({ ...name, ...patch }), result);
83
+ }
84
+ assert.deepEqual(Object.keys(result), ["classification", "modelId"]);
85
+ assert.ok(!("modelVersion" in result));
86
+ });
87
+
88
+ test("catalog-public selections are counted without claiming actual route identity", async () => {
89
+ const catalogName = await lookupPiCatalogName(name);
90
+ const metrics = new RuntimeMetrics();
91
+ const missing = { state: "unavailable" } as const;
92
+ for (const [index, identity] of [catalogName, { ...name, origin: "builtin", api: model.api, baseUrl: model.baseUrl }].entries()) {
93
+ const record = {
94
+ kind: "final_assistant_response", responseId: String(index), identity, selectedModelId: model.id,
95
+ executor: "worker", provider: model.provider, modelFamily: "gpt", effort: "high", error: "none",
96
+ tokens: { input: missing, output: missing, cacheRead: missing, cacheWrite: missing },
97
+ responseHeadersMs: missing, fullResponseMs: missing,
98
+ };
99
+ assert.equal(metrics.record(record as FinalResponse), "recorded");
100
+ }
101
+ const [bucket] = metrics.snapshot();
102
+ assert.equal(bucket.selectedModelId, model.id);
103
+ assert.ok(!("modelId" in bucket));
104
+ assert.ok(!("route" in bucket));
105
+ assert.equal(bucket.responses, 2);
106
+ assert.equal(bucket.hostAgent, "pi");
107
+ });
108
+
109
+ test("selection classification rejects fabricated public labels and private IDs", async () => {
110
+ await lookupPiCatalogName(name);
111
+ assert.equal(classifyPiCatalogName({ ...name, modelId: "private-id", classification: "catalog_public" }).modelId, "custom");
112
+ assert.equal(classifyPiCatalogName({ ...name, provider: "anthropic" }).modelId, "custom");
113
+ });
@@ -0,0 +1,62 @@
1
+ import assert from "node:assert/strict";
2
+ import test from "node:test";
3
+ import { readRuntimeMetricsPolicy, runtimeMetricsEnvAllows } from "../lib/runtime-metrics-policy.ts";
4
+
5
+ const grant = { schema: "gentle-ai.telemetry-policy/v1", operation: "policy", enabled: true, source: "state", reason: "enabled" };
6
+ const result = (value: unknown) => ({ stdout: JSON.stringify(value), stderr: "", exitCode: 0,
7
+ signal: null, timedOut: false, outputLimitExceeded: false });
8
+
9
+ test("only strict enabled policy grants; uses bounded policy invocation exclusively", async () => {
10
+ let calls = 0;
11
+ assert.equal(await readRuntimeMetricsPolicy("/workspace", {
12
+ resolve: () => "/fixture/gentle-ai", exec: async request => {
13
+ calls++;
14
+ assert.deepEqual(request.arguments, ["telemetry", "policy", "--json"]);
15
+ assert.equal(request.file, "/fixture/gentle-ai");
16
+ assert.equal(request.cwd, "/workspace");
17
+ assert.equal(request.timeoutMs, 1000);
18
+ assert.equal(request.maxBufferBytes, 2048);
19
+ return result(grant);
20
+ },
21
+ }), true);
22
+ assert.equal(calls, 1);
23
+ for (const patch of [{ schema: "wrong" }, { operation: "status" }, { enabled: "true" },
24
+ { enabled: false, reason: "disabled" }, { reason: "enrollment_pending" }, { reason: "state_unavailable" },
25
+ { source: "private" }, { raw: "private" }, { reason: undefined }]) {
26
+ assert.equal(await readRuntimeMetricsPolicy("/workspace", {
27
+ resolve: () => "fixture", exec: async () => result({ ...grant, ...patch }),
28
+ }), false);
29
+ }
30
+ });
31
+
32
+ test("missing, old, timed-out and malformed binary responses fail closed without fallback", async () => {
33
+ for (const patch of [{ exitCode: 1 }, { timedOut: true }, { outputLimitExceeded: true },
34
+ { signal: "SIGTERM" }, { stderr: "private error" }, { stdout: "unsupported command" },
35
+ { stdout: "x".repeat(2049) }, { stdout: "null" }, { stdout: "[]" }]) {
36
+ let calls = 0;
37
+ assert.equal(await readRuntimeMetricsPolicy("/workspace", { resolve: () => "fixture",
38
+ exec: async () => { calls++; return { ...result(grant), ...patch } as any; } }), false);
39
+ assert.equal(calls, 1);
40
+ }
41
+ assert.equal(await readRuntimeMetricsPolicy("/workspace", {
42
+ resolve: () => { throw new Error("missing"); }, exec: async () => assert.fail("must not spawn"),
43
+ }), false);
44
+ assert.equal(await readRuntimeMetricsPolicy("/workspace", {
45
+ resolve: () => "fixture", exec: async () => { throw new Error("ENOENT"); },
46
+ }), false);
47
+ });
48
+
49
+ test("outer deadline releases callers even when an exec adapter never settles", async () => {
50
+ let signal: AbortSignal | undefined;
51
+ assert.equal(await readRuntimeMetricsPolicy("/workspace", { resolve: () => "fixture",
52
+ exec: request => { signal = request.signal; return new Promise(() => {}); } }), false);
53
+ assert.equal(signal?.aborted, true);
54
+ });
55
+
56
+ test("native-compatible environment veto accepts broad truthy spellings", () => {
57
+ for (const key of ["DO_NOT_TRACK", "CI", "GITHUB_ACTIONS"]) {
58
+ for (const value of ["1", "true", "TRUE", " yes ", "on", "t", "unknown"]) assert.equal(runtimeMetricsEnvAllows({ [key]: value }), false);
59
+ }
60
+ assert.equal(runtimeMetricsEnvAllows({ GENTLE_AI_TELEMETRY: "0" }), false);
61
+ assert.equal(runtimeMetricsEnvAllows({ DO_NOT_TRACK: "false", CI: "0" }), true);
62
+ });
@@ -0,0 +1,184 @@
1
+ import assert from "node:assert/strict";
2
+ import test from "node:test";
3
+ import { RuntimeMetrics, type FinalResponse } from "../lib/runtime-metrics.ts";
4
+
5
+ function response(responseId = "local-response"): FinalResponse {
6
+ return {
7
+ kind: "final_assistant_response", responseId,
8
+ executor: "worker", provider: "openai-codex", modelFamily: "gpt",
9
+ effort: "high", error: "none",
10
+ tokens: {
11
+ input: { state: "reported", value: 0 }, output: { state: "reported", value: 12 },
12
+ cacheRead: { state: "unavailable" }, cacheWrite: { state: "unsupported" },
13
+ },
14
+ responseHeadersMs: { state: "measured", value: 25.5 },
15
+ fullResponseMs: { state: "measured", value: 100 },
16
+ };
17
+ }
18
+
19
+ test("selected provider is independent from response provider without relabeling", async () => {
20
+ const { lookupPiCatalogName } = await import("../lib/runtime-metrics-pi-identity.ts");
21
+ const { OPENAI_CODEX_MODELS } = await import("@earendil-works/pi-ai/providers/openai-codex.models");
22
+ const model = Object.values(OPENAI_CODEX_MODELS)[0];
23
+ await lookupPiCatalogName({ provider: model.provider, modelId: model.id });
24
+ const metrics = new RuntimeMetrics();
25
+ assert.equal(recordRaw(metrics, { ...response(), selectedProvider: model.provider,
26
+ selectedModelId: model.id, provider: "anthropic" }), "recorded");
27
+ const [row] = metrics.snapshot();
28
+ assert.equal(row.selectedModelId, model.id);
29
+ assert.equal(row.selectedProvider, model.provider);
30
+ assert.equal(row.provider, "anthropic");
31
+ assert.equal(recordRaw(metrics, { ...response("private"), selectedProvider: "private-provider",
32
+ selectedModelId: model.id }), "recorded");
33
+ assert.equal(metrics.snapshot()[1].selectedProvider, "custom");
34
+ assert.equal(metrics.snapshot()[1].selectedModelId, "custom");
35
+ });
36
+
37
+ test("mirrored registry identities survive an unavailable Pi catalog", () => {
38
+ const metrics = new RuntimeMetrics({ classifyModel: () => ({ classification: "unknown", modelId: "unknown" }) });
39
+ assert.equal(metrics.record({ ...response(), selectedProvider: "openai-codex", selectedModelId: "gpt-5.6-terra",
40
+ responseModelId: "gpt-5.6-sol" }), "recorded");
41
+ const [row] = metrics.snapshot();
42
+ assert.equal(row.selectedModelId, "gpt-5.6-terra");
43
+ assert.equal(row.responseModelId, "gpt-5.6-sol");
44
+ });
45
+
46
+ // Deliberately bypass static types to exercise the runtime boundary.
47
+ function recordRaw(metrics: RuntimeMetrics, value: unknown) {
48
+ return metrics.record(value as FinalResponse);
49
+ }
50
+
51
+ test("optional native token fields preserve absence but reject explicit malformed values", () => {
52
+ for (const value of [null, -1, { state: "reported", value: NaN }, { state: "reported", value: 1_000_000_001 }]) {
53
+ assert.equal(recordRaw(new RuntimeMetrics(), { ...response(), tokens: { ...response().tokens, reasoning: value } }), "invalid");
54
+ }
55
+ const metrics = new RuntimeMetrics();
56
+ assert.equal(metrics.record(response()), "recorded");
57
+ assert.equal(metrics.snapshot()[0].tokens.reasoning.unavailable, 1);
58
+ });
59
+
60
+ test("accounts final responses with explicit zero, missing states and separate durations", () => {
61
+ const metrics = new RuntimeMetrics();
62
+ assert.equal(metrics.record(response()), "recorded");
63
+ assert.equal(metrics.record(response("second")), "recorded");
64
+ const [row] = metrics.snapshot();
65
+ assert.equal(row.responses, 2);
66
+ assert.deepEqual(row.tokens.input, { reported: 2, unavailable: 0, unsupported: 0, sum: 0 });
67
+ assert.equal(row.tokens.output.sum, 24);
68
+ assert.equal(row.tokens.cacheRead.unavailable, 2);
69
+ assert.equal(row.tokens.cacheWrite.unsupported, 2);
70
+ assert.equal(row.responseHeadersMs.sum, 51);
71
+ assert.equal(row.fullResponseMs.sum, 200);
72
+ assert.equal(row.executor, "worker");
73
+ });
74
+
75
+ test("deduplicates distinct local IDs without retaining them in snapshots", () => {
76
+ const metrics = new RuntimeMetrics();
77
+ assert.equal(metrics.record(response()), "recorded");
78
+ assert.equal(metrics.record({ ...response(), effort: "low" }), "duplicate");
79
+ assert.equal(metrics.record(response("second")), "recorded");
80
+ assert.equal(metrics.snapshot()[0].responses, 2);
81
+ assert.ok(!JSON.stringify(metrics.snapshot()).includes("local-response"));
82
+ const copy = metrics.snapshot();
83
+ copy[0].tokens.output.sum = 999;
84
+ assert.equal(metrics.snapshot()[0].tokens.output.sum, 24);
85
+ });
86
+
87
+ test("effort selections and absence states remain distinct", () => {
88
+ const metrics = new RuntimeMetrics();
89
+ const levels = ["off", "minimal", "low", "medium", "high", "xhigh", "max",
90
+ "not_selected", "unsupported", "unavailable"] as const;
91
+ for (const effort of levels) assert.equal(metrics.record({ ...response(effort), effort }), "recorded");
92
+ assert.deepEqual(metrics.snapshot().map(row => row.effort), levels);
93
+ assert.equal(recordRaw(metrics, { ...response("bad"), effort: "ultra" }), "invalid");
94
+ });
95
+
96
+ test("only closed dimensions survive; no prompt-based executor inference", () => {
97
+ const metrics = new RuntimeMetrics();
98
+ assert.equal(recordRaw(metrics, {
99
+ ...response(), provider: "private-provider", modelFamily: "private-model-id",
100
+ executor: "SDD apply executor", systemPrompt: "reviewer", errorMessage: "private failure",
101
+ }), "recorded");
102
+ const [row] = metrics.snapshot();
103
+ assert.equal(row.provider, "custom");
104
+ assert.equal(row.modelFamily, "custom");
105
+ assert.equal(row.executor, "unknown");
106
+ assert.ok(!JSON.stringify(row).includes("private"));
107
+ assert.equal(recordRaw(metrics, { ...response("missing"), provider: undefined, modelFamily: null }), "recorded");
108
+ assert.equal(metrics.snapshot()[1].provider, "unknown");
109
+ assert.equal(metrics.snapshot()[1].modelFamily, "unknown");
110
+ });
111
+
112
+ test("rejects malformed numbers and ambiguous measurement states atomically", () => {
113
+ const metrics = new RuntimeMetrics();
114
+ for (const value of [-1, NaN, Infinity, "12", null, 1e20]) {
115
+ const base = response();
116
+ assert.equal(recordRaw(metrics, { ...base, tokens: { ...base.tokens, input: { state: "reported", value } } }), "invalid");
117
+ assert.equal(recordRaw(metrics, { ...base, fullResponseMs: { state: "measured", value } }), "invalid");
118
+ }
119
+ for (const measurement of [{ value: 0 }, { state: "reported" }, { state: "unsupported", value: 0 }]) {
120
+ const base = response();
121
+ assert.equal(recordRaw(metrics, { ...base, tokens: { ...base.tokens, input: measurement } }), "invalid");
122
+ }
123
+ assert.equal(recordRaw(metrics, { ...response(), tokens: { ...response().tokens, output: { state: "reported", value: 1.5 } } }), "invalid");
124
+ assert.equal(recordRaw(metrics, { ...response(), fullResponseMs: { state: "reported", value: 10 } }), "invalid");
125
+ assert.equal(recordRaw(metrics, { ...response(), fullResponseMs: { state: "measured", value: 10 } }), "invalid");
126
+ assert.deepEqual(metrics.snapshot(), []);
127
+ assert.equal(metrics.record(response()), "recorded", "invalid inputs do not reserve IDs");
128
+ });
129
+
130
+ test("rejects streaming, malformed records, identifiers and free-text errors", () => {
131
+ const metrics = new RuntimeMetrics();
132
+ for (const value of [null, {}, [], { ...response(), kind: "message_update" },
133
+ { ...response(), responseId: "" }, { ...response(), responseId: "x".repeat(129) },
134
+ { ...response(), error: "raw error text" }, { ...response(), tokens: null }]) {
135
+ assert.equal(recordRaw(metrics, value), "invalid");
136
+ }
137
+ for (const error of ["none", "aborted", "rate_limit", "authentication", "network", "provider", "unknown"] as const) {
138
+ assert.equal(metrics.record({ ...response(error), error, responseHeadersMs: { state: "unsupported" }, fullResponseMs: { state: "unavailable" } }), "recorded");
139
+ }
140
+ assert.equal(metrics.snapshot().length, 7);
141
+ });
142
+
143
+ test("host identity stays separate from role and rejects invented model dimensions", () => {
144
+ const metrics = new RuntimeMetrics();
145
+ assert.equal(recordRaw(metrics, { ...response("custom"), identity: { origin: "custom" } }), "recorded");
146
+ assert.equal(recordRaw(metrics, { ...response("forged"), identity: { hostAgent: "private-worker", modelId: "secret-model" } }), "recorded");
147
+ const rows = metrics.snapshot();
148
+ assert.deepEqual(rows.map(row => [row.hostAgent, row.executor, row.selectedModelId, row.effort]), [
149
+ ["pi", "worker", "unknown", "high"],
150
+ ]);
151
+ assert.ok(!JSON.stringify(rows).includes("secret-model"));
152
+ });
153
+
154
+ test("hard response and numeric ceilings keep aggregate totals bounded", () => {
155
+ const metrics = new RuntimeMetrics();
156
+ for (let i = 0; i < 1024; i++) {
157
+ const record = response(String(i));
158
+ record.tokens.input = { state: "reported", value: 1_000_000_000 };
159
+ record.responseHeadersMs = { state: "measured", value: 0 };
160
+ record.fullResponseMs = { state: "measured", value: 86_400_000 };
161
+ assert.equal(metrics.record(record), "recorded");
162
+ }
163
+ assert.equal(metrics.record(response("overflow")), "capacity");
164
+ const [row] = metrics.snapshot();
165
+ assert.equal(row.responses, 1024);
166
+ assert.equal(row.tokens.input.sum, 1_024_000_000_000);
167
+ assert.equal(row.fullResponseMs.sum, 88_473_600_000);
168
+ assert.equal(row.responseHeadersMs.measured, 1024);
169
+ assert.equal(row.responseHeadersMs.sum, 0);
170
+ });
171
+
172
+ test("capacity rejection keeps dedupe and accounting intact without eviction", () => {
173
+ const metrics = new RuntimeMetrics({ maxResponses: 2, maxBuckets: 1 });
174
+ assert.equal(metrics.record(response("one")), "recorded");
175
+ assert.equal(metrics.record({ ...response("two"), effort: "low" }), "capacity");
176
+ assert.equal(metrics.record(response("two")), "recorded");
177
+ assert.equal(metrics.record(response("three")), "capacity");
178
+ assert.equal(metrics.record(response("one")), "duplicate");
179
+ assert.equal(metrics.snapshot()[0].responses, 2);
180
+ for (const maxResponses of [0, -1, 1.5, Infinity, 1025]) {
181
+ assert.throws(() => new RuntimeMetrics({ maxResponses }), RangeError);
182
+ }
183
+ assert.throws(() => new RuntimeMetrics({ maxBuckets: 65 }), RangeError);
184
+ });
@@ -67,7 +67,7 @@ const requiredToolsByAgent: Record<string, string[]> = {
67
67
  "sdd-init.md": ["read", "grep", "find", "edit", "write", "bash", "mem_search", "mem_get_observation", "mem_save", "mem_update"],
68
68
  "sdd-onboard.md": ["read", "grep", "find", "edit", "write", "bash", "mem_search", "mem_get_observation", "mem_save", "mem_update"],
69
69
  "sdd-proposal.md": ["read", "grep", "find", "edit", "write", "mem_search", "mem_get_observation", "mem_save"],
70
- "sdd-research.md": ["read", "grep", "find", "edit", "write", "mem_search", "mem_get_observation", "mem_save"],
70
+ "sdd-research.md": ["read", "grep", "find", "edit", "write", "mem_search", "mem_get_observation", "mem_save", "fetch_content", "web_search", "source_check", "get_search_content"],
71
71
  "sdd-spec.md": ["read", "grep", "find", "edit", "write", "mem_search", "mem_get_observation", "mem_save"],
72
72
  "sdd-status.md": ["read", "grep", "find", "bash", "mem_search", "mem_get_observation"],
73
73
  "sdd-sync.md": ["read", "grep", "find", "edit", "write", "bash", "mem_search", "mem_get_observation", "mem_save", "mem_update"],
@@ -103,6 +103,15 @@ test("artifact-producing SDD agents can persist OpenSpec files while status rema
103
103
  assert.ok(!statusTools.includes("write"), "sdd-status.md must remain read-only");
104
104
  });
105
105
 
106
+ test("research instructions require executed evidence rather than blanket denial", () => {
107
+ const source = readFileSync(join(assetsAgentsDir, "sdd-research.md"), "utf8");
108
+ assert.doesNotMatch(source, /documentation=\[\]; open-web=\[\]/);
109
+ assert.match(source, /Actually call approved tools/);
110
+ assert.match(source, /claim maps to source IDs/);
111
+ assert.match(source, /proposal_ready: false/);
112
+ assert.ok(!readTools(join(assetsAgentsDir, "sdd-research.md")).includes("bash"));
113
+ });
114
+
106
115
  test("project does not ship local SDD agent overrides", () => {
107
116
  for (const relativeDir of [join(".pi", "agents"), join(".pi", "subagents")]) {
108
117
  const dir = join(repoRoot, relativeDir);
@@ -0,0 +1,28 @@
1
+ import assert from "node:assert/strict";
2
+ import { readFileSync } from "node:fs";
3
+ import test from "node:test";
4
+
5
+ const guidancePaths = [
6
+ "assets/support/sdd-status-contract.md",
7
+ "assets/sdd-orchestrator-workflow.md",
8
+ "assets/agents/sdd-status.md",
9
+ "assets/agents/sdd-verify.md",
10
+ "assets/agents/sdd-archive.md",
11
+ ];
12
+
13
+ for (const path of guidancePaths) {
14
+ test(`${path}: native v2 status remains read-only and authoritative`, () => {
15
+ const guidance = readFileSync(new URL(`../${path}`, import.meta.url), "utf8");
16
+ assert.match(guidance, /native.*(?:status|v2)|gentle-ai\.sdd-status/i);
17
+ assert.match(guidance, /read-only/i);
18
+ assert.doesNotMatch(guidance, /resolve-via-engram/i);
19
+ assert.doesNotMatch(guidance, /local SDD status engine|manual (?:fallback )?status|reconstruct(?:ing)? (?:native )?status/i);
20
+ });
21
+ }
22
+
23
+ test("workflow preserves explicit continuation and the manual sync resolver", () => {
24
+ const workflow = readFileSync(new URL("../assets/sdd-orchestrator-workflow.md", import.meta.url), "utf8");
25
+ assert.match(workflow, /only.*sdd-continue|sdd-continue.*only/i);
26
+ assert.match(workflow, /manual sdd-sync/i);
27
+ assert.doesNotMatch(workflow, /sdd-(?:apply|verify|archive).*local|local.*sdd-(?:apply|verify|archive)/i);
28
+ });