@pi-in-go/pigpen-pi-typesafe 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (189) hide show
  1. package/CREDITS.md +14 -0
  2. package/LICENSE +22 -0
  3. package/README.md +45 -0
  4. package/extensions/pi-typesafe/branches_test.go +185 -0
  5. package/extensions/pi-typesafe/command.go +319 -0
  6. package/extensions/pi-typesafe/export_test.go +9 -0
  7. package/extensions/pi-typesafe/extension.go +188 -0
  8. package/extensions/pi-typesafe/extension_test.go +321 -0
  9. package/extensions/pi-typesafe/fakehost_test.go +548 -0
  10. package/extensions/pi-typesafe/format.go +191 -0
  11. package/extensions/pi-typesafe/format_test.go +75 -0
  12. package/extensions/pi-typesafe/go.mod +10 -0
  13. package/extensions/pi-typesafe/go.sum +2 -0
  14. package/extensions/pi-typesafe/go.work +11 -0
  15. package/extensions/pi-typesafe/harness_test.go +200 -0
  16. package/extensions/pi-typesafe/ownmodel_test.go +100 -0
  17. package/extensions/pi-typesafe/review_test.go +134 -0
  18. package/extensions/pi-typesafe/tool.go +193 -0
  19. package/extensions/pi-typesafe/twin_test.go +28 -0
  20. package/libs/pi-typesafe-api/CREDITS.md +14 -0
  21. package/libs/pi-typesafe-api/LICENSE +22 -0
  22. package/libs/pi-typesafe-api/README.md +30 -0
  23. package/libs/pi-typesafe-api/ask.go +62 -0
  24. package/libs/pi-typesafe-api/ask_test.go +76 -0
  25. package/libs/pi-typesafe-api/auth.go +249 -0
  26. package/libs/pi-typesafe-api/auth_test.go +131 -0
  27. package/libs/pi-typesafe-api/backends.go +336 -0
  28. package/libs/pi-typesafe-api/backends_test.go +404 -0
  29. package/libs/pi-typesafe-api/batch.go +202 -0
  30. package/libs/pi-typesafe-api/batch_test.go +202 -0
  31. package/libs/pi-typesafe-api/battery_test.go +41 -0
  32. package/libs/pi-typesafe-api/calibrate.go +354 -0
  33. package/libs/pi-typesafe-api/calibrate_test.go +186 -0
  34. package/libs/pi-typesafe-api/client.go +615 -0
  35. package/libs/pi-typesafe-api/client_test.go +490 -0
  36. package/libs/pi-typesafe-api/credentials.go +252 -0
  37. package/libs/pi-typesafe-api/credentials_test.go +216 -0
  38. package/libs/pi-typesafe-api/doc.go +14 -0
  39. package/libs/pi-typesafe-api/errors.go +143 -0
  40. package/libs/pi-typesafe-api/evaluation.go +86 -0
  41. package/libs/pi-typesafe-api/evaluation_schema.json +264 -0
  42. package/libs/pi-typesafe-api/gaps_test.go +77 -0
  43. package/libs/pi-typesafe-api/go.mod +9 -0
  44. package/libs/pi-typesafe-api/go.sum +2 -0
  45. package/libs/pi-typesafe-api/helpers_test.go +169 -0
  46. package/libs/pi-typesafe-api/hostmodel/hostmodel.go +87 -0
  47. package/libs/pi-typesafe-api/json.go +299 -0
  48. package/libs/pi-typesafe-api/json_test.go +92 -0
  49. package/libs/pi-typesafe-api/ownmodel_test.go +79 -0
  50. package/libs/pi-typesafe-api/package.json +40 -0
  51. package/libs/pi-typesafe-api/provenance.json +18 -0
  52. package/libs/pi-typesafe-api/review_test.go +23 -0
  53. package/libs/pi-typesafe-api/schema.go +473 -0
  54. package/libs/pi-typesafe-api/schema_test.go +262 -0
  55. package/libs/pi-typesafe-api/testdata/tools/typebox-messages.mts +5 -0
  56. package/libs/pi-typesafe-api/testdata/typebox-messages.json +285 -0
  57. package/libs/pi-typesafe-api/twin_test.go +28 -0
  58. package/libs/pi-typesafe-api/ui/fakehost_test.go +548 -0
  59. package/libs/pi-typesafe-api/ui/keyprompt.go +115 -0
  60. package/libs/pi-typesafe-api/ui/login.go +106 -0
  61. package/libs/pi-typesafe-api/ui/twin_test.go +28 -0
  62. package/libs/pi-typesafe-api/ui/ui_test.go +285 -0
  63. package/libs/pi-typesafe-api/usage.go +366 -0
  64. package/libs/pi-typesafe-api/usage_test.go +139 -0
  65. package/libs/typesafe/CONTRACT.md +125 -0
  66. package/libs/typesafe/CREDITS.md +37 -0
  67. package/libs/typesafe/LICENSE +23 -0
  68. package/libs/typesafe/README.md +19 -0
  69. package/libs/typesafe/go.mod +3 -0
  70. package/libs/typesafe/libraries/ownmodel/backend_test.go +496 -0
  71. package/libs/typesafe/libraries/ownmodel/canon.go +190 -0
  72. package/libs/typesafe/libraries/ownmodel/convert.go +199 -0
  73. package/libs/typesafe/libraries/ownmodel/doc.go +15 -0
  74. package/libs/typesafe/libraries/ownmodel/equivalence_test.go +199 -0
  75. package/libs/typesafe/libraries/ownmodel/helpers_test.go +155 -0
  76. package/libs/typesafe/libraries/ownmodel/mutation_test.go +31 -0
  77. package/libs/typesafe/libraries/ownmodel/ownmodel.go +225 -0
  78. package/libs/typesafe/libraries/ownmodel/plan.go +442 -0
  79. package/libs/typesafe/libraries/ownmodel/run.go +288 -0
  80. package/libs/typesafe/libraries/ownmodel/schema_test.go +254 -0
  81. package/libs/typesafe/libraries/ownmodel/twins_test.go +169 -0
  82. package/libs/typesafe/libraries/ownmodel/utils_test.go +125 -0
  83. package/libs/typesafe/libraries/pigmodel/pigmodel.go +264 -0
  84. package/libs/typesafe/libraries/pigmodel/pigmodel_test.go +410 -0
  85. package/libs/typesafe/libraries/typesafe/answers.go +268 -0
  86. package/libs/typesafe/libraries/typesafe/api_response_test.go +113 -0
  87. package/libs/typesafe/libraries/typesafe/batch.go +80 -0
  88. package/libs/typesafe/libraries/typesafe/batch_test.go +133 -0
  89. package/libs/typesafe/libraries/typesafe/bench_test.go +71 -0
  90. package/libs/typesafe/libraries/typesafe/client.go +561 -0
  91. package/libs/typesafe/libraries/typesafe/client_test.go +495 -0
  92. package/libs/typesafe/libraries/typesafe/crosscheck_test.go +464 -0
  93. package/libs/typesafe/libraries/typesafe/crosscheck_workflowevals_test.go +219 -0
  94. package/libs/typesafe/libraries/typesafe/doc.go +27 -0
  95. package/libs/typesafe/libraries/typesafe/entry.go +142 -0
  96. package/libs/typesafe/libraries/typesafe/env.go +11 -0
  97. package/libs/typesafe/libraries/typesafe/errors.go +310 -0
  98. package/libs/typesafe/libraries/typesafe/errors_test.go +175 -0
  99. package/libs/typesafe/libraries/typesafe/helpers_test.go +294 -0
  100. package/libs/typesafe/libraries/typesafe/live_test.go +96 -0
  101. package/libs/typesafe/libraries/typesafe/logging.go +160 -0
  102. package/libs/typesafe/libraries/typesafe/logging_test.go +259 -0
  103. package/libs/typesafe/libraries/typesafe/marshal_test.go +112 -0
  104. package/libs/typesafe/libraries/typesafe/mutation_test.go +39 -0
  105. package/libs/typesafe/libraries/typesafe/questions.go +490 -0
  106. package/libs/typesafe/libraries/typesafe/questions_test.go +166 -0
  107. package/libs/typesafe/libraries/typesafe/regressions_test.go +159 -0
  108. package/libs/typesafe/libraries/typesafe/reliability_test.go +649 -0
  109. package/libs/typesafe/libraries/typesafe/retry.go +350 -0
  110. package/libs/typesafe/libraries/typesafe/retry_test.go +297 -0
  111. package/libs/typesafe/libraries/typesafe/runtime_test.go +26 -0
  112. package/libs/typesafe/libraries/typesafe/transport_test.go +163 -0
  113. package/libs/typesafe/libraries/typesafe/twins_test.go +127 -0
  114. package/libs/typesafe/libraries/typesafe/types_test.go +165 -0
  115. package/libs/typesafe/libraries/typesafe/version.go +10 -0
  116. package/libs/typesafe/package.json +37 -0
  117. package/libs/typesafe/provenance.json +49 -0
  118. package/package.json +42 -0
  119. package/port/PORT.md +98 -0
  120. package/port/accepted-gaps.json +3 -0
  121. package/port/golden/enable-confirm.jsonl +11 -0
  122. package/port/golden/enable-decline.jsonl +20 -0
  123. package/port/golden/enable-missing-key.jsonl +4 -0
  124. package/port/golden/login-shadow.jsonl +4 -0
  125. package/port/golden/logout-env-key.jsonl +6 -0
  126. package/port/golden/playground-cancel.jsonl +4 -0
  127. package/port/golden/playground-invalid-json.jsonl +5 -0
  128. package/port/golden/playground-invalid-questions.jsonl +5 -0
  129. package/port/golden/status-env-key.jsonl +6 -0
  130. package/port/golden/status-no-key.jsonl +6 -0
  131. package/port/golden/tool-disabled.jsonl +18 -0
  132. package/port/golden/trailing-words.jsonl +10 -0
  133. package/port/library-mutations.py +44 -0
  134. package/port/mutations.json +302 -0
  135. package/port/oracle/.env.example +4 -0
  136. package/port/oracle/CHANGELOG.md +91 -0
  137. package/port/oracle/CONTRIBUTING.md +35 -0
  138. package/port/oracle/LICENSE +21 -0
  139. package/port/oracle/README.md +159 -0
  140. package/port/oracle/docs/api.md +143 -0
  141. package/port/oracle/docs/ci-cd.md +97 -0
  142. package/port/oracle/examples/decision-extension.ts +41 -0
  143. package/port/oracle/extensions/index.js +2 -0
  144. package/port/oracle/package.json +89 -0
  145. package/port/oracle/scripts/dev-pi.mjs +23 -0
  146. package/port/oracle/scripts/live-smoke.mjs +35 -0
  147. package/port/oracle/src/ask.ts +42 -0
  148. package/port/oracle/src/auth.ts +171 -0
  149. package/port/oracle/src/backends.ts +196 -0
  150. package/port/oracle/src/batch.ts +170 -0
  151. package/port/oracle/src/calibrate.ts +237 -0
  152. package/port/oracle/src/client.ts +310 -0
  153. package/port/oracle/src/credentials.ts +136 -0
  154. package/port/oracle/src/errors.ts +53 -0
  155. package/port/oracle/src/extension.ts +204 -0
  156. package/port/oracle/src/index.ts +31 -0
  157. package/port/oracle/src/key-prompt.ts +51 -0
  158. package/port/oracle/src/login.ts +60 -0
  159. package/port/oracle/src/schema.ts +158 -0
  160. package/port/oracle/src/ui.ts +4 -0
  161. package/port/oracle/src/usage.ts +258 -0
  162. package/port/oracle/tests/ask.test.ts +63 -0
  163. package/port/oracle/tests/auth.test.ts +141 -0
  164. package/port/oracle/tests/backends.test.ts +380 -0
  165. package/port/oracle/tests/batch.test.ts +156 -0
  166. package/port/oracle/tests/calibrate.test.ts +144 -0
  167. package/port/oracle/tests/client.test.ts +499 -0
  168. package/port/oracle/tests/credentials.test.ts +144 -0
  169. package/port/oracle/tests/extension.test.ts +276 -0
  170. package/port/oracle/tests/key-prompt.test.ts +47 -0
  171. package/port/oracle/tests/login.test.ts +101 -0
  172. package/port/oracle/tests/schema.test.ts +85 -0
  173. package/port/oracle/tests/usage.test.ts +106 -0
  174. package/port/oracle/tsconfig.build.json +10 -0
  175. package/port/oracle/tsconfig.json +14 -0
  176. package/port/scenarios/enable-confirm.json +5 -0
  177. package/port/scenarios/enable-decline.json +3 -0
  178. package/port/scenarios/enable-missing-key.json +2 -0
  179. package/port/scenarios/login-shadow.json +2 -0
  180. package/port/scenarios/logout-env-key.json +3 -0
  181. package/port/scenarios/playground-cancel.json +2 -0
  182. package/port/scenarios/playground-invalid-json.json +2 -0
  183. package/port/scenarios/playground-invalid-questions.json +2 -0
  184. package/port/scenarios/status-env-key.json +3 -0
  185. package/port/scenarios/status-no-key.json +3 -0
  186. package/port/scenarios/tool-disabled.json +2 -0
  187. package/port/scenarios/trailing-words.json +5 -0
  188. package/port/upstream-tests.json +160 -0
  189. package/provenance.json +18 -0
@@ -0,0 +1,143 @@
1
+ # API
2
+
3
+ Everything `pi-typesafe` exports, for extension authors. The library has no dependency on Pi's runtime and is safe in tests. The README covers the tool, the commands, and how to write questions.
4
+
5
+ ## The client
6
+
7
+ ```ts
8
+ import { createTypeSafe, choice, noul, score } from "pi-typesafe";
9
+
10
+ const typesafe = createTypeSafe({ maxRequests: 5, maxUsdPerDay: 1 });
11
+ const result = await typesafe.evaluate({
12
+ state: { title: "Login fails after update", body: "..." },
13
+ questions: {
14
+ area: choice("Which area does this report concern?", { auth: "Sign-in", ui: "Layout", other: null }),
15
+ duplicate: noul("Does the report describe the same defect as `known_issue`?"),
16
+ severity: score("How severe is the defect?", ["Cosmetic", "Workaround exists", "Blocking"]),
17
+ },
18
+ });
19
+ result.answers.area.choice; // "auth" | "ui" | "other"
20
+ result.answers.duplicate.noul; // 0..1
21
+ result.answers.severity.score; // 0..2, may be fractional
22
+ ```
23
+
24
+ | Option | Default | Meaning |
25
+ | --- | --- | --- |
26
+ | `apiKey` | the backend's key (below) | Never returned |
27
+ | `backend` | `typesafe` | `typesafe`, `openrouter`, `commandcode`, or a caller-supplied endpoint object; picks the host, the request path, the default model, and the key |
28
+ | `model` | `jev-latest` (`typesafe/jev-1.13` on OpenRouter, `typesafe/jev` on Command Code) | No model is inferred from submitted content |
29
+ | `timeoutMs` | `15000` | Per request; no automatic retries |
30
+ | `maxInputBytes` | `65536` | UTF-8 JSON bytes, not tokens |
31
+ | `maxRequests` | `20` | Attempts per client instance, failures included |
32
+ | `maxRequestsPerDay`, `maxInputTokensPerDay`, `maxUsdPerDay` | none | Local-day caps, persisted |
33
+ | `usdPerMTok` | `0.042` | Price used for the estimate and the USD cap |
34
+ | `ledger` | the store next to the key | Inject a ledger in tests |
35
+ | `fetch` | global fetch | Inject a transport for offline tests |
36
+
37
+ A `model` is mapped to the backend's own id form before it is sent: on OpenRouter a bare `jev-latest` goes as `~typesafe/jev-latest` and a bare `jev-1.13` (or `jev-1.13.0`) as `typesafe/jev-1.13`, while an id that already carries an author, such as `vendor/other`, passes unchanged, and TypeSafe sends ids as written. The same mapping applies to a per-request `model` inside `evaluate()`; the limits of 1–100 characters apply to your own id, before mapping. A caller-supplied endpoint's model is never mapped: `defaultModel` and a per-request `model` go on the wire as the caller wrote them.
38
+
39
+ `DECISIONS_BACKENDS` is the registry behind `backend`: `typesafe`, `openrouter`, and `commandcode`. Command Code serves the same Jev decisions protocol at `api.commandcode.ai` under `/provider/v1/systemone`, with the model id `typesafe/jev`; its model list is public, under `/provider/v1/models`, and arrives in `data` with ids in `id`. Each entry carries `label`, `host`, `keyEnv`, and, when the service does not serve the SDK's own paths, `path` for the judgment request plus `modelsPath`, `modelsField`, and `modelsIdField` for the model list — OpenRouter's and Command Code's lists are renamed to the `models` the SDK reads, with each entry's `id` promoted to the `name` that `listModels()` returns. `modelsVerifyKey: false` marks a backend whose model list is public, and therefore proves nothing about the key: both OpenRouter and Command Code serve theirs without checking one, so `listModels()` there leaves the auth state unverified while the answer check is unchanged and a malformed reply stays a `response` error. `DEFAULT_BACKEND` is `"typesafe"`.
40
+
41
+ `backend` also accepts a caller-supplied endpoint object (`BackendEndpoint`) wherever a backend name is accepted — `createTypeSafe`, `keySituation`, `resolveApiKey`, `authState`, `ensureApiKey`, and `safeError`. An endpoint names `label`, `host`, and `keyEnv`, and optionally `path`, `modelsPath`, `modelsField`, `modelsIdField`, `modelsVerifyKey`, and `defaultModel`; it is validated on every call, never added to the registry, and without `defaultModel` requires `model` on `createTypeSafe`. `resolveBackend(nameOrEndpoint)` resolves either form to the validated `ResolvedBackend` the client uses (registry `name` when there is one, `host` as origin only, explicit `keyEnv` and `modelsVerifyKey`), and `backendHost(nameOrEndpoint)` returns just the destination host — `api.commandcode.ai`, `gw.example.com:8443` — for consent text. Validation is in this order, and every failure is a `configuration` error whose message never quotes a caller value (a host can carry credentials in its user info), except the label after it is validated:
42
+
43
+ | Rule | Refusal |
44
+ | --- | --- |
45
+ | `label` is a string, trimmed nonempty, at most 60 characters | `Backend label must be a nonempty string of at most 60 characters.` |
46
+ | `host` is an absolute `https:` URL with no user info, path, query, or fragment (`http:` only for a loopback host) | `Backend host must be an absolute https: URL with no user info, path, query, or fragment …` |
47
+ | `path`, when present, starts with `"/"` and carries no `?` or `#` | `Backend path must be a string that starts with "/".` |
48
+ | `modelsPath`, same rule | `Backend modelsPath must be a string that starts with "/".` |
49
+ | `modelsField` / `modelsIdField`, when present, nonempty strings | `Backend modelsField must be a nonempty string.` / `Backend modelsIdField must be a nonempty string.` |
50
+ | `modelsVerifyKey`, when present, a boolean | `Backend modelsVerifyKey must be a boolean.` |
51
+ | `keyEnv` is a name of letters, digits, and underscores, not starting with a digit | `Backend keyEnv must name an environment variable: letters, digits, and underscores, not starting with a digit.` |
52
+ | `keyEnv` is not `TYPESAFE_API_KEY` in any case | `Backend keyEnv must not be TYPESAFE_API_KEY: the TypeSafe key is only sent to the typesafe backend. Give this endpoint its own variable.` |
53
+ | `defaultModel`, when present, trimmed nonempty, at most 100 characters | `Backend defaultModel must be a nonempty string of at most 100 characters.` |
54
+
55
+ Anything that is neither a registry name nor such an object is refused with `backend must be a registry name or a backend object.`
56
+
57
+ The TypeSafe backend takes its key from `TYPESAFE_API_KEY`, then the `/typesafe login` store. Every other backend — registry or endpoint — reads only its own `keyEnv` variable: the store holds a TypeSafe key, and a login verifies against api.typesafe.ai, so neither applies elsewhere, and the TypeSafe key is only ever sent to the typesafe backend. `createTypeSafe` with no `apiKey` and no key in the backend's variable fails with `No API key. Set <keyEnv> in the environment.` before any request is built. Pass the same `backend` to `authState`, `keySituation`, and `ensureApiKey` so what you report matches what you send.
58
+
59
+ `evaluate(request, { signal })` validates before sending and rejects with `TypeSafeIntegrationError`. `code` is one of `configuration`, `validation`, `budget`, `aborted`, `timeout`, `http`, `connection`, `response`; messages never contain upstream bodies, keys, or your submitted state, and no header value except a numeric `Retry-After` count in seconds (quoted by the 429 advice as `Retry after <n> seconds.`). The advice is backend-aware: a 401 names the backend's own key variable (`Check TYPESAFE_API_KEY.`, `Check OPENROUTER_API_KEY.`, `Check COMMANDCODE_API_KEY.`, or `Check <keyEnv>.` for an endpoint), and a 402 says `Check your account balance.` except on OpenRouter, which says `Insufficient credits. Add credits at https://openrouter.ai/credits.` `listModels()` verifies the key without counting toward `maxRequests`, except on a backend whose model list is public (`modelsVerifyKey: false`, or an endpoint that does not set `modelsVerifyKey: true`), which accepts any key and leaves the auth state unverified.
60
+
61
+ ## Admission
62
+
63
+ `prepareEvaluationRequest(value, { maxInputBytes })` is the one admission rule, used by the tool, the playground, and `evaluate`. It normalizes the near-miss aliases a model produces (`options` / `levels` / `choices` for `criteria`, a string Noul criterion, a label array for a Choice), validates the schema and JSON-safety, then enforces the byte budget. `DEFAULT_MAX_INPUT_BYTES`, `DEFAULT_MAX_QUESTIONS`, and `DEFAULT_MAX_REQUESTS` hold the shared defaults.
64
+
65
+ ## Batching
66
+
67
+ `evaluate` is one request: up to 32 questions about one state. Both batching calls preserve input order, bound concurrency (`concurrency`, default 4), never throw, and stop submitting once a `budget` or cancellation failure appears.
68
+
69
+ | Call | Use |
70
+ | --- | --- |
71
+ | `evaluateAll(request)` | One state, any number of questions: chunks over 32 share the state, then merge into one `answers` map with usage summed |
72
+ | `evaluateMany(requests)` | Several requests: per-request results plus merged answers, `failures`, `skipped` |
73
+ | `chunkEvaluationRequest(request, { maxQuestions })` | The splitter alone; a pure function, no validation |
74
+ | `fanOut(items, worker, { concurrency, signal, stopOn })` | The pool underneath, for your own work |
75
+
76
+ Every item comes back as `{ ok: true, index, value }` or `{ ok: false, index, error, skipped }`; `skipped` marks work that was never submitted.
77
+
78
+ ## Usage and spend
79
+
80
+ `getUsage()` returns this client's session counters (`requestsStarted`, `requestsSucceeded`, `requestsFailed`, `inputTokens`, `outputTokens`, `estimatedUsd`). `getSpend()` adds today's persisted totals, the caps in force, and the cap currently reached.
81
+
82
+ Day caps live in `~/.pi/agent/pi-typesafe/usage.json` (owner-only, atomic, best-effort: an unwritable ledger never fails a request) and roll over at local midnight.
83
+
84
+ | Option | Environment | Bounds |
85
+ | --- | --- | --- |
86
+ | `maxRequestsPerDay` | `PI_TYPESAFE_MAX_REQUESTS_PER_DAY` | requests |
87
+ | `maxInputTokensPerDay` | `PI_TYPESAFE_MAX_INPUT_TOKENS_PER_DAY` | input tokens |
88
+ | `maxUsdPerDay` | `PI_TYPESAFE_MAX_USD_PER_DAY` | estimated spend |
89
+
90
+ The environment may lower an explicit cap, never raise it. A reached cap raises a `budget` error that names the cap, the amount used, and the day, before anything is submitted. Cost is estimated from input tokens only, because output is free.
91
+
92
+ `openUsageLedger(options)`, `usagePath()`, `estimateUsd(tokens, usdPerMTok)`, `capsFromEnvironment(env)`, and `mergeCaps(explicit, environment)` expose the same arithmetic for your own display.
93
+
94
+ ## Auth state
95
+
96
+ `authState({ backend })` never throws for a valid backend. An invalid backend — an unknown name or an endpoint object that fails validation — throws the same `configuration` error as `resolveBackend`, so validate a user-supplied endpoint with `resolveBackend` first. It reports `backend` (the value you passed, `typesafe` by default — a name or your endpoint object), `kind` (`environment`, `stored`, `missing`, `unusable`), `keyName`, `path`, `reason`, `verified`, `verifiedAt`, `lastFailure`, and `usable` — `usable` is false when no key is present or the last authentication outcome was an HTTP 401/403 rejection. Name the backend you pass to `createTypeSafe`, or the report describes a key you do not send. The verification and failure record is one file shared by every backend, so after switching backends the last outcome stands until the next request: a 401 recorded while one backend is in use makes every backend's `authState` report `usable: false`.
97
+
98
+ `describeAuth(state)` turns that into `{ level: "ok" | "warning" | "error", text }` for a status line or a log. The extension calls both at session start and after a rejection, so an enabled-but-unusable setup is never reported as working.
99
+
100
+ `recordAuthVerified()` is called by `listModels()` and by the first successful request; `recordAuthFailure(error)` records what degraded TypeSafe; `clearAuthState()` forgets both, and `/typesafe logout` calls it. `keySituation(backend)` and `keySourceLabel(situation)` remain the lower-level, frozen-for-existing-callers pair, and `resolveApiKey(backend)` the pre-0.4.0 one; the `backend` argument is optional and defaults to `typesafe`. An environment situation names the variable it read in `keyEnv`.
101
+
102
+ ## Asking without throwing
103
+
104
+ ```ts
105
+ import { ask } from "pi-typesafe";
106
+
107
+ const answer = await ask(typesafe, request, { timeoutMs: 5_000, signal: mySignal });
108
+ if (!answer.ok) return { skipped: answer.errorCode === "budget" };
109
+ answer.answers; // typed, plus model, usage, elapsedMs
110
+ ```
111
+
112
+ `ask` merges its deadline into your signal, takes any object with `evaluate` (so tests pass a stub), and never throws: a failure is `{ ok: false, error, errorCode }` with pi-typesafe's own message. Unknown failures become a fixed message, so nothing from the transport reaches the user.
113
+
114
+ ## Calibration: `pi-typesafe/calibrate`
115
+
116
+ A small, domain-free toolkit for turning labelled cases into thresholds.
117
+
118
+ ```ts
119
+ import { calibrate, formatCalibration, replay, samplesOf } from "pi-typesafe/calibrate";
120
+
121
+ const results = await replay(cases, data => scoreOne(data), { concurrency: 6 });
122
+ console.log(formatCalibration(calibrate("action guard", samplesOf(results).samples, { minPrecision: 0.8 })));
123
+ ```
124
+
125
+ | Export | Purpose |
126
+ | --- | --- |
127
+ | `auc(samples)` | Rank-based AUC (Mann–Whitney, ties count half); undefined when one class is empty |
128
+ | `metricsAt(samples, threshold)`, `sweep(samples, thresholds)` | Confusion counts plus precision, recall, and flag rate |
129
+ | `defaultThresholds(samples)`, `pickThreshold(rows, floors)` | The distinct-score grid, and the lowest threshold that clears a precision and recall floor |
130
+ | `calibrate(name, samples, options)`, `formatCalibration(calibration)` | AUC, the sweep, a recommendation, and what it misses and flags, as text |
131
+ | `replay(cases, score, options)`, `samplesOf(results)` | Run labelled cases through any scorer with bounded concurrency, keep per-case failures, then extract the scored samples |
132
+
133
+ `replay` stops on a `budget` failure like the batching calls, and reports each failure with the scorer's own message unless you pass `describeError`.
134
+
135
+ ## Login helpers: `pi-typesafe/ui`
136
+
137
+ `ensureApiKey(ctx, { backend })`, `loginWithPrompt(ctx)`, and `promptForApiKey(ctx)` use the same hidden input as `/typesafe login`. `ensureApiKey(ctx)` returns the existing key source, or prompts, verifies, and stores a new TypeSafe key (`undefined` when the user cancels). For any other backend it returns the environment source or throws `configuration` naming the variable to set; it never opens the prompt, because the prompt verifies against api.typesafe.ai and writes the TypeSafe store. These need Pi's TUI, so call them only from extension command handlers.
138
+
139
+ ## One agent tool
140
+
141
+ `typesafe_evaluate` is the only tool the package registers. It already accepts typed Choice, Score, and Noul questions, including the aliases above, so a separate "ask Jev" tool would duplicate the admission seam and give the model two ways to do one thing. `ask()` is the author-facing half of that seam; both run through `prepareEvaluationRequest`, so what one accepts the others accept.
142
+
143
+ Your extension owns its own user consent and budget; `/typesafe enable` applies only to this package's tool. See [`../examples/decision-extension.ts`](../examples/decision-extension.ts) and [pi-warden](https://github.com/DevMortimer/pi-warden) for a full extension built this way.
@@ -0,0 +1,97 @@
1
+ # CI and continuous delivery
2
+
3
+ The `CI` workflow runs on every pull request to `main`, pushes to `main`, merge
4
+ queue checks, `v*` tags, and manual runs from the Actions page. There are no path
5
+ filters: documentation-only PRs also report the required check. Like pi-warden,
6
+ this project prepares draft releases but does not automatically publish to npm.
7
+
8
+ ## Contributor checks
9
+
10
+ - **Workflow lint** runs actionlint 1.7.12 and ShellCheck on the workflow commands.
11
+ - **Check (ubuntu-latest, Node 22.19.0)** tests the minimum supported Node version.
12
+ - Linux also runs Node **24** and **26**; macOS runs Node **24**. Each job installs
13
+ the lockfile with `npm ci`, then runs `npm run check`: build, typecheck, and
14
+ offline tests. Build runs first because the public-API example resolves this
15
+ package's exports through `dist/`; this also checks the generated declarations
16
+ from a clean checkout. Native Windows is not in this matrix; the existing credential and
17
+ usage tests assert POSIX file permissions.
18
+ - **Package** builds an actual npm tarball and installs it outside the checkout
19
+ with install scripts disabled. It checks root and calibration imports without
20
+ optional Pi peers, verifies all declared JavaScript and type declaration
21
+ files, then checks the extension, UI helpers, and Pi loader with the peer
22
+ versions from `package-lock.json`. It also checks package/lockfile version
23
+ agreement.
24
+ - **CI passed** runs even when another job fails or is skipped, and fails unless
25
+ all three prerequisites succeeded. This is the stable check to require in the
26
+ `main` branch rules; do not require only **Package**, which can be skipped.
27
+
28
+ No TypeSafe or OpenRouter credentials are needed. Tests do not call either
29
+ service. CI does not run `test:live` or any other billable checks. Package
30
+ installation needs access to npm; offline tests themselves use fake transports.
31
+
32
+ Actions use full commit SHAs, checkout does not retain credentials, and check
33
+ jobs have read-only repository permissions. Jobs have time limits; newer PR
34
+ runs cancel superseded runs. Fork PRs use `pull_request`, never
35
+ `pull_request_target`, and do not receive repository secrets. Dependabot opens
36
+ weekly action and dependency updates; they run the same checks and are not
37
+ merged automatically.
38
+
39
+ Run the normal gate locally with:
40
+
41
+ ```bash
42
+ npm ci
43
+ npm run check
44
+ npm pack --dry-run # builds and lists the publishable files
45
+ ```
46
+
47
+ For the complete installed-package smoke test, use the workflow's **Package**
48
+ steps or dispatch `CI` from the Actions page after this workflow is on `main`.
49
+ Successful runs retain an `npm-package` artifact for 14 days. It contains the
50
+ verified tarball and `SHA256SUMS`; tag runs also include release notes.
51
+
52
+ ## Release delivery
53
+
54
+ A tag run must pass the same checks. Before packaging, it must also prove:
55
+
56
+ 1. The tagged commit is an ancestor of `origin/main`.
57
+ 2. The package has a stable `X.Y.Z` version and the tag is exactly `vX.Y.Z`.
58
+ 3. Both root version fields in `package-lock.json` match `package.json`.
59
+ 4. `CHANGELOG.md` has a nonempty `## X.Y.Z` section, not just an HTML comment.
60
+
61
+ Only a **push of a version tag** can start **Draft release**. The job downloads
62
+ that run's checked artifact, verifies the tarball checksum, and creates a draft
63
+ GitHub release with the notes, tarball, and checksum. PRs, branch pushes, merge
64
+ queues, and manual runs cannot create a release. A manual run on a tag can check
65
+ it, but does not publish or create a draft.
66
+
67
+ Only the draft job has `contents: write`. It uses the built-in GitHub token,
68
+ does not check out or execute repository code, and uses `--verify-tag` so it
69
+ cannot create a tag. No npm token or other new repository secret is required.
70
+ An existing release for the tag makes creation fail rather than overwrite it;
71
+ inspect any existing draft and assets before retrying a failed delivery.
72
+
73
+ ## Maintainer steps
74
+
75
+ 1. In GitHub's `main` branch rules, require **CI passed** and PR review. This PR
76
+ does not change repository settings; the checks are not a merge restriction
77
+ until a maintainer enables the rule.
78
+ 2. For a release, update the version with `npm version patch --no-git-tag-version`
79
+ (or the chosen minor version), so `package.json` and `package-lock.json` stay
80
+ aligned. Move the release notes below a matching `## X.Y.Z` heading and leave
81
+ `## Unreleased` at the top. Commit these release files together. Normal
82
+ contributor PRs leave the version bump to the maintainer.
83
+ 3. Merge only after review and a passing **CI passed**. Pull the merged `main`,
84
+ ensure the checkout is clean, and run `npm ci` and `npm run check`.
85
+ 4. Create and push the matching tag: `git tag vX.Y.Z && git push origin vX.Y.Z`.
86
+ 5. Wait for its `CI` run and review the draft. Download its tarball and
87
+ `SHA256SUMS` to an empty directory, then run `sha256sum --check SHA256SUMS`
88
+ (`shasum -a 256 --check SHA256SUMS` on macOS).
89
+ 6. Publish that verified tarball yourself with `npm publish ./pi-typesafe-X.Y.Z.tgz`,
90
+ using maintainer credentials and the normal npm authentication checks. Then
91
+ publish the GitHub draft manually. Do not publish a PR artifact or run
92
+ `npm publish` from an unreviewed branch.
93
+
94
+ Merging this workflow creates no tag, npm publication, or published GitHub
95
+ release. The existing manual `npm publish` process can still be used from a
96
+ reviewed, checked release commit; publishing the tagged tarball is preferred
97
+ because it is the exact package tested by the release workflow.
@@ -0,0 +1,41 @@
1
+ import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
2
+ import { ask, authState, createTypeSafe, choice, describeAuth, noul } from "pi-typesafe";
3
+
4
+ // A separate extension using the public API, not pi-typesafe's private modules.
5
+ // It owns its own consent and request budget; it does not reuse /typesafe enable.
6
+ export default function decisionExample(pi: ExtensionAPI): void {
7
+ pi.registerCommand("decision-demo", {
8
+ description: "Send one synthetic, batched decision request to TypeSafe",
9
+ async handler(_args, ctx) {
10
+ if (!ctx.hasUI) return;
11
+ // Say which key is in effect before showing a data notice, so a missing one is not mistaken for consent.
12
+ const auth = describeAuth(authState());
13
+ if (auth.level === "error") {
14
+ ctx.ui.notify(auth.text, "warning");
15
+ return;
16
+ }
17
+ if (!await ctx.ui.confirm("Send a TypeSafe request?", "This synthetic example goes to api.typesafe.ai and may incur charges.")) return;
18
+ try {
19
+ // One client instance, one session budget, and a daily spend cap the script cannot forget.
20
+ const client = createTypeSafe({ maxRequests: 1, maxUsdPerDay: 1 });
21
+ const answer = await ask(client, {
22
+ state: "Please refund the duplicate charge.",
23
+ questions: {
24
+ team: choice("Which team should handle the request?", {
25
+ billing: "Payments and refunds", engineering: "Software defects", other: "Anything else",
26
+ }),
27
+ refund: noul("Is the sender requesting a refund?"),
28
+ },
29
+ }, { timeoutMs: 5_000 });
30
+ if (!answer.ok) {
31
+ ctx.ui.notify(answer.errorCode === "budget" ? `No request was sent: ${answer.error}` : "The example could not complete.", "warning");
32
+ return;
33
+ }
34
+ ctx.ui.notify(`Team: ${answer.answers.team.choice}; P(refund): ${answer.answers.refund.noul}; ${answer.elapsedMs} ms`, "info");
35
+ } catch (error) {
36
+ // createTypeSafe rejects an unusable key store; ask() never throws.
37
+ ctx.ui.notify(error instanceof Error ? error.message : "The example could not start.", "error");
38
+ }
39
+ },
40
+ });
41
+ }
@@ -0,0 +1,2 @@
1
+ // Pi entry point. Kept at extensions/index.js so Pi lists the extension as "pi-typesafe".
2
+ export { default } from "../dist/extension.js";
@@ -0,0 +1,89 @@
1
+ {
2
+ "name": "pi-typesafe",
3
+ "version": "0.8.0",
4
+ "description": "TypeSafe AI (Jev) decisions for Pi: batched Choice/Score/Noul evaluation tool, terminal playground, and a typed API other extensions build on.",
5
+ "type": "module",
6
+ "license": "MIT",
7
+ "author": "Ryan Gapac",
8
+ "repository": {
9
+ "type": "git",
10
+ "url": "git+https://github.com/DevMortimer/pi-typesafe.git"
11
+ },
12
+ "homepage": "https://github.com/DevMortimer/pi-typesafe#readme",
13
+ "bugs": "https://github.com/DevMortimer/pi-typesafe/issues",
14
+ "funding": "https://github.com/sponsors/DevMortimer",
15
+ "keywords": [
16
+ "pi-package",
17
+ "pi-extension",
18
+ "typesafe",
19
+ "jev",
20
+ "structured-decisions"
21
+ ],
22
+ "engines": {
23
+ "node": ">=22.19.0"
24
+ },
25
+ "exports": {
26
+ ".": {
27
+ "types": "./dist/index.d.ts",
28
+ "import": "./dist/index.js"
29
+ },
30
+ "./extension": {
31
+ "types": "./dist/extension.d.ts",
32
+ "import": "./dist/extension.js"
33
+ },
34
+ "./ui": {
35
+ "types": "./dist/ui.d.ts",
36
+ "import": "./dist/ui.js"
37
+ },
38
+ "./calibrate": {
39
+ "types": "./dist/calibrate.d.ts",
40
+ "import": "./dist/calibrate.js"
41
+ },
42
+ "./package.json": "./package.json"
43
+ },
44
+ "files": [
45
+ "dist",
46
+ "extensions",
47
+ "examples",
48
+ "README.md",
49
+ "LICENSE"
50
+ ],
51
+ "pi": {
52
+ "extensions": [
53
+ "./extensions/index.js"
54
+ ],
55
+ "image": "https://raw.githubusercontent.com/DevMortimer/pi-typesafe/main/docs/preview.png"
56
+ },
57
+ "scripts": {
58
+ "build": "tsc -p tsconfig.build.json",
59
+ "typecheck": "tsc --noEmit",
60
+ "test": "node --import tsx --test tests/*.test.ts",
61
+ "test:live": "node --env-file=.env scripts/live-smoke.mjs",
62
+ "dev:pi": "node scripts/dev-pi.mjs",
63
+ "check": "npm run build && npm run typecheck && npm test",
64
+ "prepack": "npm run build"
65
+ },
66
+ "dependencies": {
67
+ "@typesafe-ai/sdk": "^0.6.0",
68
+ "typebox": "^1.3.31"
69
+ },
70
+ "peerDependencies": {
71
+ "@earendil-works/pi-coding-agent": ">=0.85.1 <1",
72
+ "@earendil-works/pi-tui": ">=0.85.1 <1"
73
+ },
74
+ "peerDependenciesMeta": {
75
+ "@earendil-works/pi-coding-agent": {
76
+ "optional": true
77
+ },
78
+ "@earendil-works/pi-tui": {
79
+ "optional": true
80
+ }
81
+ },
82
+ "devDependencies": {
83
+ "@earendil-works/pi-coding-agent": "0.86.1",
84
+ "@earendil-works/pi-tui": "^0.86.1",
85
+ "@types/node": "^22.0.0",
86
+ "tsx": "^4.23.13",
87
+ "typescript": "^7.0.2"
88
+ }
89
+ }
@@ -0,0 +1,23 @@
1
+ import { spawn } from 'node:child_process';
2
+ import { fileURLToPath } from 'node:url';
3
+ import { join } from 'node:path';
4
+ import { loadEnvFile } from 'node:process';
5
+
6
+ const root = fileURLToPath(new URL('../', import.meta.url));
7
+ // Optional: a private .env can supply TYPESAFE_API_KEY; otherwise use /typesafe login inside Pi.
8
+ try {
9
+ loadEnvFile(join(root, '.env'));
10
+ } catch {
11
+ // No .env present; the stored key from /typesafe login (if any) is used.
12
+ }
13
+ // --no-extensions keeps an installed pi-typesafe (or other extensions) from loading alongside the working tree.
14
+ const child = spawn('pi', ['--no-extensions', '-e', root, ...process.argv.slice(2)], {
15
+ cwd: root,
16
+ env: process.env,
17
+ stdio: 'inherit',
18
+ });
19
+ child.on('error', () => {
20
+ console.error('Could not start Pi. Install the Pi CLI and make it available on PATH.');
21
+ process.exitCode = 1;
22
+ });
23
+ child.on('exit', code => { process.exitCode = code ?? 1; });
@@ -0,0 +1,35 @@
1
+ import assert from 'node:assert/strict';
2
+ import { createTypeSafe, choice, noul, score } from '../dist/index.js';
3
+
4
+ // One explicitly requested, billable call with synthetic data only.
5
+ try {
6
+ const client = createTypeSafe({ maxRequests: 1 });
7
+ const result = await client.evaluate({
8
+ state: { message: 'I was charged twice for my subscription. Please help today.' },
9
+ questions: {
10
+ category: choice('Which team should handle this message?', {
11
+ billing: 'Charges and payments', technical: 'Software failures', other: 'None of these',
12
+ }),
13
+ urgent: noul('Does the sender request help today?'),
14
+ frustration: score('How frustrated does the sender sound?', [
15
+ 'A neutral request without expressed frustration',
16
+ 'Expressed frustration while remaining civil',
17
+ 'Explicit anger or threats',
18
+ ]),
19
+ },
20
+ });
21
+ assert.equal(result.answers.category.choice, 'billing');
22
+ assert.ok(result.answers.urgent.noul > 0.5);
23
+ assert.ok(result.answers.frustration.score >= 0 && result.answers.frustration.score <= 2);
24
+ assert.equal(client.getUsage().requestsStarted, 1);
25
+ console.log(JSON.stringify({
26
+ passed: true, model: result.model, elapsedMs: result.elapsedMs, usage: result.usage,
27
+ category: result.answers.category.choice,
28
+ urgent: result.answers.urgent.noul,
29
+ frustration: result.answers.frustration.score,
30
+ }, null, 2));
31
+ } catch (error) {
32
+ // Do not dump errors, request bodies, environment, or stacks containing credentials.
33
+ console.error(JSON.stringify({ passed: false, code: error?.code ?? 'assertion-or-configuration', status: error?.status }));
34
+ process.exitCode = 1;
35
+ }
@@ -0,0 +1,42 @@
1
+ import type { Questions, SystemOneRequest } from "@typesafe-ai/sdk";
2
+ import type { Evaluation, TypeSafe } from "./client.js";
3
+ import { TypeSafeIntegrationError } from "./errors.js";
4
+ import type { IntegrationErrorCode } from "./errors.js";
5
+
6
+ /** Anything with pi-typesafe's `evaluate`: the real client in a session, a stub in tests. */
7
+ export type Judge = Pick<TypeSafe, "evaluate">;
8
+
9
+ /** Default per-ask deadline; matches the client's own request timeout. */
10
+ export const DEFAULT_ASK_TIMEOUT_MS = 15_000;
11
+
12
+ const FALLBACK_MESSAGE = "TypeSafe request failed.";
13
+
14
+ export interface AskOptions {
15
+ /** Per-ask deadline, merged with the caller's own signal. Default: DEFAULT_ASK_TIMEOUT_MS. */
16
+ timeoutMs?: number;
17
+ signal?: AbortSignal;
18
+ }
19
+
20
+ export type AskAnswer<Q extends Questions> =
21
+ | { readonly ok: true; readonly answers: Evaluation<Q>["answers"]; readonly model: string; readonly usage: Evaluation<Q>["usage"]; readonly elapsedMs: number }
22
+ | { readonly ok: false; readonly error: string; readonly errorCode?: IntegrationErrorCode };
23
+
24
+ /**
25
+ * One typed Jev request that never throws: a failure comes back as `{ ok: false }` with pi-typesafe's own message
26
+ * (which carries no upstream body, header, key, or submitted state) and its code, so a caller can stop asking after a
27
+ * `budget` error. The per-ask timeout is merged into the caller's signal, so either can cancel the request.
28
+ *
29
+ * This is the author-facing "ask Jev" seam. Agents get the same admission through the `typesafe_evaluate` tool; the two
30
+ * share `prepareEvaluationRequest`, so what one accepts the other accepts.
31
+ */
32
+ export async function ask<Q extends Questions>(judge: Judge, request: SystemOneRequest<Q>, options: AskOptions = {}): Promise<AskAnswer<Q>> {
33
+ const timeout = AbortSignal.timeout(options.timeoutMs ?? DEFAULT_ASK_TIMEOUT_MS);
34
+ const signal = options.signal ? AbortSignal.any([options.signal, timeout]) : timeout;
35
+ try {
36
+ const result = await judge.evaluate(request, { signal });
37
+ return { ok: true, answers: result.answers, model: result.model, usage: result.usage, elapsedMs: result.elapsedMs };
38
+ } catch (error) {
39
+ if (error instanceof TypeSafeIntegrationError) return { ok: false, error: error.message, errorCode: error.code };
40
+ return { ok: false, error: FALLBACK_MESSAGE };
41
+ }
42
+ }