@edgehero/pi-dispatch 2.1.0 → 3.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. package/.env.example +41 -5
  2. package/README.md +11 -5
  3. package/deploy/docker-compose.yml +12 -0
  4. package/deploy/egress-proxy.conf +28 -3
  5. package/deploy/pi-dispatch-egress-proxy.container +8 -2
  6. package/package.json +8 -1
  7. package/src/allocation.mjs +731 -0
  8. package/src/backends.mjs +243 -0
  9. package/src/budget.mjs +40 -4
  10. package/src/cli.mjs +222 -11
  11. package/src/config.mjs +126 -5
  12. package/src/daemon-facts.mjs +3 -0
  13. package/src/deployment-venue.mjs +1 -0
  14. package/src/doctor.mjs +2261 -203
  15. package/src/dollar-budget.mjs +373 -0
  16. package/src/dollar-fingerprint.mjs +83 -0
  17. package/src/egress-cli.mjs +316 -0
  18. package/src/egress-proxy-state.mjs +35 -5
  19. package/src/egress.mjs +12 -0
  20. package/src/env-allowlist.mjs +107 -6
  21. package/src/env-file.mjs +194 -25
  22. package/src/envelope.mjs +413 -0
  23. package/src/exit-code.mjs +22 -0
  24. package/src/fleet-lease.mjs +85 -25
  25. package/src/get-token.mjs +16 -5
  26. package/src/git-dirty.mjs +67 -0
  27. package/src/github-app-setup.mjs +6 -3
  28. package/src/github-host.mjs +5 -3
  29. package/src/identity.mjs +2 -1
  30. package/src/image-preflight.mjs +98 -24
  31. package/src/image-ref.mjs +37 -0
  32. package/src/import-pi.mjs +4 -2
  33. package/src/index.mjs +407 -62
  34. package/src/init.mjs +18 -0
  35. package/src/job-id.mjs +26 -3
  36. package/src/live-probes.mjs +24 -9
  37. package/src/model-catalog.mjs +297 -0
  38. package/src/model-endpoints.mjs +649 -0
  39. package/src/model-ref.mjs +151 -0
  40. package/src/models-json.mjs +262 -0
  41. package/src/money.mjs +144 -0
  42. package/src/octokit-log.mjs +65 -0
  43. package/src/outbox-plan.mjs +218 -0
  44. package/src/outbox.mjs +29 -9
  45. package/src/output-cap.mjs +157 -0
  46. package/src/pause-windows.mjs +81 -2
  47. package/src/pi-model-loader.mjs +77 -0
  48. package/src/podman-stack.mjs +16 -3
  49. package/src/portfolio-snapshot.mjs +304 -0
  50. package/src/prepare-local.mjs +247 -12
  51. package/src/prepare.mjs +35 -3
  52. package/src/priorities.mjs +569 -0
  53. package/src/processor.mjs +599 -170
  54. package/src/project-id.mjs +17 -0
  55. package/src/projects.mjs +238 -0
  56. package/src/provider-steering.mjs +179 -65
  57. package/src/queue.mjs +111 -6
  58. package/src/reserved-env.mjs +30 -0
  59. package/src/run-container.mjs +59 -5
  60. package/src/run-history.mjs +379 -24
  61. package/src/run-mirror.mjs +30 -0
  62. package/src/runtime-settings.mjs +104 -9
  63. package/src/schedules.mjs +33 -1
  64. package/src/scoped-limits.mjs +447 -27
  65. package/src/service.mjs +15 -4
  66. package/src/session-store.mjs +131 -6
  67. package/src/start.mjs +528 -40
  68. package/src/triggers-file.mjs +65 -4
  69. package/src/triggers.mjs +135 -7
  70. package/src/up.mjs +308 -34
  71. package/src/valkey-endpoint.mjs +3 -2
package/src/init.mjs CHANGED
@@ -14,6 +14,7 @@ import { parseBackendList, venuesOf } from "./backends.mjs";
14
14
  import { deploymentVenueEnv } from "./deployment-venue.mjs";
15
15
  import { setEnvKeyIfEmpty } from "./env-file.mjs";
16
16
  import { PACKAGED_EGRESS_PROXY_CONF } from "./egress-conf-copy.mjs";
17
+ import { EMPTY_MODEL_ENDPOINTS, MODEL_ENDPOINTS_FILE_NAME, MODEL_ENDPOINTS_INCLUDE_NAME, renderEndpointsInclude } from "./model-endpoints.mjs";
17
18
  import { VALKEY_PASSWORD_KEY, newValkeyPassword } from "./valkey-auth.mjs";
18
19
 
19
20
  const EMPTY_TRIGGERS = `${JSON.stringify({ triggers: [] }, null, 2)}\n`;
@@ -34,6 +35,9 @@ const EMPTY_SUBSCRIPTIONS = `${JSON.stringify({ version: 1, subscriptions: [] },
34
35
  // Versioned for the subscriptions reason, sharpened: this is enforcement config, and a silently
35
36
  // down-read newer file would be a silently widened spend limit.
36
37
  export const EMPTY_SCOPED_LIMITS = `${JSON.stringify({ version: 1, limits: [] }, null, 2)}\n`;
38
+ // Projects (issue #499): named groups of repos and folders, recorded per run. Empty is inert: every run records no
39
+ // project. Versioned for the scoped-limits reason, since part B of the issue caps a project as one.
40
+ export const EMPTY_PROJECTS = `${JSON.stringify({ version: 1, projects: [] }, null, 2)}\n`;
37
41
  /**
38
42
  * The egress allowlist (REQ-EGRESS-ALLOWLIST): the hosts a job container may reach, one bare hostname per
39
43
  * line. Scaffolded with the three a job cannot work without, and NOT empty -- unlike every other scaffold
@@ -105,6 +109,10 @@ export function runInit(cwd = process.cwd(), deps = {}) {
105
109
  // Issue #468: the new file carries this deployment's own Valkey password (the example's `# VALKEY_PASSWORD=` line
106
110
  // filled in, never shown) and is created readable by this account alone: it holds that password and, soon, the
107
111
  // provider key. `wx`: a file that appeared since the check above is never overwritten (init's contract).
112
+ // Its GROUP is the folder's choice and left so (issue #522): on macOS a new file takes the folder's group (`wheel`
113
+ // under `/private/tmp`), and on Linux a setgid folder's, which is how a shared deployment gives the service its
114
+ // group. Setting this account's group here was considered and rejected: it would undo that shared layout, and the
115
+ // writer `up` uses keeps whatever group a new file there gets, so the fresh file is never refused for it.
108
116
  const text = setEnvKeyIfEmpty(String(fs.readFileSync(source, "utf8")), VALKEY_PASSWORD_KEY, newPassword(), { platform });
109
117
  if (createOnly(fs, envPath, text, { mode: 0o600 })) results.push(["created", ".env", `from .env.example, mode 0600, with a generated ${VALKEY_PASSWORD_KEY} (value not shown): set your provider key next`]);
110
118
  else results.push(["kept", ".env", KEPT]);
@@ -115,7 +123,17 @@ export function runInit(cwd = process.cwd(), deps = {}) {
115
123
  scaffold(fs, results, join(cwd, "pi-packages.json"), EMPTY_PACKAGES, "empty pi package list (stage with import-pi --with-packages)");
116
124
  scaffold(fs, results, join(cwd, "subscriptions.json"), EMPTY_SUBSCRIPTIONS, "empty subscription list (declare plan prices for the admin's cost analytics)");
117
125
  scaffold(fs, results, join(cwd, "scoped-limits.json"), EMPTY_SCOPED_LIMITS, "empty scoped-limits list (per repo/folder caps; the folder mutex needs no file)");
126
+ scaffold(fs, results, join(cwd, "projects.json"), EMPTY_PROJECTS, "empty projects list (group repos and folders into a project, recorded per run)");
127
+ // Issue #504 part B: NO envelope.json. Unset is no delegation, the safe default, and an envelope is a decision (a
128
+ // total, a floor per project, the delegation rules) with no neutral empty form: a scaffold would either govern every
129
+ // job or refuse the boot. The .env written above documents PI_ENVELOPE_FILE (from .env.example), and
130
+ // docs/allocation.md the file.
118
131
  scaffold(fs, results, join(cwd, "egress-allowlist.conf"), DEFAULT_EGRESS_ALLOWLIST, "egress allowlist (provider + forge + registry; the egress policy is on unless PI_EGRESS=0)");
132
+ // Issue #503: the declared model endpoints, empty, and the proxy include rendered from them, which is the empty
133
+ // render (its header only). squid refuses to start on a missing include file and starts on a comments-only one
134
+ // (measured), so the include exists from the first init even with nothing declared.
135
+ scaffold(fs, results, join(cwd, MODEL_ENDPOINTS_FILE_NAME), EMPTY_MODEL_ENDPOINTS, "empty model endpoints list (local or LAN model servers a job may reach through the proxy)");
136
+ scaffold(fs, results, join(cwd, MODEL_ENDPOINTS_INCLUDE_NAME), renderEndpointsInclude([]), "the proxy's rules for those endpoints, none yet (generated: do not edit)");
119
137
  // Issue #480: the proxy's rules, the file beside the allowlist that the docker proxy mounts. Create-only like every
120
138
  // scaffold here, so a clone's own deploy/egress-proxy.conf is reported and kept. The one scaffold whose content is
121
139
  // not this module's: it is the package's file verbatim, read from the package and never through `fs` (the
package/src/job-id.mjs CHANGED
@@ -6,17 +6,40 @@ import { forgeSpec } from "./forges.mjs";
6
6
  * A deterministic jobId for a local job. BullMQ's dedup is `EXISTS jobId`, so a double-invoke of
7
7
  * the same task within the same minute produces the same id and the duplicate is ignored -- the
8
8
  * local equivalent of REQ-DEDUP-BY-DELIVERY-GUID, guarding against a hasty second Enter
9
- * double-spending, without blocking a deliberate re-run a minute later.
9
+ * double-spending, without blocking a deliberate re-run a minute later. The model fields (provider, model,
10
+ * models) are part of the task's identity: the same task on another model is a different run.
10
11
  *
11
12
  * Kept free of any bullmq import so the dedup logic is testable everywhere, not only where the
12
13
  * queue's dependencies are installed.
13
14
  */
14
- export function localJobId({ folder, flow, task, minute }) {
15
+ export function localJobId({ folder, flow, task, minute, provider, model, models }) {
16
+ // The model fields join the key (issue #544): two runs that differ only in the model are two runs, and with the
17
+ // key blind to them the second was dropped as a duplicate. Appended only when one is set, so a job that names no
18
+ // model keeps the id it always had. `models` is JSON so a list cannot collide with a differently split one.
19
+ const modelFields = provider === undefined && model === undefined && models === undefined ? [] : [provider ?? "", model ?? "", models === undefined ? "" : JSON.stringify(models)];
15
20
  // NUL-delimited so {folder:'a',task:'bc'} and {folder:'ab',task:'c'} cannot collide.
16
- const digest = createHash("sha256").update([folder, flow ?? "", task ?? "", minute].join("\0")).digest("hex");
21
+ const digest = createHash("sha256").update([folder, flow ?? "", task ?? "", minute, ...modelFields].join("\0")).digest("hex");
17
22
  return `local-${digest.slice(0, 16)}`;
18
23
  }
19
24
 
25
+ /**
26
+ * The jobId of a cron trigger fired by hand (`pi-dispatch run --trigger <id>`, issue #505): `manual:<id>:<millis>`,
27
+ * the millis floored to the minute. Floored so the same command twice inside one minute is the same id, and the
28
+ * queue keeps the first: the dedup `localJobId` gives `pi-dispatch run`, for the same reason (a hasty second Enter
29
+ * must not pay twice). A later minute is a new id, so a deliberate re-run is never blocked.
30
+ *
31
+ * The shape is BullMQ's, not a choice: a custom id that holds a `:` is refused unless it splits into exactly three
32
+ * parts (bullmq `Job.addJob`, "Custom Id cannot contain :"), which is how its own `repeat:<id>:<millis>` passes. A cron
33
+ * id can hold no `:` (the triggers loader refuses one), so this always splits into three. An ISO time would add two
34
+ * more. The `manual:` prefix keeps it apart from a scheduled run's `repeat:` id, which is what tells the event writer
35
+ * (`localEventContext`) there is no scheduled instant to report.
36
+ */
37
+ export function manualTriggerJobId({ triggerId, now }) {
38
+ if (typeof triggerId !== "string" || triggerId === "" || triggerId.includes(":")) throw new TypeError("a cron trigger id is a non-empty string with no ':'");
39
+ const millis = Math.floor(now.getTime() / 60_000) * 60_000;
40
+ return `manual:${triggerId}:${millis}`;
41
+ }
42
+
20
43
  /**
21
44
  * The retry-idempotent jobId for a chained (outbox-requested) child job: `parent id + content-hash of
22
45
  * (flow, task)`, with NO time component. BullMQ's dedup is `EXISTS jobId`, so a retried parent
@@ -452,10 +452,11 @@ export function imagePinningVerdict({ code, output, stillAbsent, bin = "docker"
452
452
  }
453
453
 
454
454
  /**
455
- * EGRESS, folded in from `doctor.mjs`'s canary: two containers on a job-shaped network behind the proxy, one that
456
- * must reach the provider and one that must not reach an unlisted host. `results` is `[{ want, reached }]` from the
457
- * canary's `readBack`; anything short of both readings -- the policy off, the proxy down, the canary skipped -- is
458
- * "not read back", never a pass.
455
+ * EGRESS, folded in from `doctor.mjs`'s canary: three containers on a job-shaped network behind the proxy, one that
456
+ * must reach the provider, one that must not reach an unlisted host, and one that must not get plain HTTP through to a
457
+ * listed host on a port other than 80 (issue #508). `results` is `[{ want, reached, probe }]` from the canary's
458
+ * `readBack`, `probe` naming which of the three it is; anything short of all three readings -- the policy off, the
459
+ * proxy down, the canary skipped -- is "not read back", never a pass.
459
460
  *
460
461
  * Each venue hands in its OWN canary's readings: docker's from doctor's egress lines, the podman venue's from the canary
461
462
  * its `--live` runs under Podman (issue #431), so "see the egress lines above" names lines about the proxy that venue's
@@ -466,15 +467,29 @@ export function egressVerdict({ armed, results, keeperBlocked = null }) {
466
467
  if (armed !== true) return notReadBack("egress", "PI_EGRESS is off, so there is no policy to read back");
467
468
  // Issue #458: the podman venue on Podman 4.x without its keeper runs no canary, since its teardown would break the proxy.
468
469
  if (keeperBlocked) return notReadBack("egress", `the egress canary was not run, because ${keeperBlocked} (see the keeper line above)`);
469
- if (!Array.isArray(results) || results.length < 2) return notReadBack("egress", "the egress canary did not run both probes (see the egress lines above)");
470
- // A WRONG reading fails first, whatever else is missing: an unlisted host that was reached is a finding even when
471
- // the provider probe did not run, and reporting it as merely unread would pass doctor over it.
470
+ // Each of the three probes exactly once, by NAME (issue #508, gate round 1): a count alone read two unlisted readings,
471
+ // or three with no probe named, as a canary that ran all three.
472
+ const probes = Array.isArray(results) ? results.map((r) => r?.probe) : [];
473
+ if (probes.length !== EGRESS_PROBES.length || !EGRESS_PROBES.every((p) => probes.includes(p))) return notReadBack("egress", "the egress canary did not run all three probes (see the egress lines above)");
474
+ // With all three probes present, a WRONG reading fails first, whatever else did not run to an answer: an unlisted
475
+ // host that was reached is a finding even when the provider probe did not run, and reporting it as merely unread
476
+ // would pass doctor over it. A probe missing altogether is "not read back" above, before any reading is judged.
472
477
  const wrong = results.filter((r) => typeof r.reached === "boolean" && r.reached !== r.want);
473
- if (wrong.length > 0) return verdict("egress", false, wrong.map((r) => (r.want ? "the provider was not reached" : "an unlisted host was reached")).join("; "));
478
+ if (wrong.length > 0) return verdict("egress", false, wrong.map((r) => EGRESS_WRONG[r.probe]).join("; "));
474
479
  if (results.some((r) => typeof r.reached !== "boolean")) return notReadBack("egress", "an egress probe did not run to an answer (see the egress lines above)");
475
- return verdict("egress", true, "the provider was reached and an unlisted host was not");
480
+ return verdict("egress", true, "the provider was reached, an unlisted host was not, and plain HTTP off port 80 was refused");
476
481
  }
477
482
 
483
+ /** The canary's probes: `CANARY_PROBE_SLUGS` in doctor.mjs, which imports this module, so a test pins the two equal. */
484
+ export const EGRESS_PROBES = Object.freeze(["provider", "unlisted", "plainhttp"]);
485
+
486
+ /** What a wrong reading says, by the canary probe it came from. */
487
+ const EGRESS_WRONG = Object.freeze({
488
+ provider: "the provider was not reached",
489
+ unlisted: "an unlisted host was reached",
490
+ plainhttp: "plain HTTP to a listed host off port 80 was let through",
491
+ });
492
+
478
493
  /**
479
494
  * EPHEMERAL (issue #344): two runs under ONE name, each detached with `--rm`, each waited on until `docker ps -a` no
480
495
  * longer lists it. `first`/`second` are `{ started, id, removal: { state, ms } }` (`state` one of `gone`, `exited`
@@ -0,0 +1,297 @@
1
+ /**
2
+ * Does this model exist (issue #502)? The worker's answer, asked among the free gates before any token mint,
3
+ * clone or reservation, so a typo in a model id costs nothing instead of a paid container that exits 2.
4
+ *
5
+ * Two sources, the two pi itself reads at the 0.99.1 pin:
6
+ * - the builtin catalog, every provider's chat, image and classifier models (`getAllBuiltinModels`), so a
7
+ * classifier or image model on an allowed-model list is found too. `getPricedModel` (pricing.mjs) answers
8
+ * chat models only, which is why this module does not reuse it;
9
+ * - the overlay `models.json` (`readOverlayModels`, model-endpoints.mjs), whose `providers.<name>.models[].id`
10
+ * declares a custom model, under a custom provider or a builtin one (an Ollama model under `openai`).
11
+ *
12
+ * Imports only `@earendil-works/pi-ai/providers/all`, which pi-ai's package.json declares side-effect-free. Never
13
+ * the root or `compat`: a lookup must not register providers as a side effect of being asked a question.
14
+ *
15
+ * Matching is EXACT and case-sensitive on both the provider and the id, because that is how pi resolves a model;
16
+ * an id differing only in case is a model pi would not find, and saying "known" would let the job spend a
17
+ * container to learn that.
18
+ *
19
+ * What this cannot see, by design and named in REQ-MODEL-POLICY: a provider an extension registers
20
+ * (`pi.registerProvider`) and a virtual model (`pi.registerVirtualModel`) exist only inside the job. A flow that
21
+ * needs one declares the physical models it routes to in the overlay `models.json`, which this module reads.
22
+ */
23
+
24
+ import { getAllBuiltinModels, getBuiltinModels, getBuiltinProviders } from "@earendil-works/pi-ai/providers/all";
25
+
26
+ let builtinIndex = null;
27
+
28
+ /**
29
+ * Two `provider -> Set(id)` maps over the builtin catalog, built ONCE per process: `any` (chat, image and classifier
30
+ * models, `getAllBuiltinModels`) and `chat` (`getBuiltinModels`). The catalog is generated data inside the pinned
31
+ * package, so it cannot change while the worker runs; walking ~1,600 ids per job would be work with no answer it
32
+ * could change. `Map`s of own entries, so `__proto__` or `constructor` is never a provider.
33
+ */
34
+ function builtins() {
35
+ if (builtinIndex !== null) return builtinIndex;
36
+ const any = new Map();
37
+ const chat = new Map();
38
+ for (const provider of getBuiltinProviders()) {
39
+ any.set(provider, new Set(getAllBuiltinModels(provider).map((m) => m?.id).filter((id) => typeof id === "string")));
40
+ chat.set(provider, new Set(getBuiltinModels(provider).map((m) => m?.id).filter((id) => typeof id === "string")));
41
+ }
42
+ builtinIndex = { any, chat };
43
+ return builtinIndex;
44
+ }
45
+
46
+ /**
47
+ * Is `provider`/`id` a builtin model at the pin? `chatOnly` for a job's MAIN model (PR #536's review): the runner
48
+ * resolves it with `modelRuntime.getModel`, which answers chat models only, so an image or classifier model as the
49
+ * main model passes a catalog-wide check and then exits 2 in a paid container. A LIST entry may be any kind, since a
50
+ * flow may call a classifier or an image model through the registry.
51
+ */
52
+ export function isBuiltinModel(provider, id, { chatOnly = false } = {}) {
53
+ if (typeof provider !== "string" || typeof id !== "string") return false;
54
+ return builtins()[chatOnly ? "chat" : "any"].get(provider)?.has(id) === true;
55
+ }
56
+
57
+ /**
58
+ * The builtin catalog's model object for `provider`/`id` (chat, image or classifier), or null (issue #503 part 7). The
59
+ * zero-rated check (`zeroRatedVerdict`, model-endpoints.mjs) reads its `cost` and `type` for a model the overlay does
60
+ * not redefine; that module never imports pi, so this is handed to it. Own entries only, exact match, like the rest.
61
+ */
62
+ export function builtinModel(provider, id) {
63
+ if (typeof provider !== "string" || typeof id !== "string") return null;
64
+ if (!getBuiltinProviders().includes(provider)) return null;
65
+ return getAllBuiltinModels(provider).find((m) => m?.id === id) ?? null;
66
+ }
67
+
68
+ /**
69
+ * The builtin catalog's CHAT models of `provider` (the ones pi applies an overlay's provider `compat` and `modelOverrides`
70
+ * to), or [] for a provider the catalog does not know (issue #571: doctor's no-usage check reads their api, compat and
71
+ * cost). Copies of the list, never the catalog's own array.
72
+ */
73
+ export function builtinChatModels(provider) {
74
+ if (typeof provider !== "string" || !getBuiltinProviders().includes(provider)) return [];
75
+ return [...getBuiltinModels(provider)];
76
+ }
77
+
78
+ /** Does the overlay `models.json` (already parsed, or null) declare `provider`/`id`? Own keys only. */
79
+ export function isOverlayModel(overlay, provider, id) {
80
+ return overlayDeclares(overlay, provider, id) && overlayProviderProblem(overlay.providers[provider], provider) === null;
81
+ }
82
+
83
+ /** Does the overlay's entry for `provider` list `id`, whether or not pi would compose it? */
84
+ function overlayDeclares(overlay, provider, id) {
85
+ const providers = overlay?.providers;
86
+ if (providers === null || typeof providers !== "object" || Array.isArray(providers)) return false;
87
+ if (!Object.hasOwn(providers, provider)) return false;
88
+ const models = providers[provider]?.models;
89
+ return Array.isArray(models) && models.some((m) => m !== null && typeof m === "object" && m.id === id);
90
+ }
91
+
92
+ /**
93
+ * Would pi COMPOSE this overlay provider (PR #536's review)? A file that passes pi's schema can still lose a provider
94
+ * at the next step: pi 0.99.1's `applyModelsJson` and `modelFromJson` (`pi-coding-agent/dist/core/provider-composer.js`)
95
+ * throw for `oauth` with no `baseUrl`, and per model no resolvable `api` or
96
+ * `baseUrl`, or a `contextWindow` or `maxTokens` at or below zero. pi then keeps the provider's BUILTIN models
97
+ * (`ModelRuntime.composeProvider` falls back to the base) and drops every model the overlay added, so such a model
98
+ * does not exist in the job. Mirrored here, the defaults search included: a model's missing `api` or `baseUrl` comes
99
+ * from a chat model already in the provider's list (the same id, else the same api, else an `openai-completions`
100
+ * one, else the first), which starts as the builtin chat models and grows with each overlay model in order. A null
101
+ * return is "composes"; a string names the first rule broken. Held to pi by the differential test in
102
+ * `models-json.test.mjs`, which asks pi's own `ModelRuntime` for every model in its corpus.
103
+ */
104
+ export function overlayProviderProblem(config, providerId) {
105
+ if (config === null || typeof config !== "object" || Array.isArray(config)) return null;
106
+ if (config.oauth && !config.baseUrl) return "oauth-without-baseUrl";
107
+ // NOT mirrored: pi's "must specify baseUrl, headers, compat, modelOverrides or models" throw. An entry that sets none
108
+ // of those carries nothing pi could drop (no model, no endpoint, no header), so the provider pi falls back to is the
109
+ // one the entry describes, and refusing its builtin models would be a false refusal. Mirroring it changed no verdict
110
+ // while only overlay models counted (PR #536's review, round 3, mutant H6), and would now refuse wrongly.
111
+ // With `oauth` set, pi's defaults search sees no base models at the pin (measured: a `radius` entry with `oauth` and
112
+ // no `api` loses its custom models to "no api specified" though radius has builtin chat models). Mirrored. Radius
113
+ // itself, pi's OAuth gateway, is otherwise not modelled: its models need an OAuth login, which a job never has.
114
+ const builtinChat = !config.oauth && getBuiltinProviders().includes(providerId) ? getBuiltinModels(providerId) : [];
115
+ // pi keeps a builtin model's own baseUrl under `oauth: "radius"`; that branch is not mirrored, because with `oauth`
116
+ // set the list above is empty (mutant H5 found it dead).
117
+ const models = builtinChat.map((m) => ({ id: m.id, api: m.api, baseUrl: config.baseUrl ?? m.baseUrl }));
118
+ for (const def of config.models ?? []) {
119
+ // Kept for fidelity to pi's defaults search, though at the pin it cannot change a verdict (mutant H8): every
120
+ // builtin chat model has an api and a baseUrl, so whichever default is found supplies both, and only WHICH
121
+ // default is found depends on `wantApi`. It decides once pi gains a builtin model lacking one of them.
122
+ const wantApi = def.api ?? config.api;
123
+ const defaults = models.find((m) => m.id === def.id) ?? (wantApi ? models.find((m) => m.api === wantApi) : undefined) ?? models.find((m) => m.api === "openai-completions") ?? models[0];
124
+ const api = def.api ?? config.api ?? defaults?.api;
125
+ if (!api) return "no-api";
126
+ const baseUrl = def.baseUrl ?? config.baseUrl ?? defaults?.baseUrl;
127
+ if (!baseUrl) return "no-baseUrl";
128
+ if (def.contextWindow !== undefined && def.contextWindow <= 0) return "contextWindow";
129
+ if (def.maxTokens !== undefined && def.maxTokens <= 0) return "maxTokens";
130
+ const composed = { id: def.id, api, baseUrl };
131
+ const at = models.findIndex((m) => m.id === def.id);
132
+ if (at >= 0) models[at] = composed;
133
+ else models.push(composed);
134
+ }
135
+ return null;
136
+ }
137
+
138
+ /**
139
+ * Is this model in the catalog or the overlay? `overlay` is the parsed overlay `models.json` or null. (The job gate,
140
+ * `checkModelsKnown`, also refuses a builtin model whose provider entry pi would not compose.)
141
+ */
142
+ export function knownModel({ provider, id, overlay = null, chatOnly = false }) {
143
+ return isBuiltinModel(provider, id, { chatOnly }) || isOverlayModel(overlay, provider, id);
144
+ }
145
+
146
+ /**
147
+ * The errnos a read of the overlay `models.json` may fail with and then succeed a moment later (issue #552): a disk or
148
+ * network filesystem error (EIO), a busy resource (EAGAIN), and the process or system out of file handles (EMFILE,
149
+ * ENFILE). An ALLOW-LIST, the other way round from `transient.mjs`: here the harmful direction is the false
150
+ * transient. An errno `readOverlayModels` does not read as absence and this list does not name is a file the operator
151
+ * wrote that pi in the job loads none of (the existence check in image/runner/run-job.mjs, or pi's own
152
+ * read, fails), and a builtin provider the file routes would then go to its public endpoint. So the model gate
153
+ * refuses every job on it.
154
+ */
155
+ export const OVERLAY_TRANSIENT_READ_CODES = new Set(["EIO", "EAGAIN", "EMFILE", "ENFILE"]);
156
+
157
+ /** True when a failed read of the overlay `models.json` with this errno may succeed if asked again. */
158
+ export function isTransientOverlayRead(code) {
159
+ return OVERLAY_TRANSIENT_READ_CODES.has(code);
160
+ }
161
+
162
+ /**
163
+ * The gate's question for one job (issue #502): are its main model and every model on its list known? `refs` is
164
+ * `[{ provider, id, main? }]`, main first. `readOverlay` is called AT MOST ONCE per job.
165
+ *
166
+ * A BUILTIN model is judged by the overlay too (PR #536's review, round 3): when its provider has an overlay entry pi
167
+ * would not compose (`overlayProviderProblem`), pi drops that WHOLE entry, its `baseUrl`, `headers`, `apiKey` and
168
+ * `compat` included, and the builtin model then runs against the provider's public endpoint while the operator's
169
+ * file reads as though it routed it. Refused as `overlay-provider-invalid`.
170
+ *
171
+ * A file pi drops WHOLE refuses EVERY job (issue #539, PR #546's review, the simpler rule after its third round). pi
172
+ * drops an overlay that exists and fails to load for any reason (a schema error under one provider, a block comment,
173
+ * a truncated write, a UTF-16 save, an empty file, a directory: all measured against pi 0.99.1's `ModelConfig.load`),
174
+ * and every provider entry goes with it, so `openai: { baseUrl: <proxy> }` stops applying and `openai/gpt-4o` runs
175
+ * against `api.openai.com`. Which providers the broken file meant to route cannot be read from a file that does not
176
+ * load, and three rounds of reading it anyway kept finding cases, so the gate does not try: every main and listed
177
+ * ref, builtin or overlay, is refused (`overlay-unparseable`, or `overlay-is-a-directory`) until the operator fixes
178
+ * the file. Louder than needed for a job whose provider the file never mentions, and safe in direction. An absent
179
+ * file is no overlay, as for pi.
180
+ *
181
+ * A file the worker cannot READ is judged by its errno (issue #552). The job reads it through the same read-only
182
+ * mount, as the worker's own uid on native Linux, and pi in the job loads none of the file (the runner's existence
183
+ * check in image/runner/run-job.mjs, or pi's own read, fails). `readOverlayModels` reads what the job reads as no
184
+ * file as absence (ENOENT, and ELOOP, ENOTDIR or ENAMETOOLONG on the folder's path: no content is lost). Any other errno (EACCES or
185
+ * EPERM on the file or its folder, and any errno nobody listed) is a file the operator wrote and the job loses, so
186
+ * every job is refused (`overlay-unreadable`) until the worker's user can read it. Only the errnos
187
+ * `isTransientOverlayRead` names are retried, and then for EVERY job, builtin ones included: whether the file routes
188
+ * the job's provider cannot be known without reading it. A retried job gets the queue's second attempt, then fails.
189
+ *
190
+ * A `models.json` that is a link of any kind, dangling included (PR #553's review), refuses every job (`overlay-link`):
191
+ * the job's read-only mount does not resolve a link the way the host does, so the host may read the routing while the
192
+ * job runs with no overlay. The file itself is read; the overlay folder may be a link. A `models.json` that is a named
193
+ * pipe, a socket or a device (issue #556) refuses every job too (`overlay-not-a-file`), never opened: a FIFO with no
194
+ * writer would block the read on every pickup.
195
+ *
196
+ * Returns one of:
197
+ * - `{ ok: true }`;
198
+ * - `{ unknown: { provider, id }, why }`: the first unknown ref. `why` is `"not-in-catalog"`,
199
+ * `"overlay-unparseable"` for every ref when pi would not load the overlay, `"overlay-is-a-directory"` for every
200
+ * ref when it is a directory (pi fails the same way), `"overlay-unreadable"` for every ref when it cannot be read
201
+ * for a reason no retry changes, `"overlay-link"` for every ref when `models.json` is a link, `"overlay-not-a-file"` for every ref when it is a named pipe,
202
+ * socket or device (issue #556), or `"overlay-provider-invalid"` when the ref's provider has an overlay entry pi
203
+ * would not compose, so the operator learns the file is the problem rather than the id;
204
+ * - `{ fallbackUnlisted: { provider, id }, why: "fallback-unlisted" }`: the job has a list, every ref is known, and
205
+ * a listed model declares a server-side fallback (`declaredFallbacks`) that is not on the list under its provider;
206
+ * - `{ unavailable: code }`: the overlay could not be READ for a transient reason (`isTransientOverlayRead`). The
207
+ * caller retries rather than refusing, because a refusal is permanent and public and the next attempt may read
208
+ * the file.
209
+ */
210
+ export function checkModelsKnown(refs, { readOverlay = () => null } = {}) {
211
+ let overlay = null;
212
+ let overlayRead = false;
213
+ let unavailable = null; // the errno of a transient read, else null
214
+ let unparseable = null; // the `why` when the overlay could not be used, else null
215
+ const load = () => {
216
+ if (overlayRead) return;
217
+ overlayRead = true;
218
+ try {
219
+ overlay = readOverlay();
220
+ } catch (err) {
221
+ // A configError (pi would not load the text, or `models.json` is a link) is the file itself, which no
222
+ // retry changes. An errno is judged by what the job then loads (issue #552): EISDIR is named as a directory
223
+ // (PR #536's review), the few errnos a moment can clear are retried, and every other one, EACCES and any
224
+ // errno nobody listed, is a file the job loads none of, so every job is refused. Fail closed: an unknown
225
+ // errno never lets a job run.
226
+ if (err?.overlayLink === true) unparseable = "overlay-link";
227
+ else if (err?.overlayNotAFile === true) unparseable = "overlay-not-a-file";
228
+ else if (err?.code === "EISDIR") unparseable = "overlay-is-a-directory";
229
+ else if (isTransientOverlayRead(err?.code)) unavailable = err.code;
230
+ else if (typeof err?.code === "string") unparseable = "overlay-unreadable";
231
+ else if (err?.piDispatchConfig !== true) throw err; // a defect, not a state: never a permanent refusal
232
+ else unparseable = "overlay-unparseable";
233
+ overlay = null;
234
+ }
235
+ };
236
+ for (const ref of refs) {
237
+ load();
238
+ // pi drops the whole file, every provider entry with it: no job runs until it is fixed (see above).
239
+ if (unparseable !== null) return { unknown: { provider: ref.provider, id: ref.id }, why: unparseable };
240
+ // A transient read retries every job (issue #552): the file may route this ref's provider, builtin or not.
241
+ if (unavailable !== null) return { unavailable };
242
+ // The provider's overlay entry, when the file was read and has one: pi drops all of it if it cannot compose it.
243
+ const providers = overlay?.providers;
244
+ const entry = providers !== null && typeof providers === "object" && Object.hasOwn(providers, ref.provider) ? providers[ref.provider] : undefined;
245
+ if (entry !== undefined && overlayProviderProblem(entry, ref.provider) !== null) return { unknown: { provider: ref.provider, id: ref.id }, why: "overlay-provider-invalid" };
246
+ // `main: true` marks the job's main model, which must be a CHAT model (see `isBuiltinModel`). The overlay's
247
+ // models are chat models: pi registers every models.json entry as one.
248
+ if (isBuiltinModel(ref.provider, ref.id, { chatOnly: ref.main === true })) continue;
249
+ if (!isOverlayModel(overlay, ref.provider, ref.id)) return { unknown: { provider: ref.provider, id: ref.id }, why: "not-in-catalog" };
250
+ }
251
+ // A job WITH a list (any ref beyond the main one): every listed model's server-side fallbacks must be listed too.
252
+ const listed = refs.filter((ref) => ref.main !== true);
253
+ if (listed.length > 0) {
254
+ const allowed = new Set(listed.map((ref) => `${ref.provider}\u0000${ref.id}`));
255
+ for (const ref of listed) {
256
+ for (const fallback of declaredFallbacks(ref.provider, ref.id, overlay)) {
257
+ if (!allowed.has(`${ref.provider}\u0000${fallback}`)) return { fallbackUnlisted: { provider: ref.provider, id: ref.id }, why: "fallback-unlisted" };
258
+ }
259
+ }
260
+ }
261
+ return { ok: true };
262
+ }
263
+
264
+ /**
265
+ * The server-side fallback ids a model declares (`compat.allowedFallbackModels`), as pi 0.99.1 composes them
266
+ * (provider-composer.js): the builtin model's own compat, then the overlay provider entry's `compat`, then an
267
+ * overlay model definition's (which replaces the builtin model of that id, with the provider's compat beneath it),
268
+ * then `modelOverrides[id].compat`; each later layer that sets the key replaces it (`mergeCompat` is shallow for it).
269
+ *
270
+ * Why the worker asks (issue #502, PR #538's review): pi sends these ids with EVERY call on the model
271
+ * (anthropic-messages `params.fallbacks`) and the provider may answer with any of them, so the runner's model guard
272
+ * refuses a listed model whose fallbacks are not listed under its provider. In pi's builtin catalog exactly one model
273
+ * declares any, anthropic/claude-fable-5 (claude-opus-4-8 and claude-opus-5). A list naming it alone would pass every
274
+ * free gate and then have every call refused inside a paid container; refused here instead, before any spend. Read
275
+ * for every api, which is stricter than the runner: only anthropic-messages sends fallbacks, and the runner judges them
276
+ * there only. Accepted (PR #538's review, round 3): a list is the operator saying which models may be named, and the
277
+ * refusal says only that the model declares fallbacks not on the list, which is true on every api.
278
+ */
279
+ export function declaredFallbacks(provider, id, overlay = null) {
280
+ const fallbackIds = (compat) => (Array.isArray(compat?.allowedFallbackModels) ? compat.allowedFallbackModels.map((entry) => entry?.model) : undefined);
281
+ const has = (compat) => compat !== null && typeof compat === "object" && Object.hasOwn(compat, "allowedFallbackModels");
282
+ let found;
283
+ if (getBuiltinProviders().includes(provider)) {
284
+ const model = getAllBuiltinModels(provider).find((m) => m?.id === id);
285
+ if (model) found = fallbackIds(model.compat);
286
+ }
287
+ const providers = overlay?.providers;
288
+ const entry = providers !== null && typeof providers === "object" && !Array.isArray(providers) && Object.hasOwn(providers, provider) ? providers[provider] : null;
289
+ if (entry !== null && typeof entry === "object") {
290
+ if (has(entry.compat)) found = fallbackIds(entry.compat);
291
+ const definition = Array.isArray(entry.models) ? entry.models.find((m) => m !== null && typeof m === "object" && m.id === id) : undefined;
292
+ if (definition !== undefined) found = has(definition.compat) ? fallbackIds(definition.compat) : has(entry.compat) ? fallbackIds(entry.compat) : undefined;
293
+ const override = entry.modelOverrides !== null && typeof entry.modelOverrides === "object" && Object.hasOwn(entry.modelOverrides, id) ? entry.modelOverrides[id] : null;
294
+ if (has(override?.compat)) found = fallbackIds(override.compat);
295
+ }
296
+ return found ?? [];
297
+ }