@edgehero/pi-dispatch 0.1.2 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/prepare.mjs CHANGED
@@ -1,4 +1,4 @@
1
- import { mkdirSync, mkdtempSync } from "node:fs";
1
+ import { mkdirSync, mkdtempSync, rmSync } from "node:fs";
2
2
  import { rm } from "node:fs/promises";
3
3
  import { join } from "node:path";
4
4
  import { resolveJobImage } from "./image-preflight.mjs";
@@ -11,6 +11,17 @@ import { buildGitLabPrompt } from "./gitlab-prompt.mjs";
11
11
  import { buildForgejoPrompt } from "./forgejo-prompt.mjs";
12
12
  import { buildAzurePrompt } from "./azure-prompt.mjs";
13
13
  import { prepareLocalWorkspace } from "./prepare-local.mjs";
14
+ import { copySkillTree } from "./copy-tree.mjs";
15
+
16
+ /**
17
+ * The subdirectory of the per-job dir a trigger's injected skills are copied into, so they reach the
18
+ * container at `/job/trigger-skills` on the `/job:ro` bind that already exists.
19
+ *
20
+ * The runner spells the same last segment from its own side (TRIGGER_SKILLS_DIR in
21
+ * image/runner/src/loader.mjs). It is not on this host and this file is not in that container, so the
22
+ * duplication is forced. CHANGE BOTH, IN THE SAME COMMIT.
23
+ */
24
+ export const TRIGGER_SKILLS_SUBDIR = "trigger-skills";
14
25
 
15
26
  /**
16
27
  * The `prepareWorkspace` dispatcher the processor injects. Creates a per-job dir under `jobsDir`
@@ -45,10 +56,34 @@ export function makePrepareWorkspace({
45
56
  // Keyed by `job.kind`, so a new forge is one entry rather than a new `if`. A kind with no entry falls
46
57
  // through to the throw below, which is what makes an unrouted job loud instead of a silent no-op.
47
58
  preparers = { github: prepareGithubWorkspace },
59
+ // REQ-PER-TRIGGER-SKILLS. Injected so the copy is testable without a real host tree, and defaulted so
60
+ // an unwired dispatcher behaves exactly as it did: a job with no `run.skillsDir` never calls it.
61
+ injectSkills = copySkillTree,
62
+ log = () => {},
48
63
  }) {
49
64
  mkdirSync(jobsDir, { recursive: true });
50
65
  return async function prepareWorkspace(job, token, { queueJobId, piVersion = null } = {}) {
51
66
  const jobDir = mkdtempSync(join(jobsDir, "job-"));
67
+ // The trigger's injected skills (REQ-PER-TRIGGER-SKILLS, issue #60), COPIED here rather than
68
+ // mounted, and copied ONCE for every job kind because this is where local and forge converge.
69
+ //
70
+ // Copied, not bind-mounted, and that is the decision rather than the implementation. `:ro` bounds
71
+ // the CONTAINER, not the host, and pi reads a skill's body on demand through the read tool, so a
72
+ // live bind could change under a running agent -- the copy is what pins the instruction set for the
73
+ // life of the job, exactly as materialising .pi/ at a fixed sha does for the repo's own. It also
74
+ // answers symlinks once, on the side that can (copy-tree.mjs), instead of handing pi a tree whose
75
+ // links it would follow. And it adds NO mount, so CONST-ISOLATION-CONTAINER-PER-JOB's enumeration
76
+ // is untouched -- the same trade DES-OPERATOR-GLOBAL-OVERLAY made for staged packages.
77
+ if (job.skillsDir) {
78
+ const injected = injectSkills(job.skillsDir, join(jobDir, TRIGGER_SKILLS_SUBDIR));
79
+ if (injected?.refused) {
80
+ rmSync(jobDir, { recursive: true, force: true });
81
+ return { outcome: "policy", reason: injected.refused };
82
+ }
83
+ // Counts only. `injected` carries no name and no path by construction, so this line cannot grow
84
+ // a host path by a later edit (the same shape packages.mjs's `dropped` record has).
85
+ log("trigger_skills_injected", { dirs: injected.dirs, files: injected.files, bytes: injected.bytes });
86
+ }
52
87
  // What `cleanup` needs to retain this run's directory, stamped here because this is the only place
53
88
  // that holds all three at once. Applied to the RESULT rather than mutated in, so a preparer's
54
89
  // `{ outcome: "policy" }` refusal -- which carries no jobDir -- is passed through untouched.
@@ -63,23 +98,26 @@ export function makePrepareWorkspace({
63
98
  ? `Use the "${job.flow}" skill for this task.\n\n${pointer}${job.task ?? ""}`
64
99
  : `${pointer}${job.task ?? ""}`;
65
100
  const event = localEventContext(job, queueJobId, findPreviousRun);
66
- return stampSandbox(await prepareLocal({ folder: job.folder, task, jobDir, event }), sandbox);
101
+ return discardOnPolicy(stampSandbox(await prepareLocal({ folder: job.folder, task, jobDir, event }), sandbox), jobDir);
67
102
  }
68
103
  const prepare = preparers[job.kind];
69
104
  if (prepare) {
70
105
  const host = forgeFor?.(job)?.host;
71
- return stampSandbox(
72
- await prepare(job, token, {
73
- jobDir,
74
- resolveDefaultBranchSha: host?.resolveDefaultBranchSha,
75
- // The head ref a pull/merge-request job keys on comes from the FORGE API, never the webhook
76
- // payload: an issue_comment on a PR carries no head at all, and a payload-supplied head repo
77
- // is attacker-controlled data that must not decide which transcript a job is handed.
78
- resolvePullRequestHead: host?.resolvePullRequestHead,
79
- resolveSession,
80
- piVersion,
81
- }),
82
- sandbox,
106
+ return discardOnPolicy(
107
+ stampSandbox(
108
+ await prepare(job, token, {
109
+ jobDir,
110
+ resolveDefaultBranchSha: host?.resolveDefaultBranchSha,
111
+ // The head ref a pull/merge-request job keys on comes from the FORGE API, never the webhook
112
+ // payload: an issue_comment on a PR carries no head at all, and a payload-supplied head repo
113
+ // is attacker-controlled data that must not decide which transcript a job is handed.
114
+ resolvePullRequestHead: host?.resolvePullRequestHead,
115
+ resolveSession,
116
+ piVersion,
117
+ }),
118
+ sandbox,
119
+ ),
120
+ jobDir,
83
121
  );
84
122
  }
85
123
  throw new Error(`unknown job kind: ${job.kind}`);
@@ -131,6 +169,27 @@ function stampSandbox(prepared, sandbox) {
131
169
  return { ...prepared, sandbox };
132
170
  }
133
171
 
172
+ /**
173
+ * Remove the mkdtemp'd job dir when the preparer REFUSED, because nothing downstream will.
174
+ *
175
+ * A determinate refusal carries no `jobDir` (see stampSandbox), and both teardown paths -- `cleanup`
176
+ * and `makeCleanup`'s retention branch -- guard on `prepared?.jobDir`. So the directory this function
177
+ * created two dozen lines up, which by then may hold a partial clone, was simply left on disk: one
178
+ * per refusal, forever. That has been true of `sha-gone` since it shipped and was only ever invisible
179
+ * because refusals are rare; issue #60 adds cap refusals that a misconfigured repo hits on EVERY
180
+ * delivery, which turns a slow leak into a fast one.
181
+ *
182
+ * Deliberately not folded into stampSandbox: that function's job is to decide what a RESULT carries,
183
+ * and a filesystem side effect hidden inside it would be the kind of thing the next reader has to
184
+ * discover. Deliberately `rmSync` rather than the async `rm`, so the directory is gone before the
185
+ * refusal is returned and no teardown ordering has to be reasoned about.
186
+ */
187
+ function discardOnPolicy(prepared, jobDir) {
188
+ if (prepared?.outcome !== "policy") return prepared;
189
+ rmSync(jobDir, { recursive: true, force: true });
190
+ return prepared;
191
+ }
192
+
134
193
  /** Remove a per-job dir after the run. The workspace (the operator's folder) is never touched here. */
135
194
  export async function cleanup(prepared) {
136
195
  if (prepared?.jobDir) await rm(prepared.jobDir, { recursive: true, force: true });
package/src/processor.mjs CHANGED
@@ -1,3 +1,4 @@
1
+ import { lstatSync } from "node:fs";
1
2
  import { checkTokenCap, recordTokenSpend, releaseBudget, reserveBudget } from "./budget.mjs";
2
3
  import { configError } from "./config.mjs";
3
4
  import { EXIT_COMPLETED, EXIT_INFRA, EXIT_POLICY } from "./exit-code.mjs";
@@ -55,6 +56,17 @@ export async function runJob(job, deps) {
55
56
  // cannot disagree about whether a store exists. A wiring may still pass `sessionsDir` explicitly to
56
57
  // make the seam visible; it resolves to the same value.
57
58
  sessionsDir = process.env.PI_SESSIONS_DIR || null,
59
+ // REQ-PER-TRIGGER-SKILLS. Injected so the pre-spend gate is testable without a real directory, and
60
+ // lstat rather than stat so a symlinked skillsDir is judged on its own inode -- the habit copy-tree.mjs,
61
+ // outbox.mjs and sandbox-store.mjs all keep. A throw is a refusal: an unreadable path is still absent
62
+ // as far as this job is concerned.
63
+ isReadableDir = (p) => {
64
+ try {
65
+ return lstatSync(p).isDirectory();
66
+ } catch {
67
+ return false;
68
+ }
69
+ },
58
70
  // (job) => scoped short-lived token. Takes the JOB, not the repo: which forge mints -- and therefore
59
71
  // which credential the container gets -- is a property of `job.kind`, and only the wiring knows the
60
72
  // map. Called for forge-backed jobs and for local jobs opted in via `github: true`; unflagged local
@@ -156,8 +168,8 @@ export async function runJob(job, deps) {
156
168
  }
157
169
 
158
170
  // REQ-RESUMABLE-SESSION's one fail-CLOSED case. Everything else in that feature fails OPEN and
159
- // NAMES itself -- absent, expired, too-large, unparseable, locked, no key -- because a cold start is
160
- // a correct run. This one cannot be: with no `sessionsDir`, resolveSession returns null
171
+ // NAMES itself -- absent, expired, too-large, unparseable, locked, promote-failed -- because a cold
172
+ // start is a correct run. This one cannot be: with no `sessionsDir`, resolveSession returns null
161
173
  // (session-store.mjs), so nothing is staged, no /session is mounted, the transcript dies with the
162
174
  // container, and the NEXT job on that key cold-starts too. The job would exit 0 and look like the
163
175
  // feature worked. That is an operator who believes a disclosure is on while it is off, with a green
@@ -190,6 +202,23 @@ export async function runJob(job, deps) {
190
202
  return { outcome: "policy", reason: "sessions-dir-unset", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
191
203
  }
192
204
 
205
+ // REQ-PER-TRIGGER-SKILLS. A trigger that named a skills directory the worker cannot see would run
206
+ // its flow WITHOUT the skills it was written against, produce a plausible report, and exit 0. Free
207
+ // and determinate -- one lstat, no credential needed to know the answer -- so it belongs among the
208
+ // free refusals and strictly before anything that spends: before the mint (no token is created only
209
+ // to be discarded), before the clone, before the token-cap read and before reserveBudget
210
+ // (CONST-BUDGET-BEFORE-TOKENS). Last among the free gates because it is the NARROWEST: a missing
211
+ // image blocks every job on this host, an unset sessions dir blocks every armed trigger, a bad
212
+ // skillsDir blocks one trigger.
213
+ if (job.skillsDir && !isReadableDir(job.skillsDir)) {
214
+ await comment(job, "Refused: this trigger set `run.skillsDir`, and that path is absent or is not a directory on the worker host. The job would have run without the skills the flow was written against. Not run.");
215
+ // The FIELD name, never its value. `comment` posts publicly on the issue, so a host path here
216
+ // would publish the operator's filesystem layout to anyone reading the thread; the log line is
217
+ // the same restraint refused_sessions_dir_unset keeps.
218
+ log("refused_skills_dir_missing", { kind: job.kind ?? null });
219
+ return { outcome: "policy", reason: "skills-dir-missing", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
220
+ }
221
+
193
222
  if (wantsForgeToken) {
194
223
  token = await mintToken(job);
195
224
 
@@ -216,8 +245,9 @@ export async function runJob(job, deps) {
216
245
 
217
246
  prepared = await prepareWorkspace(job, token, { piVersion }); // resolves SHA, clones, materialises .pi/, writes prompt
218
247
 
219
- // A determinate prepare refusal (e.g. sha-gone: the default branch advanced past the resolved
220
- // tip) is POLICY -- return before reserveBudget so it burns no cap slot and is never retried.
248
+ // A determinate prepare refusal -- sha-gone (the default branch advanced past the resolved tip),
249
+ // or a `pi-*` materialiser cap breach (the repo's .pi/ is too large to place in /job, issue #60)
250
+ // -- is POLICY: return before reserveBudget so it burns no cap slot and is never retried.
221
251
  // Mirrors the branch-protection policy return above. Spread-plus-attribution: the prepare
222
252
  // result keeps its own reason and fields, and the host-effective provider/model land beside
223
253
  // them exactly as on every other terminal result.
package/src/queue.mjs CHANGED
@@ -16,7 +16,7 @@ export function makeQueue(connection) {
16
16
  * removeOnComplete keeps the dedup window ~= the retention. Unlike webhooks, local jobs are not
17
17
  * redelivered, so a modest window is enough.
18
18
  */
19
- export async function enqueueLocalJob(queue, { folder, flow, task, provider, model, maxTurns, image, chainDepth, parentJobId, jobId, now = new Date() }) {
19
+ export async function enqueueLocalJob(queue, { folder, flow, task, provider, model, maxTurns, image, skillsDir, chainDepth, parentJobId, jobId, now = new Date() }) {
20
20
  const minute = now.toISOString().slice(0, 16); // YYYY-MM-DDTHH:MM -- the dedup window
21
21
  // A caller-supplied jobId (the outbox collector's retry-idempotent chainedJobId) wins; otherwise the
22
22
  // minute-windowed localJobId is the dedup key.
@@ -33,6 +33,11 @@ export async function enqueueLocalJob(queue, { folder, flow, task, provider, mod
33
33
  model,
34
34
  maxTurns,
35
35
  ...(image !== undefined && { image }),
36
+ // The host directory of operator-authored skills this trigger injects (REQ-PER-TRIGGER-SKILLS).
37
+ // Conditional like `image`, so an unflagged job's data stays byte-identical, and at JOB level rather
38
+ // than inside `trigger` because a worker-host path is an execution knob, not a fact about the
39
+ // delivery -- and `trigger` is the object copied into /job/event.json.
40
+ ...(skillsDir !== undefined && { skillsDir }),
36
41
  ...(chainDepth !== undefined && { chainDepth }),
37
42
  ...(parentJobId !== undefined && { parentJobId }),
38
43
  };
@@ -110,7 +115,7 @@ export async function enqueueGitLabJob(queue, fields) {
110
115
  * window, replicas never coalesce against each other, and an unflagged job's dedup id is the same string it
111
116
  * has always been.
112
117
  */
113
- export async function enqueueForgeJob(queue, kind, { repo, projectId, azure, target, flow, trigger, provider, model, maxTurns, packages, image, resume, replica, replicas }) {
118
+ export async function enqueueForgeJob(queue, kind, { repo, projectId, azure, target, flow, trigger, provider, model, maxTurns, packages, image, skillsDir, instructions, resume, replica, replicas }) {
114
119
  const jobId = forgeDeliveryJobId(kind, trigger?.deliveryId, replica);
115
120
  // `packages` (whether to load the operator-staged pi packages) and `image` (which container image to run)
116
121
  // come off the MATCHED trigger (INT-TRIGGERS-FILE-CONTRACT / REQ-GLOBAL-PI-OVERLAY) and land on `data`
@@ -133,6 +138,15 @@ export async function enqueueForgeJob(queue, kind, { repo, projectId, azure, tar
133
138
  maxTurns,
134
139
  ...(packages !== undefined && { packages }),
135
140
  ...(image !== undefined && { image }),
141
+ // The host directory of operator-authored skills this trigger injects (REQ-PER-TRIGGER-SKILLS).
142
+ // Conditional like `image`, so an unflagged job's data stays byte-identical, and at JOB level rather
143
+ // than inside `trigger` because a worker-host path is an execution knob, not a fact about the
144
+ // delivery -- and `trigger` is the object copied into /job/event.json.
145
+ ...(skillsDir !== undefined && { skillsDir }),
146
+ // The operator's standing instruction for this trigger (REQ-PER-TRIGGER-INSTRUCTION). Conditional like
147
+ // the rest, and at JOB level rather than inside `trigger`: it is operator config, not a fact about the
148
+ // delivery, and `trigger` is what /job/event.json is built from.
149
+ ...(instructions !== undefined && { instructions }),
136
150
  ...(resume !== undefined && { resume }),
137
151
  // Conditional for the same reason packages/image/resume are: an unflagged job's data must keep
138
152
  // exactly the keys it has today. `replica` is this job's 1-based index and `replicas` the set size;
@@ -33,7 +33,10 @@ export function makeRunContainer({
33
33
  spawnFn = spawn,
34
34
  globalPiDir = null, // REQ-GLOBAL-PI-OVERLAY: operator's global pi overlay dir, mounted :ro; null = off
35
35
  allowGlobalExtensions = true, // REQ-GLOBAL-PI-OVERLAY: the staged overlay's extensions load unless PI_GLOBAL_ALLOW_EXTENSIONS=0
36
- packagePaths = [], // REQ-GLOBAL-PI-OVERLAY: container paths of the operator-staged packages, resolved once at boot
36
+ // REQ-GLOBAL-PI-OVERLAY: container paths of the operator-staged packages. An array, or a RESOLVER called
37
+ // once per job (issue #102): the wired worker passes a resolver so a re-stage lands on the next job with
38
+ // no restart, while the array form stays valid for every caller that has a fixed set.
39
+ packagePaths = [],
37
40
  forwardEnv = [],
38
41
  authFromPi = false, // fall back to ~/.pi/agent/auth.json for the provider key when the env has none
39
42
  forgeHosts = {}, // per-forge self-hosted instance URLs, so a forge CLI in the container talks to the right one
@@ -63,7 +66,9 @@ export function makeRunContainer({
63
66
  // (INT-TRIGGERS-FILE-CONTRACT). The strictness that used to live in this `=== true` did not
64
67
  // disappear, it moved: parseTriggers refuses any non-boolean run.packages fail-loud at load, so a
65
68
  // hand-edited string "false" never becomes job data this comparison could misread as an opt-out.
66
- packagePaths: job.packages === false ? [] : packagePaths,
69
+ // The opt-out short-circuits BEFORE the resolver runs: a trigger that withheld the staged set has no
70
+ // reason to make the worker read the manifest on its behalf.
71
+ packagePaths: job.packages === false ? [] : typeof packagePaths === "function" ? packagePaths() : packagePaths,
67
72
  forwardEnv, // extra host var names to forward (e.g. a custom provider's key)
68
73
  // REQ-RESUMABLE-SESSION: the fixed container path, emitted only when this job HAS a transcript.
69
74
  // The constant is imported rather than re-typed so the mount below and this variable name one
@@ -295,6 +295,19 @@ export function buildRecord({ job, result, error, startedAt, endedAt }) {
295
295
  // deliberately not stored, for the reason `session` states one group below.
296
296
  replica: data.replica ?? null,
297
297
  replicas: data.replicas ?? null,
298
+ // Trigger attribution (INT-RUN-HISTORY-FILE-CONTRACT, issue #54): additive and nullable, explicit
299
+ // literals beside the replica fields whose admissibility argument they reuse, no spread. Both read
300
+ // the receiver's harness-computed `matched` from this job's own `job.data.trigger`: `index` is the
301
+ // raw triggers-array position of the entry that fired (cron entries counted) and `type` that entry's
302
+ // `on.type` -- an INTEGER and a FIXED ENUM ("label" | "comment" | "pull_request"), nothing
303
+ // attacker-chosen. The third `matched` key (`label`/`phrase`/`action`) is DELIBERATELY absent:
304
+ // a label that satisfied an `any` predicate is collaborator-applied payload text, and `type`
305
+ // already names the route. Cron jobs carry `trigger: { id, pattern }` with no `matched`, so both
306
+ // stay null there on purpose -- a cron run's attribution is already exact via its
307
+ // `repeat:<id>:<millis>` jobId (see makeFindPreviousRun), and that join also works retroactively
308
+ // over the whole retention window, which a new record field cannot.
309
+ triggerIndex: data.trigger?.matched?.index ?? null,
310
+ triggerType: data.trigger?.matched?.type ?? null,
298
311
  // Session telemetry (INT-RUN-HISTORY-FILE-CONTRACT): additive, nullable, an explicit literal, no
299
312
  // spread. `{ resumed, reason, bytes }` -- a boolean, a fixed enum and an integer. THE KEY AND THE
300
313
  // BRANCH NAME ARE DELIBERATELY ABSENT: this record's PII-free-by-construction property rests on it
package/src/schedules.mjs CHANGED
@@ -13,6 +13,7 @@
13
13
  */
14
14
 
15
15
  import { existsSync as fsExistsSync, readFileSync as fsReadFileSync } from "node:fs";
16
+ import { isAbsolute } from "node:path";
16
17
  import { configError } from "./config.mjs";
17
18
  import { parseTriggers } from "./triggers.mjs";
18
19
 
@@ -42,6 +43,20 @@ function normalizeCronSchedule({ on, run }, path, existsSync) {
42
43
  throw configError(`cron trigger "${on.id}": run.folder does not exist: ${run.folder} (${path})`);
43
44
  }
44
45
 
46
+ // `run.skillsDir` gets the same treatment, and for the same reason (REQ-PER-TRIGGER-SKILLS): the pure
47
+ // validator cannot check a host path, because the RECEIVER parses the same file and may run on another
48
+ // machine entirely. Absoluteness is checked here rather than there for a second reason -- `isAbsolute`
49
+ // is OS-dependent, so a shared check would let a Windows worker and a Linux receiver disagree about the
50
+ // same reviewed file. A broken cron trigger refuses the worker's BOOT rather than failing at 03:00.
51
+ if (run.skillsDir !== undefined) {
52
+ if (!isAbsolute(run.skillsDir)) {
53
+ throw configError(`cron trigger "${on.id}": run.skillsDir must be an absolute path: ${run.skillsDir} (${path})`);
54
+ }
55
+ if (!existsSync(run.skillsDir)) {
56
+ throw configError(`cron trigger "${on.id}": run.skillsDir does not exist: ${run.skillsDir} (${path})`);
57
+ }
58
+ }
59
+
45
60
  // Absent provider/model/maxTurns stay absent (undefined) so the value resolves at job start against the
46
61
  // settings overlay/env, not a default frozen here (INT-CONFIG-OVERLAY-CONTRACT). data key order matches
47
62
  // queue.mjs -- the shape the processor's runJob consumes. The three per-trigger fields ride along the same
@@ -53,7 +68,7 @@ function normalizeCronSchedule({ on, run }, path, existsSync) {
53
68
  // cron-only field: it is carried into the local `/job/event.json` (INT-CONTAINER-JOB-INPUTS) so a
54
69
  // scheduled job can name its own trigger; the INT-TRIGGERS-FILE-CONTRACT byte-match acceptance is
55
70
  // amended for exactly this field.
56
- const data = { kind: "local", folder: run.folder, flow: run.flow, task: run.task, provider: run.provider, model: run.model, maxTurns: run.maxTurns, github: run.github, packages: run.packages, image: run.image, resume: run.resume, trigger: { id: on.id, pattern: on.pattern } };
71
+ const data = { kind: "local", folder: run.folder, flow: run.flow, task: run.task, provider: run.provider, model: run.model, maxTurns: run.maxTurns, github: run.github, packages: run.packages, image: run.image, ...(run.skillsDir !== undefined && { skillsDir: run.skillsDir }), resume: run.resume, trigger: { id: on.id, pattern: on.pattern } };
57
72
  // Retention only; the deterministic repeat:<id>:<millis> jobId supplies dedup, so no jobId here, and
58
73
  // scheduler jobs are not retried (DES-CRON-VIA-BULLMQ-SCHEDULER) so no attempts/backoff.
59
74
  const opts = { removeOnComplete: { age: 24 * 3600 }, removeOnFail: { age: 7 * 24 * 3600 } };
@@ -115,6 +115,12 @@ export function makeSessionStore({
115
115
  * agents' turns into one transcript.
116
116
  */
117
117
  function promoteSession(session, { piVersion = null } = {}) {
118
+ // The second DI-seam backstop, and unreachable for the same reason as the `!sessionsDir` return
119
+ // above: sessionKeyFor is total and binary (null, or 32 hex chars), so resolveSession returns null
120
+ // rather than a keyless session, and processor.mjs only calls this when prepare handed it one. Kept
121
+ // because the store and the preparer are separately injected and neither can assume the other. It is
122
+ // NOT in INT-RUN-HISTORY-FILE-CONTRACT's session.reason enum, deliberately: a token no wired worker
123
+ // can emit does not belong in the record's vocabulary, and `promote-failed` below does.
118
124
  if (!session?.key) return { promoted: false, reason: "no-key" };
119
125
  try {
120
126
  const staged = join(session.hostDir, SESSION_FILE_NAME);
package/src/start.mjs CHANGED
@@ -320,14 +320,51 @@ export async function startWorker(
320
320
  // (CONST-RETRY-INFRA-ONLY). The processor calls it as the sole COMPLETED-path chain step.
321
321
  const collectChain = makeCollectChain({ queue: runtimeQueue, config, log });
322
322
 
323
- // REQ-GLOBAL-PI-OVERLAY staged packages: read the operator's stage manifest ONCE at boot. The staged set
324
- // is deploy-time state under the :ro overlay -- identical for every job -- so a per-job re-read would buy
325
- // nothing and put a filesystem read on the hot path. A missing or unreadable manifest yields [] plus one
326
- // log line and NEVER a boot failure: a deployment that never opted into packages must not be blocked by
327
- // it, and `pi-dispatch doctor` is what fails loud on a mismatch between the overlay and the triggers.
328
- const stagedPackages = config.globalPiDir ? readStageManifest({ globalPiDir: config.globalPiDir }) : null;
329
- const packagePaths = stagedPackages ? containerPackagePaths(stagedPackages) : [];
330
- if (config.globalPiDir && !stagedPackages) log("packages_manifest_absent", { overlay: config.globalPiDir });
323
+ // REQ-GLOBAL-PI-OVERLAY staged packages: read the operator's stage manifest at EACH job start, like
324
+ // getSettings above and the pause-window ref below.
325
+ //
326
+ // This was a boot-time read until issue #102, and the argument for that was sound while it held: the
327
+ // staged set was deploy-time state under a :ro mount, identical for every job, so a per-job read bought
328
+ // nothing. What changed is that `import-pi --with-packages` now discovers what the operator installed in
329
+ // pi, which makes `pi install X` then re-stage a ROUTINE act rather than a rare one. Under the boot read
330
+ // the jobs after such a re-stage keep the old set until someone restarts the worker, and when the re-stage
331
+ // DROPS a package the symptom is worse than staleness: the runner refuses a missing staged dir at
332
+ // container start (exit 2), and budget is reserved before the container, so every job burns a daily-cap
333
+ // slot until the restart. A free filesystem read that prevents a reserved-and-wasted slot is exactly what
334
+ // CONST-BUDGET-BEFORE-TOKENS asks for.
335
+ //
336
+ // Last-known-good on a failed read, never []: an empty set emits no PI_PACKAGES at all, so the runner's
337
+ // assertPackagePathsExist has nothing to refuse and the job would run WITHOUT its tools and still exit 0.
338
+ // That is the silent no-op this project refuses. And never a throw: a transient overlay fault must not
339
+ // become a queue retry (CONST-RETRY-INFRA-ONLY).
340
+ let lastGoodPackagePaths = [];
341
+ let lastPackageKey = null;
342
+ // Logged once per CHANGE, not once per job: a line every job would drown the log it is meant to serve.
343
+ // EVERY resolved read records its key, including the empty one, so "nothing staged" becoming "one package
344
+ // staged" is the change it obviously is rather than a first read that logs nothing.
345
+ const notePackageKey = (key) => {
346
+ if (lastPackageKey !== null && key !== lastPackageKey) log("packages_stage_changed", { count: key === "" ? 0 : key.split(":").length });
347
+ lastPackageKey = key;
348
+ };
349
+ const getPackagePaths = () => {
350
+ if (!config.globalPiDir) return [];
351
+ const staged = readStageManifest({ globalPiDir: config.globalPiDir });
352
+ if (!staged) {
353
+ if (lastGoodPackagePaths.length > 0) {
354
+ log("packages_manifest_unreadable", { overlay: config.globalPiDir, keeping: lastGoodPackagePaths.length });
355
+ return lastGoodPackagePaths;
356
+ }
357
+ notePackageKey("");
358
+ return [];
359
+ }
360
+ const paths = containerPackagePaths(staged);
361
+ notePackageKey(paths.join(":"));
362
+ lastGoodPackagePaths = paths;
363
+ return paths;
364
+ };
365
+ // One read at boot, for the same log line the boot read always emitted, and to seed last-known-good.
366
+ if (config.globalPiDir && !readStageManifest({ globalPiDir: config.globalPiDir })) log("packages_manifest_absent", { overlay: config.globalPiDir });
367
+ getPackagePaths();
331
368
 
332
369
  const worker = createWorkerFn({
333
370
  connection: parseConnection(config.valkeyUrl),
@@ -363,7 +400,10 @@ export async function startWorker(
363
400
  openJobLog,
364
401
  globalPiDir: config.globalPiDir, // REQ-GLOBAL-PI-OVERLAY: :ro overlay mount when configured
365
402
  allowGlobalExtensions: config.allowGlobalExtensions,
366
- packagePaths, // REQ-GLOBAL-PI-OVERLAY: staged package paths; every job receives them unless its trigger set packages:false
403
+ // REQ-GLOBAL-PI-OVERLAY: staged package paths; every job receives them unless its trigger set
404
+ // packages:false. A RESOLVER, not the array: the factory is still constructed exactly once, only
405
+ // the value it reads became a call, so a re-stage takes effect on the next job without a restart.
406
+ packagePaths: getPackagePaths,
367
407
  forwardEnv: config.forwardEnv,
368
408
  authFromPi: config.authFromPi, // source the provider key from ~/.pi/agent/auth.json when env has none
369
409
  // Self-hosted instance URLs, keyed by forge. A MAP rather than one scalar per forge: the table says