omp-conductor 0.15.11 → 0.15.12

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/doctor.ts CHANGED
@@ -462,7 +462,7 @@ async function labelProbe(probes: Probes, project: ProjectConfig): Promise<Findi
462
462
  if (!read.ok) {
463
463
  return failFinding(
464
464
  "labels",
465
- `cannot read the labels of ${project.tracker.repo}${read.detail === undefined ? "" : ` (${read.detail})`}`,
465
+ `[${project.name}] cannot read the labels of ${project.tracker.repo}${read.detail === undefined ? "" : ` (${read.detail})`}`,
466
466
  "check `gh` can list that repo's labels — the daemon reads them the same way",
467
467
  );
468
468
  }
@@ -474,7 +474,7 @@ async function labelProbe(probes: Probes, project: ProjectConfig): Promise<Findi
474
474
  }
475
475
  const missing = wanted.filter((w) => !present.has(w.name));
476
476
  if (missing.length === 0) {
477
- return passFinding("labels", `${wanted.length} configured label(s) on ${project.tracker.repo} match exactly`);
477
+ return passFinding("labels", `[${project.name}] ${wanted.length} configured label(s) on ${project.tracker.repo} match exactly`);
478
478
  }
479
479
  const detail = missing.map((m) => {
480
480
  const near = nearByName.get(norm(m.name));
@@ -486,17 +486,60 @@ async function labelProbe(probes: Probes, project: ProjectConfig): Promise<Findi
486
486
  });
487
487
  return failFinding(
488
488
  "labels",
489
- `${detail.join("; ")} — ${INCIDENTS.labels}`,
489
+ `[${project.name}] ${detail.join("; ")} — ${INCIDENTS.labels}`,
490
490
  allNearMisses
491
491
  ? "fix the label case/format in the tracker repo (or change the config to the actual spelling)"
492
492
  : "create the missing label(s) in the tracker repo (`gh label create`, or re-run `omp-conductor setup`)",
493
493
  );
494
494
  }
495
495
 
496
- /** Line-level drift between an installed unit and the canonical render. */
496
+ /**
497
+ * A unit line whose canonical value is a property of the *rendering process*,
498
+ * not the deployment. The herdr unit's `Environment="SHELL=…"` carries the
499
+ * account's login shell as it resolved in the process that staged it (#463 —
500
+ * the pane-shell pin that keeps panes off dash). Doctor cannot re-derive that
501
+ * value: in a non-login/tick context `userInfo().shell` resolves to nothing and
502
+ * renders `SHELL=unknown`, while the unit that was actually staged carries
503
+ * `SHELL=/bin/bash`. Comparing against a value that answers to "whoever asked"
504
+ * can never pass, so it is excluded from the drift comparison and reported as
505
+ * not-comparable, never as drift (#511).
506
+ */
507
+ const PROCESS_DERIVED_UNIT_LINE = /^Environment="SHELL=/;
508
+
509
+ /**
510
+ * Whether a canonical unit's pane-shell value could not be resolved — the
511
+ * `SHELL=` pin rendered when the invoking process has no login shell (an empty
512
+ * value, or the literal `unknown` userInfo returns in a non-login context).
513
+ * Excluded from the drift comparison, such a value is reported as
514
+ * not-comparable rather than implied to match (#511).
515
+ */
516
+ function unresolvableShell(unit: string): boolean {
517
+ const line = unit
518
+ .split("\n")
519
+ .map((l) => l.trim())
520
+ .find((l) => PROCESS_DERIVED_UNIT_LINE.test(l));
521
+ if (line === undefined) return false;
522
+ const value = /^Environment="SHELL=(.*)"$/.exec(line)?.[1] ?? "";
523
+ return value === "" || value === "unknown";
524
+ }
525
+
526
+ /**
527
+ * Line-level drift between an installed unit and the canonical render, ignoring
528
+ * lines whose canonical value is not re-derivable by doctor (the
529
+ * {@link PROCESS_DERIVED_UNIT_LINE} pane-shell pin). The pane shell is host
530
+ * state resolved at staging time, so a rendered value doctor cannot reproduce
531
+ * is excluded from both sides rather than reported as a difference — an
532
+ * unresolvable canonical is "cannot compare", never drift (#511).
533
+ */
497
534
  export function unitDrift(installed: string, canonical: string): { differences: string[] } {
498
- const want = canonical.split("\n").map((l) => l.trim()).filter((l) => l.length > 0);
499
- const have = installed.split("\n").map((l) => l.trim()).filter((l) => l.length > 0);
535
+ const want = canonical
536
+ .split("\n")
537
+ .map((l) => l.trim())
538
+ .filter((l) => l.length > 0 && !PROCESS_DERIVED_UNIT_LINE.test(l));
539
+ const have = installed
540
+ .split("\n")
541
+ .map((l) => l.trim())
542
+ .filter((l) => l.length > 0 && !PROCESS_DERIVED_UNIT_LINE.test(l));
500
543
  const wantSet = new Set(want);
501
544
  const haveSet = new Set(have);
502
545
  const differences = [
@@ -516,8 +559,9 @@ function driftLines(name: string, installed: string, canonical: string): string[
516
559
 
517
560
  /**
518
561
  * Installed unit vs the canonical staged rendering: any drift in User=,
519
- * Environment, WorkingDirectory, ExecStart, Restart/SuccessExitStatus, or
520
- * memory lines is the unit-drift class that once killed workers at turn 0.
562
+ * Environment (save the process-derived pane shell, which is not comparable),
563
+ * WorkingDirectory, ExecStart, Restart/SuccessExitStatus, or memory lines is
564
+ * the unit-drift class that once killed workers at turn 0.
521
565
  */
522
566
  function unitProbe(probes: Probes, project: ProjectConfig | undefined, cfg: ConductorConfig | undefined): Finding {
523
567
  if (!probes.hasSystemd()) return passFinding("systemd-unit", "no systemd directory on this host — nothing to check");
@@ -530,16 +574,21 @@ function unitProbe(probes: Probes, project: ProjectConfig | undefined, cfg: Cond
530
574
  return warnFinding("systemd-unit", `${STAGED_SERVICE_NAME} is not installed`, "install it: run `omp-conductor setup host` from the fleet account");
531
575
  }
532
576
  const problems = driftLines(STAGED_SERVICE_NAME, installedDaemon, canonical.daemon);
577
+ const uncomparable: string[] = [];
533
578
  if (canonical.herdr !== undefined) {
534
579
  const installedHerdr = probes.readUnit(join(SYSTEMD_UNIT_DIR, DEFAULT_HERDR_UNIT));
535
580
  if (installedHerdr === undefined) {
536
581
  problems.push(`${DEFAULT_HERDR_UNIT} is not installed (the staged plan provisions it)`);
537
582
  } else {
583
+ if (unresolvableShell(canonical.herdr)) {
584
+ uncomparable.push("the pane-shell SHELL value is not comparable (it resolves from this process, not the deployment)");
585
+ }
538
586
  problems.push(...driftLines(DEFAULT_HERDR_UNIT, installedHerdr, canonical.herdr));
539
587
  }
540
588
  }
541
589
  if (problems.length === 0) {
542
- return passFinding("systemd-unit", "installed units match the staged render");
590
+ const note = uncomparable.length === 0 ? "" : ` — ${uncomparable.join("; ")}`;
591
+ return passFinding("systemd-unit", `installed units match the staged render${note}`);
543
592
  }
544
593
  return failFinding(
545
594
  "systemd-unit",
@@ -548,6 +597,46 @@ function unitProbe(probes: Probes, project: ProjectConfig | undefined, cfg: Cond
548
597
  );
549
598
  }
550
599
 
600
+ /** The `OnFailure=` targets an installed unit names, each a systemd unit file. */
601
+ function onFailureTargets(unitText: string): string[] {
602
+ return (unitText.match(/^OnFailure=(.*)$/gm) ?? []).flatMap((line) =>
603
+ line.slice("OnFailure=".length).trim().split(/\s+/).filter(Boolean),
604
+ );
605
+ }
606
+
607
+ /**
608
+ * Every `OnFailure=` target an installed fleet unit names must itself be
609
+ * installed. The fleet units carry the line from day one (#485); the failure
610
+ * this check exists for is the line present while the recovery oneshot it
611
+ * names is not installed — so systemd silently refuses to enqueue the job and
612
+ * a failed unit never recovers (#509).
613
+ */
614
+ function recoveryProbe(probes: Probes): Finding {
615
+ if (!probes.hasSystemd()) return passFinding("systemd-recovery", "no systemd directory on this host — nothing to check");
616
+ const problems: string[] = [];
617
+ const fleetUnits: readonly [string, string][] = [
618
+ [STAGED_SERVICE_NAME, join(SYSTEMD_UNIT_DIR, STAGED_SERVICE_NAME)],
619
+ [DEFAULT_HERDR_UNIT, join(SYSTEMD_UNIT_DIR, DEFAULT_HERDR_UNIT)],
620
+ ];
621
+ for (const [name, path] of fleetUnits) {
622
+ const text = probes.readUnit(path);
623
+ if (text === undefined) continue; // not installed — the unit/ownership probes name it
624
+ for (const target of onFailureTargets(text)) {
625
+ if (probes.readUnit(join(SYSTEMD_UNIT_DIR, target)) === undefined) {
626
+ problems.push(`${name} names OnFailure=${target}, which is not installed`);
627
+ }
628
+ }
629
+ }
630
+ if (problems.length === 0) {
631
+ return passFinding("systemd-recovery", "every OnFailure= target names an installed unit");
632
+ }
633
+ return failFinding(
634
+ "systemd-recovery",
635
+ `${problems.join("; ")} — systemd cannot enqueue the recovery a failed fleet unit needs (#485/#509)`,
636
+ "install the recovery unit and playbook: run `omp-conductor setup host` from the fleet account",
637
+ );
638
+ }
639
+
551
640
  function installedUnitUser(probes: Probes): string | undefined {
552
641
  const unit = probes.readUnit(join(SYSTEMD_UNIT_DIR, STAGED_SERVICE_NAME));
553
642
  if (unit === undefined) return undefined;
@@ -609,13 +698,13 @@ function timezoneProbe(project: ProjectConfig | undefined): Finding {
609
698
  return passFinding(
610
699
  "reporting-timezone",
611
700
  configured.length === 0
612
- ? "no reporting timezone configured"
613
- : `reporting timezone(s) valid: ${configured.map(([, tz]) => `"${tz}"`).join(", ")}`,
701
+ ? `[${project.name}] no reporting timezone configured`
702
+ : `[${project.name}] reporting timezone(s) valid: ${configured.map(([, tz]) => `"${tz}"`).join(", ")}`,
614
703
  );
615
704
  }
616
705
  return failFinding(
617
706
  "reporting-timezone",
618
- `${bad.map(([label, tz]) => `${label}: "${tz}" is not a known IANA timezone`).join("; ")} — an invalid zone silently skips the availability window and the daily digest`,
707
+ `[${project.name}] ${bad.map(([label, tz]) => `${label}: "${tz}" is not a known IANA timezone`).join("; ")} — an invalid zone silently skips the availability window and the daily digest`,
619
708
  "set a valid IANA timezone (e.g. Europe/London) in reporting.availability or reporting.digest",
620
709
  );
621
710
  }
@@ -651,11 +740,11 @@ async function telegramProbe(probes: Probes, project: ProjectConfig | undefined,
651
740
  if (project === undefined) return passFinding("telegram", "no project resolved — nothing to check");
652
741
  const chatId = (project.escalation.telegramChatId ?? "").trim();
653
742
  if (chatId === "") {
654
- return passFinding("telegram", "no escalation.telegramChatId configured — nothing to probe");
743
+ return passFinding("telegram", `[${project.name}] no escalation.telegramChatId configured — nothing to probe`);
655
744
  }
656
745
  const health = await probes.telegramHealth(project.name);
657
746
  const failures: string[] = [];
658
- const notes: string[] = [];
747
+ const notes: string[] = [`[${project.name}]`];
659
748
  if (health.kind === "down") failures.push(`bot health: down (${health.detail ?? "getMe failed"})`);
660
749
  else if (health.kind === "unconfigured") notes.push(`bot health: unconfigured (${health.detail ?? "no token"})`);
661
750
  else if (health.kind === "degraded") notes.push(`bot health: degraded (${health.detail ?? ""})`);
@@ -678,7 +767,7 @@ async function telegramProbe(probes: Probes, project: ProjectConfig | undefined,
678
767
  }
679
768
  return failFinding(
680
769
  "telegram",
681
- failures.join("; "),
770
+ `[${project.name}] ${failures.join("; ")}`,
682
771
  "fix the bot token / omp-telegram install; `doctor --probe-telegram` proves delivery end to end",
683
772
  );
684
773
  }
@@ -706,7 +795,13 @@ function spendProbe(rows: RunSpendRow[], limit: number): Finding {
706
795
  /**
707
796
  * Run every probe and assemble the stable report.
708
797
  *
709
- * @param projectName the `--project NAME` value (undefined on single-project hosts)
798
+ * Host-wide facts (config backup, store integrity, gh auth, systemd units,
799
+ * ownership, spend telemetry) are probed once; the per-project facts (labels,
800
+ * reporting timezone, telegram) are probed for every resolved project, each
801
+ * named (#530).
802
+ *
803
+ * @param projectName the `--project NAME` value; undefined checks every
804
+ * configured project
710
805
  * @param opts injected seams; every probe defaults to the production wiring
711
806
  */
712
807
  export async function runDoctor(projectName: string | undefined, opts: DoctorDeps = {}): Promise<DoctorReport> {
@@ -722,42 +817,79 @@ export async function runDoctor(projectName: string | undefined, opts: DoctorDep
722
817
  configProblem = messageOf(err);
723
818
  }
724
819
 
725
- let project: ProjectConfig | undefined;
820
+ // Resolve the project set: the named project when `--project NAME` is given,
821
+ // otherwise *(every)* configured project. Doctor is fleet-scoped and covers
822
+ // the whole host, so a no-name run checks every project's per-project facts
823
+ // once each, not the first one and not nothing (#530).
824
+ let projects: ProjectConfig[] = [];
726
825
  let projectProblem: string | undefined;
727
826
  if (cfg !== undefined) {
728
- try {
729
- project = findProject(cfg, projectName);
730
- } catch (err) {
731
- projectProblem = messageOf(err);
827
+ if (projectName !== undefined) {
828
+ try {
829
+ projects = [findProject(cfg, projectName)];
830
+ } catch (err) {
831
+ projectProblem = messageOf(err);
832
+ }
833
+ } else {
834
+ projects = cfg.projects;
732
835
  }
733
836
  }
734
837
 
735
838
  const findings: Finding[] = [];
736
839
  findings.push(configProbe(configProblem));
737
- findings.push(
738
- cfg === undefined
739
- ? passFinding("project", "config unreadable — no project to resolve")
740
- : project === undefined
741
- ? failFinding(
742
- "project",
743
- `no project resolved: ${projectProblem ?? "unknown error"}`,
744
- "pass --project NAME (doctor checks one project per run)",
745
- )
746
- : passFinding("project", project.name),
747
- );
840
+ if (cfg === undefined) {
841
+ findings.push(passFinding("project", "config unreadable — no project to resolve"));
842
+ } else if (projectProblem !== undefined) {
843
+ findings.push(
844
+ failFinding(
845
+ "project",
846
+ `no project resolved: ${projectProblem}`,
847
+ "pass --project NAME (doctor checks the named project; with no name it checks every configured project)",
848
+ ),
849
+ );
850
+ } else if (projects.length === 1) {
851
+ findings.push(passFinding("project", projects[0]!.name));
852
+ } else {
853
+ findings.push(passFinding("project", `${projects.map((p) => p.name).join(", ")} (${projects.length} projects)`));
854
+ }
855
+
856
+ // Host-wide facts, probed once per run — never once per project (the daemon
857
+ // unit, the store and the token are shared across projects, so re-probing
858
+ // them per project and deduping the output would hide a real per-project
859
+ // difference behind a constant count).
748
860
  findings.push(backupProbe(probes));
749
861
  findings.push(dbProbe(probes));
750
862
  findings.push(await ghAuthProbe(probes, configuredRepos(cfg)));
751
- findings.push(
752
- project === undefined
753
- ? passFinding("labels", projectProblem === undefined ? "no project resolved — nothing to check" : `labels uncheckable: ${projectProblem}`)
754
- : await labelProbe(probes, project),
755
- );
756
- findings.push(unitProbe(probes, project, cfg));
863
+ // The collected run sample across the resolved project set, newest-first;
864
+ // spend telemetry is a single host-wide finding, not one per project.
865
+ const spendRows: RunSpendRow[] = [];
866
+ for (const p of projects) spendRows.push(...probes.recentRuns(p.name, SPEND_SAMPLE_RUNS));
867
+
868
+ // Per-project probes run for every resolved project, each named in its
869
+ // finding so a multi-project run stays legible. When nothing resolved
870
+ // (config unreadable, or --project named an unknown project), a no-op pass
871
+ // keeps the stable report shape rather than silently dropping the row.
872
+ if (projects.length === 0) {
873
+ findings.push(
874
+ passFinding("labels", projectProblem === undefined ? "no project resolved — nothing to check" : `labels uncheckable: ${projectProblem}`),
875
+ );
876
+ } else {
877
+ for (const p of projects) findings.push(await labelProbe(probes, p));
878
+ }
879
+ // The installed-unit check is host-global: one shared daemon. Any project
880
+ // renders the same canonical units (the daemon/units carry no project), so
881
+ // the first one stands in for the rendering seam.
882
+ findings.push(unitProbe(probes, projects[0], cfg));
883
+ findings.push(recoveryProbe(probes));
757
884
  findings.push(ownershipProbe(probes));
758
- findings.push(timezoneProbe(project));
759
- findings.push(await telegramProbe(probes, project, checkedAt));
760
- findings.push(spendProbe(project === undefined ? [] : probes.recentRuns(project.name, SPEND_SAMPLE_RUNS), SPEND_SAMPLE_RUNS));
885
+ if (projects.length === 0) {
886
+ findings.push(timezoneProbe(undefined));
887
+ findings.push(await telegramProbe(probes, undefined, checkedAt));
888
+ } else {
889
+ for (const p of projects) findings.push(timezoneProbe(p));
890
+ for (const p of projects) findings.push(await telegramProbe(probes, p, checkedAt));
891
+ }
892
+ findings.push(spendProbe(spendRows, SPEND_SAMPLE_RUNS));
761
893
 
762
894
  const status: ReportStatus = findings.some((f) => f.status === "fail")
763
895
  ? "fail"
@@ -765,7 +897,12 @@ export async function runDoctor(projectName: string | undefined, opts: DoctorDep
765
897
  ? "warn"
766
898
  : "pass";
767
899
  return {
768
- project: project === undefined ? projectName ?? "(none resolved)" : project.name,
900
+ project:
901
+ projects.length === 0
902
+ ? projectName ?? "(none resolved)"
903
+ : projects.length === 1
904
+ ? projects[0]!.name
905
+ : projects.map((p) => p.name).join(", "),
769
906
  checkedAt: checkedAtIso,
770
907
  status,
771
908
  findings,
@@ -222,6 +222,27 @@ export function dispatchInfra(
222
222
  // front of the thrown message, so a bare /^git / would never match a real row.
223
223
  const DISPATCH_GIT_ERROR = /^(?:Error: )?git .+ exited \d+/s;
224
224
 
225
+ /** The prefix the draining/restarting process writes to a run it killed. */
226
+ const ADMIN_RESTART_MARKER = "admin restart:";
227
+
228
+ /**
229
+ * The actor a draining restart recorded on a run it killed, or `undefined`
230
+ * when the row carries no such attribution. Read from `lastError` for the
231
+ * same reason {@link dispatchInfra} is read from it: it is the one surface a
232
+ * run's own row already persists, and the restarting process writes it the
233
+ * way `recordOperatorStop` writes "operator stopped: …". Recognised as an
234
+ * administrative kill, never a charging failure, because nothing the worker
235
+ * did ended it — #512's gap was that a killed session left no attribution.
236
+ */
237
+ export function adminRestartAttribution(lastError: string | undefined): string | undefined {
238
+ if (lastError === undefined) return undefined;
239
+ const marker = lastError.indexOf(ADMIN_RESTART_MARKER);
240
+ if (marker === -1) return undefined;
241
+ const firstLine = lastError.slice(marker + ADMIN_RESTART_MARKER.length).trim().split("\n")[0];
242
+ if (firstLine === undefined || firstLine.trim() === "") return undefined;
243
+ return firstLine.trim();
244
+ }
245
+
225
246
  export function classifyRun(
226
247
  run: RunRecord,
227
248
  facts: ClassifyFacts,
@@ -300,6 +321,22 @@ export function classifyRun(
300
321
  }
301
322
  }
302
323
 
324
+ // A run the draining/restarting process attributed before it restarted the
325
+ // session host: killed under its own ceilings by an administrative action,
326
+ // not by anything the worker did. Costs no budget and requeues, naming the
327
+ // actor (#512). A killed session used to leave no attribution and fell
328
+ // through to `unknown`, charging a failure for an operator's restart.
329
+ if (run.state === "failed" || run.state === "killed") {
330
+ const actor = adminRestartAttribution(run.lastError);
331
+ if (actor !== undefined) {
332
+ return {
333
+ cls: "admin-kill",
334
+ recovery: "requeue",
335
+ evidence: `killed by an administrative restart (${actor}) — an operator's restart, not the run's work`,
336
+ };
337
+ }
338
+ }
339
+
303
340
  if (run.state === "blocked") {
304
341
  return {
305
342
  cls: "question",
package/src/gitops.ts CHANGED
@@ -17,6 +17,7 @@
17
17
  * branch cannot disagree.
18
18
  */
19
19
 
20
+ import { existsSync } from "node:fs";
20
21
  import { join } from "node:path";
21
22
 
22
23
  import { parseChainSource, type ChainEntry } from "./chain-check.ts";
@@ -321,3 +322,159 @@ export async function readBaseChain(
321
322
  return { ok: false, stderr: err instanceof Error ? err.message : String(err) };
322
323
  }
323
324
  }
325
+
326
+ // ------------------------------------------- the shared-host critical-base guard
327
+
328
+ /**
329
+ * The stale-base admission verdict (#428): whether a preserved continuation
330
+ * branch may be reattached at all, once a project marks a base commit as
331
+ * critical. Fail closed — a branch that cannot be *proven* to contain every
332
+ * marker is refused.
333
+ */
334
+ export type CriticalBaseVerdict =
335
+ /** No preserved branch for this issue in the mirror — a clean first attempt. */
336
+ | { state: "no-branch" }
337
+ /** The reattach source contains every configured marker. */
338
+ | { state: "fresh" }
339
+ /** The reattach source predates a marker and cannot be safely advanced to it,
340
+ * so the dispatcher must not reattach it. `range` names the base commits
341
+ * the branch has not absorbed, for the hold's recovery wording. */
342
+ | { state: "stale"; marker: string; range: string[] }
343
+ /** The marker could not be verified at all (unresolvable, fetch failed). */
344
+ | { state: "unknown"; error: string };
345
+
346
+ /** The stubborn-session view of the probe, injectable for deterministic tests. */
347
+ export type CriticalBaseProbe = (
348
+ repo: RepoTarget,
349
+ markers: readonly string[],
350
+ branch: string,
351
+ ) => Promise<CriticalBaseVerdict>;
352
+
353
+ /** The bounded sample of base commits a held branch is missing. */
354
+ const CRITICAL_BASE_RANGE_MAX = 8;
355
+
356
+ /**
357
+ * Answer whether the preserved branch {@link addRunRepo} would reattach
358
+ * contains every configured critical-base marker (#428). A base safety fix
359
+ * protects only branches forked after it landed; a continuation forked before
360
+ * it still carries the dangerous code, and re-running the lifecycle suite it
361
+ * retains on a shared host is what SIGTERMed the production daemon.
362
+ *
363
+ * The reattach source is the mirror's `refs/heads/<branch>`. When it lacks a
364
+ * marker but the live remote branch now carries it and the reattach source is
365
+ * that live head's ancestor, the operator advanced the branch on GitHub (e.g.
366
+ * merged base into the PR branch): the fast-forward is folded into the mirror
367
+ * — safe and work-preserving, the same reconcile `pushRunBranch` performs —
368
+ * and the marker is accepted. Judged against the reattach source, never the
369
+ * live head alone, so a stale local copy cannot smuggle pre-fix code back in.
370
+ */
371
+ export async function probeCriticalBase(
372
+ project: Pick<ProjectConfig, "mirrorRoot">,
373
+ repo: RepoTarget,
374
+ branch: string,
375
+ markers: readonly string[],
376
+ exec: Exec = spawnCaptured,
377
+ ): Promise<CriticalBaseVerdict> {
378
+ const mirror = mirrorPath(project, repo);
379
+ // A mirror is where preserved branches live; without one there is nothing to
380
+ // reattach. Judged "no branch" rather than "unknown" so a fleet that has not
381
+ // yet created a mirror is not deadlocked at admission.
382
+ if (!existsSync(mirror)) return { state: "no-branch" };
383
+ const branchRef = `refs/heads/${branch}`;
384
+ const env = credentialedEnv();
385
+
386
+ const present = await exec(
387
+ ["git", "--git-dir", mirror, "show-ref", "--verify", "--quiet", branchRef],
388
+ { env },
389
+ );
390
+ if (present.code !== 0) return { state: "no-branch" };
391
+
392
+ // The configured marker is a commit that landed on base, so base has to be
393
+ // resolvable to prove anything about the branch against it. Fetch today's
394
+ // base tip into the mirror's tracking refs first, so the judgement is against
395
+ // the current base rather than whatever a run last left behind.
396
+ const baseRef = `refs/remotes/origin/${repo.defaultBranch}`;
397
+ const fetched = await exec(
398
+ ["git", "--git-dir", mirror, "fetch", "--no-tags", "origin", `+refs/heads/${repo.defaultBranch}:${baseRef}`],
399
+ { env },
400
+ );
401
+ if (fetched.code !== 0) {
402
+ return {
403
+ state: "unknown",
404
+ error: scrubUserinfo(fetched.stderr.trim() || fetched.stdout.trim() || `git fetch exited ${String(fetched.code)}`),
405
+ };
406
+ }
407
+
408
+ for (const marker of markers) {
409
+ const resolved = await exec(
410
+ ["git", "--git-dir", mirror, "rev-parse", "--verify", `${marker}^{commit}`],
411
+ { env },
412
+ );
413
+ if (resolved.code !== 0) {
414
+ return {
415
+ state: "unknown",
416
+ error: `critical-base marker "${marker}" is not resolvable in the ${repo.name} mirror`,
417
+ };
418
+ }
419
+ const markerSha = resolved.stdout.trim();
420
+ if (markerSha === "") {
421
+ return { state: "unknown", error: `critical-base marker "${marker}" resolved to no commit` };
422
+ }
423
+
424
+ const localHas = await exec(
425
+ ["git", "--git-dir", mirror, "merge-base", "--is-ancestor", markerSha, branchRef],
426
+ { env },
427
+ );
428
+ if (localHas.code === 0) continue;
429
+
430
+ // The reattach source lacks the marker. If the live branch now carries it
431
+ // and the reattach source is its ancestor, the operator advanced the branch
432
+ // on GitHub (e.g. merged base into the PR branch): fold that fast-forward
433
+ // into the reattach source — safe, work-preserving — and accept the marker.
434
+ const tracked = `refs/remotes/origin/${branch}`;
435
+ const liveFetched = await exec(
436
+ ["git", "--git-dir", mirror, "fetch", "--no-tags", "origin", `+refs/heads/${branch}:${tracked}`],
437
+ { env },
438
+ );
439
+ if (liveFetched.code !== 0) {
440
+ return {
441
+ state: "unknown",
442
+ error: scrubUserinfo(liveFetched.stderr.trim() || liveFetched.stdout.trim() || `git fetch ${branch} exited ${String(liveFetched.code)}`),
443
+ };
444
+ }
445
+ const liveHas = await exec(
446
+ ["git", "--git-dir", mirror, "merge-base", "--is-ancestor", markerSha, tracked],
447
+ { env },
448
+ );
449
+ if (liveHas.code === 0) {
450
+ const localIsAncestor = await exec(
451
+ ["git", "--git-dir", mirror, "merge-base", "--is-ancestor", branchRef, tracked],
452
+ { env },
453
+ );
454
+ if (localIsAncestor.code === 0) {
455
+ await exec(["git", "--git-dir", mirror, "fetch", "--no-tags", "origin", `+${branchRef}:${branchRef}`], { env });
456
+ continue;
457
+ }
458
+ }
459
+
460
+ // Predates the marker and cannot be safely advanced to it. Name the base
461
+ // commits the branch has not absorbed so the hold's recovery is concrete.
462
+ const range: string[] = [];
463
+ const missing = await exec(
464
+ ["git", "--git-dir", mirror, "log", "--oneline", "--format=%h", `${branchRef}..${baseRef}`],
465
+ { env },
466
+ );
467
+ if (missing.code === 0) {
468
+ range.push(
469
+ ...missing.stdout
470
+ .trim()
471
+ .split("\n")
472
+ .filter((line) => line.length > 0)
473
+ .slice(0, CRITICAL_BASE_RANGE_MAX),
474
+ );
475
+ }
476
+ return { state: "stale", marker: markerSha, range };
477
+ }
478
+
479
+ return { state: "fresh" };
480
+ }