@north-light/crouter 0.3.255 → 0.3.256

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. package/dist/api/client.d.ts +13 -1
  2. package/dist/api/client.js +17 -0
  3. package/dist/api/dto/worktree.d.ts +17 -0
  4. package/dist/api/routes.d.ts +1 -0
  5. package/dist/api/routes.js +1 -0
  6. package/dist/clients/attach/viewer.js +936 -797
  7. package/dist/commands/api-client.js +4 -0
  8. package/dist/commands/sys/worktrees.d.ts +1 -0
  9. package/dist/commands/sys/worktrees.js +53 -0
  10. package/dist/commands/sys.js +2 -1
  11. package/dist/core/__tests__/integration/worktree-land.test.js +71 -1
  12. package/dist/core/__tests__/integration/worktree-reap.test.js +101 -0
  13. package/dist/core/__tests__/worktree-landing.test.d.ts +1 -0
  14. package/dist/core/__tests__/worktree-landing.test.js +11 -0
  15. package/dist/core/canvas/canvas.js +24 -8
  16. package/dist/core/canvas/pid.d.ts +6 -0
  17. package/dist/core/canvas/pid.js +48 -3
  18. package/dist/core/canvas/types.d.ts +41 -1
  19. package/dist/core/exclusive-lock.d.ts +6 -0
  20. package/dist/core/exclusive-lock.js +12 -0
  21. package/dist/core/git.d.ts +1 -1
  22. package/dist/core/git.js +5 -1
  23. package/dist/core/runtime/fleet.d.ts +6 -0
  24. package/dist/core/worktree-landing.d.ts +44 -0
  25. package/dist/core/worktree-landing.js +56 -0
  26. package/dist/core/worktree-quarantine.d.ts +19 -0
  27. package/dist/core/worktree-quarantine.js +54 -0
  28. package/dist/core/worktree-sweep.d.ts +60 -0
  29. package/dist/core/worktree-sweep.js +493 -0
  30. package/dist/core/worktree.d.ts +37 -0
  31. package/dist/core/worktree.js +149 -23
  32. package/dist/daemon/__tests__/startup-block-marker.test.d.ts +1 -0
  33. package/dist/daemon/__tests__/startup-block-marker.test.js +86 -0
  34. package/dist/daemon/api/handlers/worktree.js +5 -0
  35. package/dist/daemon/crtrd-cli.js +12 -7
  36. package/dist/daemon/crtrd.js +6 -1
  37. package/dist/daemon/fleet.d.ts +1 -0
  38. package/dist/daemon/fleet.js +3 -0
  39. package/dist/daemon/manage.js +22 -1
  40. package/dist/daemon/reconcilers/managed-worktree-sweep.d.ts +24 -0
  41. package/dist/daemon/reconcilers/managed-worktree-sweep.js +190 -0
  42. package/dist/daemon/reconcilers/storage-maintenance.d.ts +11 -2
  43. package/dist/daemon/reconcilers/storage-maintenance.js +7 -2
  44. package/dist/daemon/startup-block-marker.d.ts +25 -0
  45. package/dist/daemon/startup-block-marker.js +99 -0
  46. package/package.json +1 -1
  47. package/runtime.lock.json +2 -2
@@ -0,0 +1,190 @@
1
+ // The daemon lane that drives managed-worktree reconciliation.
2
+ //
3
+ // It owns WHEN and HOW MANY; `core/worktree-sweep.ts` owns what a single
4
+ // reconciliation does. The lane never blocks the supervision tick: it registers
5
+ // one detached pass and returns immediately, and only one pass is ever in
6
+ // flight, so passes cannot overlap however long a repository takes to answer.
7
+ //
8
+ // NOTHING that spawns a subprocess runs on the tick, and nothing spawns one PER
9
+ // CANDIDATE. `run()` does no work beyond timing; the detached pass reads the
10
+ // canvas, applies the budget, and then takes ONE asynchronous `ps` snapshot
11
+ // that classifies every budgeted candidate's engine. A per-pid probe here would
12
+ // stall the daemon once per stuck worktree.
13
+ //
14
+ // Cost is flat regardless of backlog. The candidate scan touches only terminal
15
+ // nodes, and only those carrying an unreconciled worktree record reach a Git
16
+ // call at all — normally none, so the pass costs zero subprocesses. When there
17
+ // is a backlog, a fixed per-pass budget walks the candidates round-robin, so
18
+ // work per minute is the same whether three worktrees are stuck or three
19
+ // hundred.
20
+ import { getNode, getRow, listNodes } from '../../core/canvas/index.js';
21
+ import { isSafeNodeId } from '../../core/canvas/paths.js';
22
+ import { captureLivenessSnapshotAsync, recordedPidLiveness } from '../../core/canvas/pid.js';
23
+ import { emitEvent } from '../../core/events/emit.js';
24
+ import { operationIdContext } from '../../core/events/operation-id.js';
25
+ import { isDueForSweep, isReconcilable, reconcileManagedWorktree } from '../../core/worktree-sweep.js';
26
+ export const WORKTREE_SWEEP_INTERVAL_MS = 60 * 1000;
27
+ /** Candidates reconciled per pass. Bounds the work a large backlog can demand. */
28
+ export const WORKTREE_SWEEP_BUDGET = 3;
29
+ const TERMINAL_STATUSES = ['done', 'dead', 'canceled'];
30
+ function isTerminal(row) {
31
+ return TERMINAL_STATUSES.includes(row.status);
32
+ }
33
+ /** Is this node's engine definitively gone?
34
+ *
35
+ * Terminal status alone is not enough: `push final` marks a node `done` while
36
+ * its broker is still running out of the managed checkout, which is exactly
37
+ * the case that made cleanup deferred in the first place. So the sweep waits
38
+ * for terminal status AND the absence of any live engine, and an
39
+ * INDETERMINATE pid probe counts as live — uncertainty must never authorize
40
+ * deleting a checkout something may still be writing to.
41
+ *
42
+ * `isClaimed` rather than `has`: a revive that has reserved its slot but not
43
+ * yet registered its broker is a node coming back to life, and the checkout it
44
+ * is about to run in must not be removed underneath it. */
45
+ function engineIsGone(row, fleet, snapshot) {
46
+ if (fleet.isClaimed(row.node_id))
47
+ return false;
48
+ if (row.frozen_at !== null)
49
+ return false;
50
+ if (!isTerminal(row))
51
+ return false;
52
+ return recordedPidLiveness(row.pi_pid, row.pi_pid_identity, snapshot) === 'dead';
53
+ }
54
+ /** The eligibility recheck the reconciliation runs immediately before EVERY Git
55
+ * mutation, synchronously and with no subprocess: a revive that won the race
56
+ * makes this false and the mutation never happens.
57
+ *
58
+ * It re-reads the row rather than trusting the classification the pass began
59
+ * with, and it compares the recorded engine identity against the one that was
60
+ * classified dead — a row whose pid or pid identity moved has been relaunched,
61
+ * whatever its status column currently says. */
62
+ function eligibilityGuard(row, fleet) {
63
+ const nodeId = row.node_id;
64
+ const pid = row.pi_pid;
65
+ const identity = row.pi_pid_identity;
66
+ return {
67
+ stillEligible: () => {
68
+ if (fleet.isClaimed(nodeId))
69
+ return false;
70
+ const current = getRow(nodeId);
71
+ if (current === null || !isTerminal(current) || current.frozen_at !== null)
72
+ return false;
73
+ return current.pi_pid === pid && current.pi_pid_identity === identity;
74
+ },
75
+ };
76
+ }
77
+ export class ManagedWorktreeSweepReconciler {
78
+ lastSweepAt = Number.NEGATIVE_INFINITY;
79
+ inFlight = false;
80
+ /** Round-robin cursor over candidate node ids, so a permanently stuck record
81
+ * at the head of the list cannot starve the ones behind it. */
82
+ cursor = 0;
83
+ run(now, ctx) {
84
+ if (this.inFlight)
85
+ return;
86
+ if (now - this.lastSweepAt < WORKTREE_SWEEP_INTERVAL_MS)
87
+ return;
88
+ if (!ctx.lifecycle.acceptsDetachedWork())
89
+ return;
90
+ this.lastSweepAt = now;
91
+ this.inFlight = true;
92
+ const pass = this.sweep(now, ctx.fleet).finally(() => {
93
+ this.inFlight = false;
94
+ });
95
+ ctx.lifecycle.registerDetached(pass);
96
+ }
97
+ /** Terminal nodes whose managed worktree is unreconciled and whose backoff has
98
+ * elapsed. Canvas reads only — engine liveness is classified later, for the
99
+ * budgeted few, from one shared process snapshot. Returns null when the
100
+ * canvas could not be read, which authorizes nothing. */
101
+ candidates(now) {
102
+ let rows;
103
+ try {
104
+ rows = listNodes({ status: [...TERMINAL_STATUSES] });
105
+ }
106
+ catch (err) {
107
+ operationIdContext.fresh(() => {
108
+ emitEvent({ level: 'error', event: 'worktree.sweep.scan_failed', error: err });
109
+ });
110
+ return null;
111
+ }
112
+ const due = [];
113
+ for (const row of rows) {
114
+ let wt;
115
+ try {
116
+ wt = getNode(row.node_id)?.managed_worktree;
117
+ }
118
+ catch {
119
+ continue; // an unreadable record proves no ownership; leave it alone
120
+ }
121
+ if (!isReconcilable(wt) || wt == null)
122
+ continue;
123
+ if (!isDueForSweep(wt, now))
124
+ continue;
125
+ due.push(row.node_id);
126
+ }
127
+ return due;
128
+ }
129
+ async sweep(now, fleet) {
130
+ const candidates = this.candidates(now);
131
+ if (candidates === null || candidates.length === 0)
132
+ return;
133
+ if (this.cursor >= candidates.length)
134
+ this.cursor = 0;
135
+ const start = this.cursor;
136
+ const take = Math.min(WORKTREE_SWEEP_BUDGET, candidates.length);
137
+ const budgeted = [];
138
+ for (let i = 0; i < take; i += 1)
139
+ budgeted.push(candidates[(start + i) % candidates.length]);
140
+ this.cursor = (start + take) % candidates.length;
141
+ // One process-table read for the whole pass, off the tick. A null snapshot
142
+ // is unknown liveness, and unknown never authorizes destroying a checkout,
143
+ // so the pass simply ends and the next one retries.
144
+ const snapshot = await captureLivenessSnapshotAsync();
145
+ if (snapshot === null)
146
+ return;
147
+ for (const nodeId of budgeted) {
148
+ const row = getRow(nodeId);
149
+ if (row === null || !engineIsGone(row, fleet, snapshot))
150
+ continue;
151
+ await this.reconcileOne(row, now, fleet);
152
+ }
153
+ }
154
+ async reconcileOne(row, now, fleet) {
155
+ const nodeId = row.node_id;
156
+ const target = isSafeNodeId(nodeId) ? { node_id: nodeId } : {};
157
+ const affected = isSafeNodeId(nodeId) ? {} : { affected_node_id: nodeId };
158
+ try {
159
+ const outcome = await reconcileManagedWorktree(nodeId, eligibilityGuard(row, fleet), now);
160
+ operationIdContext.fresh(() => {
161
+ if (outcome.status === 'complete') {
162
+ emitEvent({
163
+ level: 'info',
164
+ event: 'worktree.sweep.reconciled',
165
+ outcome: 'succeeded',
166
+ ...target,
167
+ fields: { ...affected, ...(outcome.stash === undefined ? {} : { base_checkout_stash: outcome.stash }) },
168
+ });
169
+ return;
170
+ }
171
+ if (outcome.status === 'skipped')
172
+ return;
173
+ if (outcome.status === 'refused' && outcome.unchanged === true)
174
+ return; // nothing moved; nothing to say
175
+ emitEvent({
176
+ level: outcome.status === 'failed' ? 'error' : 'warn',
177
+ event: outcome.status === 'failed' ? 'worktree.sweep.failed' : 'worktree.sweep.refused',
178
+ outcome: outcome.status === 'failed' ? 'failed' : 'skipped',
179
+ ...target,
180
+ fields: { ...affected, reason: outcome.reason, ...(outcome.detail === undefined ? {} : { detail: outcome.detail }) },
181
+ });
182
+ });
183
+ }
184
+ catch (err) {
185
+ operationIdContext.fresh(() => {
186
+ emitEvent({ level: 'error', event: 'worktree.sweep.failed', ...target, error: err, fields: affected });
187
+ });
188
+ }
189
+ }
190
+ }
@@ -1,10 +1,19 @@
1
+ import type { FleetRegistry } from '../../core/runtime/fleet.js';
2
+ import type { DetachedWorkLifecycle } from './broker-supervision.js';
1
3
  export declare const DEAD_REAP_GRACE_MS: number;
2
- export type StorageMaintenanceContext = Record<string, never>;
4
+ export interface StorageMaintenanceContext {
5
+ /** Carries the managed-worktree sweep's detached Git work off this tick. */
6
+ lifecycle: DetachedWorkLifecycle;
7
+ /** A live engine handle is the sweep's proof that a node is NOT disposable. */
8
+ fleet: FleetRegistry;
9
+ }
3
10
  /** Recurring storage cleanup. The in-memory throttle clocks intentionally reset
4
- * on daemon boot, so the first tick runs both full sweeps. */
11
+ * on daemon boot, so the first tick runs every full sweep — which is also how
12
+ * managed-worktree reconciliation gets its start-up pass. */
5
13
  export declare class StorageMaintenanceReconciler {
6
14
  private lastSpareSweepAt;
7
15
  private lastGhostSweepAt;
16
+ private readonly worktreeSweep;
8
17
  run(now: number, ctx: StorageMaintenanceContext): void;
9
18
  /** Drop rows whose on-disk node dir is gone. Such a row can never be revived,
10
19
  * focused, or closed — every one of those resolves the node through meta.json —
@@ -6,18 +6,20 @@ import { emitEvent } from '../../core/events/emit.js';
6
6
  import { operationIdContext } from '../../core/events/operation-id.js';
7
7
  import { listLivePanes, tearDownNode } from '../../core/runtime/placement.js';
8
8
  import { reapStaleSpares } from '../../core/runtime/warm-pool.js';
9
+ import { ManagedWorktreeSweepReconciler } from './managed-worktree-sweep.js';
9
10
  // How long a dead node's on-disk record must be quiet before its leftover
10
11
  // placement is reaped. A fresh crash keeps its pane for inspection.
11
12
  export const DEAD_REAP_GRACE_MS = 10 * 60_000;
12
13
  const SPARE_SWEEP_INTERVAL_MS = 60 * 1000;
13
14
  const GHOST_SWEEP_INTERVAL_MS = 60 * 1000;
14
15
  /** Recurring storage cleanup. The in-memory throttle clocks intentionally reset
15
- * on daemon boot, so the first tick runs both full sweeps. */
16
+ * on daemon boot, so the first tick runs every full sweep — which is also how
17
+ * managed-worktree reconciliation gets its start-up pass. */
16
18
  export class StorageMaintenanceReconciler {
17
19
  lastSpareSweepAt = Number.NEGATIVE_INFINITY;
18
20
  lastGhostSweepAt = Number.NEGATIVE_INFINITY;
21
+ worktreeSweep = new ManagedWorktreeSweepReconciler();
19
22
  run(now, ctx) {
20
- void ctx; // no consumer today — kept only for six-way call-site uniformity
21
23
  // Placement reap MUST precede focus GC, so GC catches a focus row the reap
22
24
  // strands. Both run after the lifecycle tick, so a same-tick revive has
23
25
  // already had the chance to re-anchor its focus.
@@ -25,6 +27,9 @@ export class StorageMaintenanceReconciler {
25
27
  this.gcStaleFocuses();
26
28
  this.reapGhostRows(now);
27
29
  this.gcWarmPool(now);
30
+ // Registers detached async Git work and returns immediately: this lane is
31
+ // called un-awaited from a 2-second supervision tick and must not block it.
32
+ this.worktreeSweep.run(now, ctx);
28
33
  }
29
34
  /** Drop rows whose on-disk node dir is gone. Such a row can never be revived,
30
35
  * focused, or closed — every one of those resolves the node through meta.json —
@@ -0,0 +1,25 @@
1
+ import { MigrationBlockedError } from '../migrations/activation.js';
2
+ export declare function daemonStartupBlockMarkerPath(): string;
3
+ /** Record that startup is blocked, so no OTHER process spawns into the same
4
+ * wall. Best-effort: a home we cannot write to still gets in-process
5
+ * suppression, which is strictly better than nothing. */
6
+ export declare function recordDaemonStartupBlocked(message: string, blockerPaths: readonly string[]): void;
7
+ /** Clear the marker — for a repair path that knows startup should be retried
8
+ * (a successful migration, a successful daemon start, an explicit reset). */
9
+ export declare function clearDaemonStartupBlocked(): void;
10
+ /** The recorded block message when startup is still blocked by the SAME corpus
11
+ * that produced it, or null when nothing is recorded or the corpus has since
12
+ * changed. A changed corpus retires the marker as a side effect of the read,
13
+ * so the very next spawner is free to try again.
14
+ *
15
+ * A marker naming no blocker paths at all is not self-healing — nothing about
16
+ * it can be observed to change — so it is treated as advisory only and
17
+ * retired on read rather than left to wedge autostart forever. */
18
+ export declare function daemonStartupBlocked(): string | null;
19
+ /** The blocked-migration failure, wherever it ended up in a thrown value.
20
+ * Startup releases its ownership claim before rethrowing, and a claim release
21
+ * that ALSO fails is reported as an aggregate — so the one error the spawner
22
+ * must react to can arrive nested. Matching only the bare instance would let
23
+ * the aggregate path fall through to a generic exit, and every suppression
24
+ * downstream is keyed on getting that exit code right. */
25
+ export declare function blockedMigrationIn(error: unknown): MigrationBlockedError | null;
@@ -0,0 +1,99 @@
1
+ // A blocked daemon startup is a STANDING condition, and the processes that
2
+ // keep retrying it are not one process. Every CLI invocation, every scheduled
3
+ // job tick, and every long-lived client is its own process with its own module
4
+ // state, so an in-memory suppression flag caps exactly one of them and lets the
5
+ // rest storm unchanged. This marker is that suppression made durable: one file
6
+ // in the canvas home, written by whoever first proved startup is blocked, and
7
+ // read by every would-be spawner before it spawns.
8
+ //
9
+ // The hazard of a durable "stop trying" record is that it outlives the problem
10
+ // and wedges a repaired install. So the marker is self-healing rather than
11
+ // permanent: it records the identity (path + size + mtime) of each blocker the
12
+ // daemon named, and any change to any of them retires the marker on the next
13
+ // read. Repairing the corpus — by `crtr sys migrate`, an editor, or deleting
14
+ // the offending file — is itself the thing that lifts the suppression, so no
15
+ // user is ever required to know this file exists.
16
+ import { rmSync } from 'node:fs';
17
+ import { statSync } from 'node:fs';
18
+ import { join } from 'node:path';
19
+ import { crtrHome } from '../core/canvas/paths.js';
20
+ import { MigrationBlockedError } from '../migrations/activation.js';
21
+ import { atomicWriteJson, readJsonOrNull } from '../core/fs-utils.js';
22
+ export function daemonStartupBlockMarkerPath() {
23
+ return join(crtrHome(), 'daemon-startup-blocked.json');
24
+ }
25
+ function fingerprint(path) {
26
+ try {
27
+ const stat = statSync(path);
28
+ return { path, sizeBytes: stat.size, mtimeMs: stat.mtimeMs };
29
+ }
30
+ catch {
31
+ return { path, sizeBytes: null, mtimeMs: null };
32
+ }
33
+ }
34
+ function sameFingerprint(a, b) {
35
+ return a.sizeBytes === b.sizeBytes && a.mtimeMs === b.mtimeMs;
36
+ }
37
+ /** Record that startup is blocked, so no OTHER process spawns into the same
38
+ * wall. Best-effort: a home we cannot write to still gets in-process
39
+ * suppression, which is strictly better than nothing. */
40
+ export function recordDaemonStartupBlocked(message, blockerPaths) {
41
+ const marker = {
42
+ recordedAt: Date.now(),
43
+ message,
44
+ blockers: blockerPaths.map(fingerprint),
45
+ };
46
+ try {
47
+ atomicWriteJson(daemonStartupBlockMarkerPath(), marker);
48
+ }
49
+ catch { /* best effort */ }
50
+ }
51
+ /** Clear the marker — for a repair path that knows startup should be retried
52
+ * (a successful migration, a successful daemon start, an explicit reset). */
53
+ export function clearDaemonStartupBlocked() {
54
+ try {
55
+ rmSync(daemonStartupBlockMarkerPath(), { force: true });
56
+ }
57
+ catch { /* best effort */ }
58
+ }
59
+ /** The recorded block message when startup is still blocked by the SAME corpus
60
+ * that produced it, or null when nothing is recorded or the corpus has since
61
+ * changed. A changed corpus retires the marker as a side effect of the read,
62
+ * so the very next spawner is free to try again.
63
+ *
64
+ * A marker naming no blocker paths at all is not self-healing — nothing about
65
+ * it can be observed to change — so it is treated as advisory only and
66
+ * retired on read rather than left to wedge autostart forever. */
67
+ export function daemonStartupBlocked() {
68
+ const marker = readJsonOrNull(daemonStartupBlockMarkerPath());
69
+ if (marker === null || typeof marker.message !== 'string' || !Array.isArray(marker.blockers))
70
+ return null;
71
+ if (marker.blockers.length === 0) {
72
+ clearDaemonStartupBlocked();
73
+ return null;
74
+ }
75
+ const changed = marker.blockers.some((recorded) => !sameFingerprint(recorded, fingerprint(recorded.path)));
76
+ if (changed) {
77
+ clearDaemonStartupBlocked();
78
+ return null;
79
+ }
80
+ return marker.message;
81
+ }
82
+ /** The blocked-migration failure, wherever it ended up in a thrown value.
83
+ * Startup releases its ownership claim before rethrowing, and a claim release
84
+ * that ALSO fails is reported as an aggregate — so the one error the spawner
85
+ * must react to can arrive nested. Matching only the bare instance would let
86
+ * the aggregate path fall through to a generic exit, and every suppression
87
+ * downstream is keyed on getting that exit code right. */
88
+ export function blockedMigrationIn(error) {
89
+ if (error instanceof MigrationBlockedError)
90
+ return error;
91
+ if (error instanceof AggregateError) {
92
+ for (const nested of error.errors) {
93
+ const found = blockedMigrationIn(nested);
94
+ if (found !== null)
95
+ return found;
96
+ }
97
+ }
98
+ return null;
99
+ }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@north-light/crouter",
3
- "version": "0.3.255",
3
+ "version": "0.3.256",
4
4
  "description": "crtr — agent runtime with memory, plugins, and marketplaces",
5
5
  "type": "module",
6
6
  "main": "dist/index.js",
package/runtime.lock.json CHANGED
@@ -1,12 +1,12 @@
1
1
  {
2
2
  "name": "@north-light/crouter",
3
- "version": "0.3.255",
3
+ "version": "0.3.256",
4
4
  "lockfileVersion": 3,
5
5
  "requires": true,
6
6
  "packages": {
7
7
  "": {
8
8
  "name": "@north-light/crouter",
9
- "version": "0.3.255",
9
+ "version": "0.3.256",
10
10
  "hasInstallScript": true,
11
11
  "license": "MIT",
12
12
  "dependencies": {