canopycms 0.0.65 → 0.0.66-int.82

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (60) hide show
  1. package/README.md +1 -3
  2. package/dist/ai/generate.d.ts +13 -0
  3. package/dist/ai/generate.js +9 -3
  4. package/dist/ai/to-plain-text.js +4 -0
  5. package/dist/ai/types.d.ts +20 -1
  6. package/dist/api/admin-branch-health.js +13 -12
  7. package/dist/api/branch.js +2 -2
  8. package/dist/branch-health.js +1 -1
  9. package/dist/branch-metadata-file.d.ts +58 -0
  10. package/dist/branch-metadata-file.js +65 -0
  11. package/dist/branch-metadata.d.ts +2 -26
  12. package/dist/branch-metadata.js +6 -33
  13. package/dist/branch-registry.js +4 -2
  14. package/dist/build/generate-ai-content.js +103 -0
  15. package/dist/cli/cli.js +432 -317
  16. package/dist/cli/generate-ai-content.js +264 -160
  17. package/dist/content-id-index.js +22 -0
  18. package/dist/content-listing.js +3 -0
  19. package/dist/content-reader.d.ts +25 -1
  20. package/dist/content-reader.js +15 -0
  21. package/dist/content-store.d.ts +50 -1
  22. package/dist/content-store.js +102 -4
  23. package/dist/context.d.ts +36 -4
  24. package/dist/context.js +15 -4
  25. package/dist/editor/admin/SystemHealthPanel.js +2 -2
  26. package/dist/entry-schema.d.ts +22 -0
  27. package/dist/git-manager.d.ts +9 -8
  28. package/dist/git-manager.js +35 -9
  29. package/dist/github-service.d.ts +1 -1
  30. package/dist/github-service.js +1 -1
  31. package/dist/paths/branch-name.d.ts +1 -1
  32. package/dist/paths/branch-name.js +1 -1
  33. package/dist/schema/index.d.ts +1 -1
  34. package/dist/schema/index.js +1 -1
  35. package/dist/schema/meta-loader.js +1 -1
  36. package/dist/services.d.ts +22 -0
  37. package/dist/types.d.ts +1 -1
  38. package/dist/url-exclusivity-fixtures.d.ts +76 -0
  39. package/dist/url-exclusivity-fixtures.js +119 -0
  40. package/dist/url-path-resolver.js +8 -4
  41. package/dist/utils/content-write-lock.d.ts +2 -1
  42. package/dist/utils/content-write-lock.js +2 -1
  43. package/dist/utils/error.d.ts +17 -0
  44. package/dist/utils/error.js +17 -0
  45. package/dist/utils/git.d.ts +1 -1
  46. package/dist/utils/git.js +1 -1
  47. package/dist/utils/occ-json-write.js +1 -1
  48. package/dist/worker/cms-worker.d.ts +22 -288
  49. package/dist/worker/cms-worker.js +118 -1995
  50. package/dist/worker/git-sync.d.ts +166 -0
  51. package/dist/worker/git-sync.js +554 -0
  52. package/dist/worker/history-rewrite.d.ts +129 -0
  53. package/dist/worker/history-rewrite.js +216 -0
  54. package/dist/worker/rebase.d.ts +171 -0
  55. package/dist/worker/rebase.js +859 -0
  56. package/dist/worker/task-runner.d.ts +93 -0
  57. package/dist/worker/task-runner.js +529 -0
  58. package/dist/worker/worker-context.d.ts +133 -0
  59. package/dist/worker/worker-context.js +1 -0
  60. package/package.json +1 -1
@@ -1,3 +1,4 @@
1
+ export { PermanentTaskError, isPermanentTaskFailure } from './task-runner.js';
1
2
  export { workerLog, workerLogWarn, workerLogError, installWorkerLogger } from './log.js';
2
3
  /**
3
4
  * Auth cache refresh function type.
@@ -63,37 +64,6 @@ export interface CmsWorkerConfig {
63
64
  */
64
65
  lockStaleMs?: number;
65
66
  }
66
- /**
67
- * An error inherent to the task itself (malformed payload, unknown action):
68
- * retrying can never succeed, so the task should fail fast instead of
69
- * burning its retry budget.
70
- */
71
- export declare class PermanentTaskError extends Error {
72
- }
73
- /**
74
- * Classify a task failure as permanent (fail fast) or transient (retry).
75
- *
76
- * Transient — worth retrying with backoff:
77
- * - network errors / anything without an HTTP status (git failures included:
78
- * most push/fetch failures are connectivity or contention and the retry
79
- * budget bounds the pathological cases)
80
- * - HTTP 408 (request timeout) and 429 (rate limited)
81
- * - HTTP 403 that carries a rate-limit signal (see `isRateLimitSignal403`):
82
- * GitHub returns 403, not 429, for both primary and secondary/abuse rate
83
- * limits. The throttling plugin (see github-service.ts createCanopyOctokit)
84
- * proactively retries short waits, but this carve-out remains the safety
85
- * net for waits the plugin gives up on (`shouldRetryRateLimit`/
86
- * `shouldRetrySecondaryRateLimit`) and for errors it never sees — without
87
- * it a rate-limited push-and-create-or-update-pr task would fail
88
- * permanently and wedge the branch (`sync-failed`, no retry).
89
- * - HTTP 5xx (server-side, usually recovers)
90
- *
91
- * Permanent — retrying the identical request cannot succeed:
92
- * - PermanentTaskError (malformed payload, unknown action)
93
- * - other HTTP 4xx (e.g. 401/404/422): the request itself is bad
94
- * - plain HTTP 403 with no rate-limit signal: a real permission denial
95
- */
96
- export declare function isPermanentTaskFailure(err: unknown): boolean;
97
67
  /**
98
68
  * CMS Worker daemon.
99
69
  * Handles operations that Lambda (with no internet) cannot perform:
@@ -165,6 +135,21 @@ export declare class CmsWorker {
165
135
  * resolver is pure, so a later call returns the identical string.
166
136
  */
167
137
  private ensureSettingsBranch;
138
+ /**
139
+ * Build the {@link WorkerContext} handed to the extracted clusters
140
+ * (task-runner.ts, git-sync.ts, rebase.ts, history-rewrite.ts).
141
+ *
142
+ * Built FRESH on every call rather than once in the constructor, and the
143
+ * instance-backed members are functions rather than copied values. Both
144
+ * choices exist for the same reason: the test suite drives this class by
145
+ * replacing `octokit` and `buildGitHubUrl` ON THE INSTANCE, by setting
146
+ * `running` directly, and by subclassing to override the two rebase test
147
+ * hooks. A context that captured any of those at construction time would hand
148
+ * the extracted code the pre-test value -- which for `buildGitHubUrl` means a
149
+ * test's push going to github.com for real instead of its local fixture repo.
150
+ * See WorkerContext's doc comment for the full list.
151
+ */
152
+ private ctx;
168
153
  start(): Promise<void>;
169
154
  stop(): Promise<void>;
170
155
  /**
@@ -255,47 +240,10 @@ export declare class CmsWorker {
255
240
  * guard existed.
256
241
  */
257
242
  private ensureRemoteGit;
258
- /**
259
- * Staleness threshold for recoverOrphanedTasks, derived from the
260
- * configured task timeout rather than fixed: the safety argument for
261
- * running recovery on every poll cycle is "no legitimately in-flight task
262
- * can be this old, because executeTaskWithTimeout bounds every attempt by
263
- * taskTimeoutMs" -- which is only true if this threshold scales with
264
- * taskTimeoutMs. 2x leaves the same comfortable margin the defaults have
265
- * (60s timeout vs 5min threshold); the 5-minute floor preserves the
266
- * long-standing default for a replacement instance's boot window.
267
- */
268
- private orphanRecoveryMaxAgeMs;
269
- /**
270
- * Process queued tasks from Lambda.
271
- * Polls .tasks/pending/ directory and executes each task.
272
- * Processes up to maxTasksPerCycle tasks per invocation.
273
- * Retries transient failures with exponential backoff.
274
- */
275
243
  processTaskQueue(): Promise<void>;
276
- /**
277
- * Execute a task bounded by taskTimeoutMs (DEP-H1). Two layers:
278
- * - An AbortSignal cancels Octokit HTTP calls promptly.
279
- * - A Promise.race rejects when the timeout fires, so work that cannot
280
- * observe the signal (git subprocesses via simple-git) still fails the
281
- * attempt and the worker moves on instead of stalling forever.
282
- * pushBranchToGitHub additionally kills stalled git processes via
283
- * simple-git's block timeout, so a hung push doesn't leak a process.
284
- */
285
- private executeTaskWithTimeout;
286
244
  private executeTask;
287
- /**
288
- * Update branch metadata after successful task completion.
289
- * Writes PR URL/number and sets syncStatus to 'synced'.
290
- */
291
245
  private updateBranchMetadata;
292
- /**
293
- * Update branch metadata after permanent task failure.
294
- * Sets syncStatus to 'sync-failed' and records `error` (already redacted
295
- * by the caller, processTaskQueue -- see [HIGH-1] there) as
296
- * syncFailureReason, so the editor can show WHY, not just that it failed.
297
- */
298
- private updateBranchMetadataOnFailure;
246
+ private pushBranchToGitHub;
299
247
  private buildGitHubUrl;
300
248
  /**
301
249
  * The workspace directory for a branch named by its GIT REF name -- the
@@ -314,225 +262,6 @@ export declare class CmsWorker {
314
262
  * is sanitized.
315
263
  */
316
264
  private branchWorkspacePath;
317
- /**
318
- * What `remote.git` currently holds for `branchRef`, or null when this
319
- * branch was never published there (never submitted) or the ref is
320
- * unreadable.
321
- *
322
- * Uses the same explicit `--git-dir` shape as verifyBaseBranchExists()
323
- * above: reading a bare repo that way does not depend on
324
- * `safe.bareRepository` being permissive, which is why prod code takes
325
- * this route rather than the config override the test-only `openBareRepo`
326
- * helper uses.
327
- */
328
- private readPublishedSha;
329
- /**
330
- * Record that this worker rewrote `expectedSha` out of a branch's already
331
- * published history (see BranchMetadata.historyRewrittenFrom).
332
- *
333
- * Set-once: if a marker is already present it is LEFT ALONE. Across two
334
- * rebases before any GitHub push lands, GitHub still holds the commit the
335
- * FIRST rebase replaced, so advancing the marker would aim the lease at a
336
- * commit GitHub never had and permanently wedge the branch.
337
- *
338
- * Re-reads metadata rather than trusting the caller's loop-top snapshot:
339
- * the task loop runs concurrently with syncGit() (see scheduleLoop) and may
340
- * have cleared the marker while this branch was rebasing.
341
- */
342
- private markHistoryRewritten;
343
- /**
344
- * Clear the marker once GitHub is confirmed to hold the rewritten history.
345
- *
346
- * Best-effort only in the sense that a failure here destroys nothing: the
347
- * lease still refuses anything unexpected, and the plain-push fallback only
348
- * ever fast-forwards. It is NOT harmless. A marker that outlives its
349
- * episode can wedge the NEXT one -- if the base advances before the stale
350
- * marker is revisited, the queued push leases a commit GitHub has already
351
- * moved off, falls back to a plain push of a rebased (non-ancestor)
352
- * history, and fails permanently with a "something else moved it on
353
- * GitHub" diagnosis that is false: we did.
354
- *
355
- * Tracked, with the concurrent-clear race that can drop a marker
356
- * mid-arming, in
357
- * .claude/future-tasks/worker-history-rewrite-marker-races.md.
358
- */
359
- private clearHistoryRewrittenMarker;
360
- /**
361
- * Publish a branch clone's rebased history into `remote.git`, replacing
362
- * EXACTLY `expectedSha` and nothing else.
363
- *
364
- * The lease is the entire safety argument. `--force-with-lease=<ref>:<sha>`
365
- * refuses unless `remote.git` still stands at `<sha>`, so this can only
366
- * ever undo the commit our own rebase rewrote away. Callers must never pass
367
- * "whatever remote.git currently holds" -- see the arming guard in
368
- * rebaseActiveBranches() for the interleaving where that would silently
369
- * delete a reviewer's direct push.
370
- *
371
- * Returns whether the push landed. A refused lease means a concurrent
372
- * Lambda push moved the ref; that is logged and retried by the self-heal
373
- * pass on a later cycle, never thrown.
374
- */
375
- private forcePublishToLocalRemote;
376
- /**
377
- * Queue the GitHub hop for a branch whose rewritten history now sits in
378
- * `remote.git`, so an open PR's head follows the rebase within a cycle
379
- * instead of waiting for the editor's next submit.
380
- *
381
- * Deliberately NOT skipped when a marker is already set: inferring "a task
382
- * must already be queued" from the marker starves this hop whenever a task
383
- * was lost, failed permanently, or was never written. Duplicate push tasks
384
- * are bounded by base-branch advances and are idempotent (a repeat push is
385
- * a no-op once GitHub holds the tip).
386
- */
387
- private enqueueGitHubPush;
388
- /**
389
- * Complete a rewrite this worker started but did not finish: get the
390
- * rebased history into `remote.git` and queued for GitHub.
391
- *
392
- * Runs from the rebase loop for any branch carrying a marker, whether or
393
- * not it is behind base this cycle, so an interrupted publish converges
394
- * without waiting for the next base-branch advance.
395
- *
396
- * Always leases on the MARKER, never on remote.git's current tip -- the
397
- * marker is the one commit we know our own rebase replaced.
398
- */
399
- private reconcilePendingRewrite;
400
- private pushBranchToGitHub;
401
- /**
402
- * Push THIS deployment's own settings branch (`ensureSettingsBranch()`) from
403
- * remote.git to GitHub. Non-fatal: a no-op push for an up-to-date branch
404
- * just succeeds quietly.
405
- *
406
- * Deliberately narrowed to one branch — this used to push EVERY local
407
- * branch matching `canopycms-settings-*`. With the tracking-namespace fetch
408
- * fix (GITHUB_TRACKING_REF_PREFIX), `reconcileTrackedBranches` creates local
409
- * heads for branches that exist on GitHub, so ANOTHER deployment's settings
410
- * branch (sharing this same GitHub repo) can legitimately show up as a
411
- * local head here too. Pushing it would be this deployment shipping
412
- * settings state it doesn't own.
413
- */
414
- private pushSettingsBranches;
415
- /**
416
- * Bring `refs/heads/*` in `remote.git` toward what was just fetched into
417
- * `GITHUB_TRACKING_REF_PREFIX` -- WITHOUT ever force-rewinding or deleting
418
- * a local head. This is the non-destructive replacement for what the old
419
- * `+refs/heads/*:refs/heads/*` fetch refspec used to do implicitly (and
420
- * destructively) as part of the fetch itself; see
421
- * `GITHUB_TRACKING_REF_PREFIX`'s doc comment for the two failure modes
422
- * that refspec caused.
423
- *
424
- * Per tracked branch:
425
- * - no local `refs/heads/<name>` yet -> create it at the tracked commit
426
- * (a branch created on GitHub, or by another deployment sharing this
427
- * GitHub repo, becomes visible locally).
428
- * - local is a strict ancestor of tracked (behind) -> fast-forward it.
429
- * - local === tracked -> nothing to do.
430
- * - tracked is a strict ancestor of local (ahead) -> LEAVE IT ALONE. This
431
- * is unpushed editor/settings work; the queued push task (or
432
- * pushSettingsBranches) ships it. This is exactly the branch state the
433
- * old refspec used to destroy.
434
- * - neither is an ancestor of the other (diverged) -> LEAVE IT ALONE and
435
- * count/log it. A real collision (e.g. another deployment moved the
436
- * same branch name on GitHub); the next push attempt will be rejected
437
- * non-fast-forward, which is the correct, visible outcome -- this
438
- * method must never silently pick a winner. The one expected, benign
439
- * form of this -- our own rebase loop having published a rewrite into
440
- * remote.git with the GitHub push still queued -- is split out into the
441
- * `rewritten` bucket so the collision warning stays meaningful.
442
- *
443
- * Never deletes a local head: a branch removed on GitHub simply stops
444
- * being tracked here; the local ref persists until removed through its
445
- * own explicit path (the sync loop must not be one of them).
446
- *
447
- * `remote.git` is bare, so there is no worktree to invalidate by moving
448
- * these refs -- unlike a non-bare repo, updating the ref that happens to
449
- * be "checked out" is a non-issue here.
450
- *
451
- * Concurrency: `remote.git` is bare and on EFS, and the Lambda pushes into
452
- * it concurrently (`GitManager.push()`'s `target:target` refspec) while
453
- * this runs. Every `update-ref` below passes the expected old value (the
454
- * all-zeros OID for "must not exist yet" on creation, the previously-read
455
- * SHA for the fast-forward case) so a concurrent Lambda write landing in
456
- * the gap between the read and the write loses the ref update instead of
457
- * being silently clobbered -- the branch is simply revisited next cycle.
458
- */
459
- private reconcileTrackedBranches;
460
- /**
461
- * Whether this branch carries a pending history rewrite -- the rebase loop
462
- * rewrote already-published history and the GitHub push has not landed yet
463
- * (see BranchMetadata.historyRewrittenFrom).
464
- *
465
- * `branchName` is a git ref name; branch workspaces are directories named
466
- * with the sanitized form, hence the conversion. Best-effort: a settings
467
- * branch (no workspace at all), a missing directory or an unreadable
468
- * branch.json all mean "no known rewrite", which is the conservative
469
- * answer -- it keeps the branch in the louder `diverged` bucket.
470
- */
471
- private hasPendingHistoryRewrite;
472
- syncGit(): Promise<void>;
473
- /**
474
- * [C1] Remove `.trash-*` branch directories (created by the admin purge
475
- * action, api/admin-branch-health.ts) whose name-embedded stamp is older
476
- * than {@link TRASH_RETENTION_MS}. Names that don't match the expected
477
- * `.trash-{dirName}-{STAMP}` shape, or whose stamp fails to parse, are
478
- * left alone (logged once per cycle, not per file, to avoid flooding logs
479
- * if something odd accumulates) -- purge is the only writer of this
480
- * naming scheme, so an unparseable name is unexpected and worth a human
481
- * looking rather than a silent skip.
482
- */
483
- private cleanupTrashedBranchDirs;
484
- /**
485
- * Fast-forward the base branch's own working-tree clone
486
- * (content-branches/<baseBranch>) to match origin/<baseBranch>.
487
- *
488
- * Previously this clone was refreshed only incidentally, by the generic
489
- * rebase loop below (rebaseActiveBranches): for a branch with status
490
- * 'editing', rebasing onto origin/<baseBranch> degenerates to a
491
- * fast-forward when the clone IS the base branch. But that loop's skip
492
- * paths -- a dirty tree, a missing .git -- are silent, which is the
493
- * suspected live failure mode: a wedged base clone with no diagnosable
494
- * signal in the logs. This dedicated step makes the refresh explicit,
495
- * ff-only, and loud, so a stuck base view (an editor forking a new branch
496
- * "from base" that's actually a stale snapshot) is diagnosable from logs.
497
- * This runs every sync cycle so the drift window is bounded by
498
- * gitSyncInterval.
499
- *
500
- * ff-only on purpose: this clone must stay a linear mirror of
501
- * origin/<baseBranch>, so a merge that isn't a fast-forward (diverged
502
- * local history) is treated as a should-never-happen condition and left
503
- * untouched rather than force-resolved.
504
- */
505
- private refreshBaseBranchWorkspace;
506
- /**
507
- * Poll GitHub for a submitted/approved branch's PR resolution.
508
- *
509
- * submitted/approved branches sit outside the rebase loop and get no
510
- * other signal that their PR resolved on GitHub -- nothing pushes a
511
- * merge/close webhook back into the branch workspace. merged ->
512
- * auto-archive via buildMergedBranchUpdate (shared with the manual
513
- * markAsMerged API so both paths produce identical archived-branch
514
- * metadata). closed-without-merge -> record pullRequestState only; an
515
- * admin decides the workflow transition from there. Best-effort: any
516
- * failure here is logged and swallowed, retried next sync cycle.
517
- */
518
- private pollMergeState;
519
- /**
520
- * Persist a per-branch rebase failure to branch.json (PR-W2), bounded to
521
- * roughly one save per failing branch per hour: a branch stuck failing
522
- * every cycle must not turn into unbounded save-per-cycle x N-failing-
523
- * branches write amplification -- save() eager-regenerates the branch
524
- * registry (branch-metadata.ts's invalidateRegistry(), O(branch count) EFS
525
- * reads), the same concern the `alreadyClean` no-op guard above exists
526
- * for.
527
- *
528
- * Best-effort and non-fatal like every other metadata write in this
529
- * loop's error paths: a corrupt branch.json, a lock-contention error, or
530
- * any other save failure here must never abort the per-branch iteration.
531
- * This matters doubly at the two call sites -- one is inside the outer
532
- * per-branch catch, with no further catch of its own around this call --
533
- * so the whole method is wrapped, not just the load.
534
- */
535
- private recordRebaseFailure;
536
265
  /**
537
266
  * Test-only seam: awaited inside `rebaseActiveBranches()`'s conflict round,
538
267
  * after `git rebase` reported conflicted files and BEFORE the
@@ -555,5 +284,10 @@ export declare class CmsWorker {
555
284
  */
556
285
  protected afterRebaseCompletedForTesting(): Promise<void>;
557
286
  private rebaseActiveBranches;
287
+ syncGit(): Promise<void>;
288
+ private pushSettingsBranches;
289
+ private refreshBaseBranchWorkspace;
290
+ private cleanupTrashedBranchDirs;
291
+ private pollMergeState;
558
292
  refreshAuthCache(): Promise<void>;
559
293
  }