mercury-agent 0.18.2 → 0.20.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (107) hide show
  1. package/container/Dockerfile +25 -0
  2. package/container/build.sh +23 -3
  3. package/docs/behavior-layers.md +25 -14
  4. package/docs/configuration.md +139 -11
  5. package/docs/container-lifecycle.md +151 -1
  6. package/docs/context-architecture.md +6 -2
  7. package/docs/extensions.md +9 -1
  8. package/docs/goals/football-reporter-profile/decisions.md +79 -0
  9. package/docs/goals/rehearsal-bench/decisions.md +41 -3
  10. package/docs/goals/rehearsal-bench/roadmap.md +3 -1
  11. package/docs/goals/release-gate/decisions.md +86 -0
  12. package/docs/goals/release-gate/roadmap.md +36 -3
  13. package/docs/live-testing.md +22 -0
  14. package/docs/pending-verification.md +178 -0
  15. package/docs/permissions.md +1 -1
  16. package/docs/profile-guide.md +28 -8
  17. package/examples/extensions/archive/backends/local.ts +6 -3
  18. package/examples/extensions/archive/queue.ts +51 -10
  19. package/examples/extensions/feed-watch/config.ts +42 -2
  20. package/examples/extensions/feed-watch/digest.ts +4 -7
  21. package/examples/extensions/feed-watch/items.ts +18 -7
  22. package/examples/extensions/feed-watch/match.ts +125 -9
  23. package/examples/extensions/feed-watch/skill/SKILL.md +25 -0
  24. package/examples/extensions/feed-watch/watch.ts +39 -5
  25. package/examples/extensions/gws/index.ts +126 -8
  26. package/examples/extensions/longview/hook.ts +72 -7
  27. package/examples/extensions/longview/index.ts +2 -0
  28. package/examples/extensions/morning/README.md +26 -15
  29. package/examples/extensions/morning/index.ts +29 -6
  30. package/examples/extensions/morning/lib/hosts.ts +16 -0
  31. package/examples/extensions/morning/lib/morning.ts +78 -0
  32. package/examples/extensions/morning/lib/upload.ts +584 -0
  33. package/examples/extensions/morning/skill/SKILL.md +34 -4
  34. package/examples/extensions/napkin/index.ts +12 -3
  35. package/examples/extensions/napkin/pi-spawn.ts +5 -1
  36. package/examples/extensions/overview/index.ts +20 -0
  37. package/examples/extensions/overview/skill/SKILL.md +9 -1
  38. package/examples/extensions/pinchtab/index.ts +36 -7
  39. package/examples/extensions/pinchtab/skill/SKILL.md +28 -1
  40. package/examples/profiles/_template/AGENTS.md +6 -1
  41. package/examples/profiles/football-reporter/AGENTS.md +16 -4
  42. package/examples/profiles/football-reporter/README.md +1 -1
  43. package/examples/profiles/football-reporter/config.yaml +42 -6
  44. package/examples/profiles/football-reporter/seed/MEMORY.md +1 -1
  45. package/examples/profiles/football-reporter/seed/episodes/beitar-jerusalem-2026-27.md +1 -1
  46. package/examples/profiles/football-reporter/seed/episodes/maccabi-tel-aviv-2026-27.md +24 -0
  47. package/examples/profiles/football-reporter/seed/napkin-distill.md +30 -29
  48. package/examples/profiles/football-reporter/standard.json +45 -8
  49. package/package.json +11 -6
  50. package/resources/skills/tasks/SKILL.md +17 -1
  51. package/resources/templates/AGENTS.md +4 -3
  52. package/resources/templates/mercury.example.yaml +13 -4
  53. package/src/adapters/whatsapp-ingress.ts +15 -3
  54. package/src/agent/container-entry.ts +299 -29
  55. package/src/agent/container-env.ts +26 -8
  56. package/src/agent/container-runner.ts +497 -76
  57. package/src/agent/image-contract.ts +77 -0
  58. package/src/agent/image-manifest.ts +180 -0
  59. package/src/agent/image-refresh.ts +318 -0
  60. package/src/cli/mercury.ts +225 -28
  61. package/src/cli/mrctl-http.ts +5 -0
  62. package/src/cli/mrctl.ts +57 -5
  63. package/src/cli/service-unit.ts +109 -0
  64. package/src/config-file.ts +22 -1
  65. package/src/config.ts +105 -14
  66. package/src/core/api.ts +11 -2
  67. package/src/core/commands.ts +11 -4
  68. package/src/core/connection-health.ts +377 -0
  69. package/src/core/direct-send.ts +235 -14
  70. package/src/core/exec.ts +10 -0
  71. package/src/core/history-window.ts +142 -0
  72. package/src/core/model-command.ts +130 -0
  73. package/src/core/operator-alerts.ts +326 -23
  74. package/src/core/permissions.ts +28 -0
  75. package/src/core/profiles.ts +32 -16
  76. package/src/core/reply-context.ts +31 -0
  77. package/src/core/routes/config-builtin.ts +11 -0
  78. package/src/core/routes/console.ts +98 -16
  79. package/src/core/routes/dashboard.ts +93 -10
  80. package/src/core/routes/model.ts +26 -3
  81. package/src/core/routes/send.ts +1 -1
  82. package/src/core/routes/tasks.ts +74 -0
  83. package/src/core/runtime.ts +302 -21
  84. package/src/core/system-messages.ts +33 -0
  85. package/src/core/task-scheduler.ts +155 -7
  86. package/src/extensions/image-builder.ts +1 -1
  87. package/src/extensions/installer.ts +72 -17
  88. package/src/extensions/load-project.ts +74 -0
  89. package/src/extensions/loader.ts +50 -9
  90. package/src/host-version.ts +32 -0
  91. package/src/main.ts +42 -21
  92. package/src/preflight/checks/credential.ts +300 -0
  93. package/src/preflight/checks/docker.ts +149 -0
  94. package/src/preflight/checks/extensions.ts +90 -0
  95. package/src/preflight/checks/host-deps.ts +247 -0
  96. package/src/preflight/checks/image-contract.ts +228 -0
  97. package/src/preflight/checks/roundtrip.ts +424 -0
  98. package/src/preflight/checks/sandbox.ts +159 -0
  99. package/src/preflight/deps.ts +107 -0
  100. package/src/preflight/probe-container.ts +156 -0
  101. package/src/preflight/report.ts +177 -0
  102. package/src/preflight/run.ts +223 -0
  103. package/src/server.ts +55 -14
  104. package/src/storage/db.ts +41 -2
  105. package/src/storage/models-json.ts +110 -0
  106. package/src/text/reporter-lint.ts +167 -23
  107. package/src/types.ts +22 -0
@@ -616,3 +616,82 @@
616
616
  - **Used by:** football-reporter-profile roadmap (M2.2 row),
617
617
  `docs/bugs/model-switch-disables-fallback.md` (the reset-verb constraint),
618
618
  `docs/ideas/scheduled-task-model-leg-policy.md`
619
+
620
+ ## D-023: The club markers are the owner's 2026-09-03 set, a marker may be two emoji, and Maccabi Tel Aviv is standing coverage (decided 2026-09-03)
621
+ - **Category:** editorial standard / process (applies [D-020](#d-020-live-tuning-is-adopted-into-the-repo-not-reverted--and-scans-carry-a-fixed-club-emoji-decided-2026-08-31) a second time)
622
+ - **Decided:** (1) The scan markers are now מנצ'סטר יונייטד 👹, ברצלונה ❤️💙,
623
+ ריאל מדריד 🤍, הפועל תל אביב ❤️, בית"ר ירושלים 💛🖤, מכבי חיפה 💚 and
624
+ מכבי תל אביב 💛💙 — seven clubs, chosen by the owner in the football chat on
625
+ 2026-09-03 between 12:35 and 12:55 IDT (messages #4210–#4223) and written by
626
+ the bot into the live `AGENTS.md` and `MEMORY.md` at 12:55. The repo adopts
627
+ the live text verbatim, as D-020 says; the diff was exactly the two owner
628
+ changes and the file contradicts itself nowhere. (2) A marker is the **whole
629
+ emoji run that opens the line**, compared as one unit and without variation
630
+ selectors. That is what keeps a bare ❤️ (Hapoel) from passing as half of
631
+ ❤️💙 (Barcelona), and what keeps a phone that drops or adds U+FE0F from
632
+ turning a correct post into a finding. (3) Maccabi Tel Aviv joins the
633
+ standing coverage (`## תפקיד`) with a seeded notebook, the same shape as the
634
+ five seeded on 2026-08-21; the live space already had one, opened by napkin
635
+ on 2026-08-31, so `apply` writes nothing there.
636
+ - **Alternatives considered:**
637
+ - *Revert the box with `apply`.* Rejected by D-020: the edit is the owner's,
638
+ dated, and the first post under it went out at 14:04 IDT the same day.
639
+ - *Keep single-emoji markers and ask the owner to pick seven singles.*
640
+ Rejected: the owner chose pairs deliberately ("תן גם רצפים של אמוג'י") and
641
+ the lint is the thing that has to follow the standard, not the reverse.
642
+ - *Compare markers with their selectors.* Rejected: the bot wrote ❤ with
643
+ U+FE0F and 🖤 without in the same line, and nothing guarantees the next
644
+ model or phone does the same. Two fixtures pin the normalisation in both
645
+ directions (`scan-marker-without-selector`, `scan-marker-with-extra-selector`)
646
+ because a mutation that removed it passed every other fixture.
647
+ - *Add Maccabi Tel Aviv as a ninth article research topic.* Not done: the
648
+ article checklist already names מכבי ת"א under item 5 and the standard's
649
+ coverage line is the authority for what the space covers; if the club's
650
+ league news stops reaching the article once its European run is over,
651
+ that is a task-prompt change on its own evidence.
652
+ - **Reasoning:** `space-profile check` reported the drift within the hour and
653
+ an `apply` would have written the 2026-08-31 emoji back over the owner's
654
+ decision — the same shape as the `feed-watch.terms` clobber of 2026-08-31.
655
+ The lint allowlist compared one emoji at a time, so every two-emoji marker
656
+ would have been reported twice (not allowed, and two emoji on one line) the
657
+ moment the fixtures caught up with the standard. The invisible selector is
658
+ the CLAUDE.md carrier trap in a new coat: it is spelled as an escape in
659
+ `standard.json`, the fixtures and the tests, and a consistency test fails if
660
+ the literal character ever appears in the JSON.
661
+ - **Revisit if:** the owner changes a marker again from chat — then the
662
+ repo-side steps are the same four files (`AGENTS.md`, `standard.json`,
663
+ `cases.json`, the seeded `MEMORY.md` list) and this entry's date moves; or
664
+ the lint ever needs a marker of three emoji, which `leadingEmoji` already
665
+ accepts but no fixture pins.
666
+ - **Used by:** football-reporter profile (`AGENTS.md` § תפקיד and § פורמט,
667
+ `standard.json`, `config.yaml` `workspace_seed`), `src/text/reporter-lint.ts`
668
+ (`leadingEmoji`, `emojiKey`), M4.1 regression map (F7c), `MEMORY.md` seed
669
+
670
+ ## D-024: `feed-watch.max_per_hour` is 3, with the 15-minute batch kept (decided 2026-09-03)
671
+ - **Category:** run budget / process (applies [D-020](#d-020-live-tuning-is-adopted-into-the-repo-not-reverted--and-scans-carry-a-fixed-club-emoji-decided-2026-08-31) a third time)
672
+ - **Decided:** The scan cap is **3 runs per hour**, not the 1 set on
673
+ 2026-08-31. The owner set it from the dashboard on 2026-09-03 at 01:19 IDT
674
+ and confirmed it in writing the same afternoon ("I want max_per_hour=3").
675
+ `batch_minutes` stays at 15, so at most four batches an hour can open and
676
+ the cap only bites on a burst. The repo adopts the value; `apply` no longer
677
+ reverts it and the sync's profile check goes clean.
678
+ - **Alternatives considered:**
679
+ - *Keep 1/hour and treat the dashboard write as an experiment.* Rejected: it
680
+ had been live for a day, the owner was asked and chose 3, and D-020 says the
681
+ repo follows the box, not the reverse.
682
+ - *Raise the cap and shrink the batch back to 4 minutes.* Not done: the
683
+ 15-minute batch is what merges several outlets carrying one story into one
684
+ run, and that is the cheaper half of the 2026-08-31 change. The cap and the
685
+ batch are separate levers; only the cap moved.
686
+ - **Reasoning:** the owner's; no reason was given and none is recorded here.
687
+ What the numbers say: the cost law is runs × prefix, the 2026-08-31 cut was
688
+ aimed at the ~70 % of runs answering NO_UPDATE, and the 15-minute batch is
689
+ the half of that change that merges one story's outlets into one run. Spend
690
+ under 3/hour with the 15-minute batch on 2026-09-03 was $4.92 by 14:04 IDT
691
+ for 18 runs (10 posted), in line with the two days before, so the cap was
692
+ not the binding factor that day.
693
+ - **Revisit if:** a day passes 30 runs or the NO_UPDATE share climbs back
694
+ above half — then the batch, not the cap, is the first lever; or the real
695
+ group is linked and the owner wants a quieter feed.
696
+ - **Used by:** football-reporter profile (`config.yaml`), profile README,
697
+ `space-profile check` on every sync
@@ -205,17 +205,55 @@
205
205
  the owner and a second WhatsApp account, with the snapshot/restore
206
206
  procedure of §5. The bench does not try to make this list empty.
207
207
  - **Alternatives considered:** a Baileys stub; a second bot number in
208
- shadow mode.
208
+ shadow mode; **a test bot** — a snapshot restored on another host
209
+ (M1.6), the candidate Mercury version installed, linked to WhatsApp with
210
+ its *own* number, and driven from the owner's phone (owner's question,
211
+ 2026-09-06).
209
212
  - **Reasoning:** the memory `mercury-member-role-cannot-be-simulated` and
210
213
  the F5 debug doc both record the same wall: `isSelfJid` reads a live socket.
211
214
  D-003 moves the *decisions* out from behind that wall; the socket itself is
212
215
  not our code and is not worth stubbing. A private lab group with a second
213
216
  account is cheap and is what the football space already is until the real
214
217
  group is linked.
218
+
219
+ The test bot is **complementary, not a replacement**, and it is not safe
220
+ as described. It exercises what the shadow cannot — the socket list
221
+ above — but a snapshot linked to a different number breaks on identity:
222
+ every space is reached through `conversations.external_id`, the live
223
+ chat JID (`src/storage/db.ts`, `conversations`), so the copied groups
224
+ belong to a bot that is not a member of them and the copied `main` space
225
+ points at the owner's real number. Three consequences, each of which
226
+ the shadow avoids by having no adapter at all: (1) a copied scheduled
227
+ task whose space resolves to the owner's DM **sends from the test number
228
+ to the owner's real phone**, and one bound to a group the test bot has
229
+ been added to reaches the group; (2) the test bot cannot be added to the
230
+ real groups without reaching real people, so it needs its own groups,
231
+ whose JIDs differ, so the copied roles, tasks, config and workspace no
232
+ longer apply until the snapshot's space→chat links are remapped; (3) D-002
233
+ makes "adapter on" mean "live", so every refusal of D-005 (broker,
234
+ publish, billing, deliver) is off — a copied task with a broker action is
235
+ real. What a test bot cannot do at all: act *as* a specific existing
236
+ member (a non-admin quoting a bot message is the F5 shape), which
237
+ `shadow say --as <member>` replays from the copied `space_roles`. What it
238
+ does better: everything in §5, plus a candidate version on a real socket
239
+ before `promote` (D-013). Where it is redundant: a migration rehearsal
240
+ (apply → diff) needs no socket and is the shadow's own job (M1.4, M3.2).
215
241
  - **Revisit if:** the owner does not want to hold a second number — then the
216
242
  §5 list becomes the standing content of `pending-verification.md` and the
217
- bench's coverage is everything else.
218
- - **Used by:** lab-space-procedure
243
+ bench's coverage is everything else. **Or** the owner wants the test bot —
244
+ then it is M4.2 and lands only with all three preconditions, each denying
245
+ on absence: a new opt-in mode value (never "`MERCURY_SHADOW` unset plus a
246
+ list" — absence is live, D-002) that keeps every D-005 refusal while
247
+ constructing the adapter; a **recipient allowlist** of literal chat JIDs
248
+ on the send side (none exists today — `direct-send.ts` resolves a
249
+ recipient, it does not restrict one), checked at the bridge so all
250
+ fifteen send sites of `docs/live-testing.md` §3 are behind it, empty
251
+ list = nothing leaves; and a `shadow snapshot --remap <live-jid>=<test-jid>`
252
+ that rewrites `conversations.external_id` (and `message_platform_ids`)
253
+ so the copied spaces point at the test groups, refusing any live JID left
254
+ unmapped. Tasks whose space maps to nothing are deactivated in the copy,
255
+ not left to fail silently.
256
+ - **Used by:** lab-space-procedure; test-bot-instance (M4.2, if chosen)
219
257
 
220
258
  ## D-008: No injection route on the live instance in this goal
221
259
  - **Category:** scope
@@ -1,7 +1,7 @@
1
1
  # Roadmap: Rehearsal Bench
2
2
 
3
3
  **Goal**: [rehearsal-bench](goal.md)
4
- **Last updated**: 2026-09-03 (re-scope after the VPS cutover review, [note](../../notes/2026-09-03-release-gate-and-test-environment.md):
4
+ **Last updated**: 2026-09-06 (M4.2 `test-bot-instance` added as an owner's-decision story after the "why not a test bot on a snapshot?" review — D-007 extended with the identity/egress/remap analysis; nothing else moved). Earlier: 2026-09-03 (re-scope after the VPS cutover review, [note](../../notes/2026-09-03-release-gate-and-test-environment.md):
5
5
  M1.3 gains the instance-label prerequisite and the fleet-box shape; M1.6
6
6
  `snapshot-across-hosts` added; M3.2 extended to tagula's space-sync path;
7
7
  M3.3 `sync-rehearse` **moved** to [`release-gate`](../release-gate/roadmap.md)
@@ -99,6 +99,7 @@ in `pending-verification.md` for that fix is replaced by the transcript.
99
99
  | ID | Story | Slug | Depends on | Status |
100
100
  |----|-------|------|------------|--------|
101
101
  | M4.1 | `lab` space: private group of bot + owner + second account; `mercury lab snapshot|restore` = `VACUUM INTO` + tar before, `messages`/`space_roles`/`token_usage` window delete + workspace restore after; `restore` proves itself by re-running `shadow diff` against the pre-snapshot and requiring an empty report | lab-space-procedure | M1.4 | backlog (needs a second WhatsApp number — owner's decision) |
102
+ | M4.2 | **Test-bot instance** (added 2026-09-06, D-007 revisit): a snapshot restored on another host (M1.6) and linked to WhatsApp with its own number, driven from the owner's phone — the socket-only list of §5 on a candidate version, before `promote`. Complementary to the shadow, never a replacement (it cannot act *as* an existing member; a migration rehearsal needs no socket). Lands only with three guards, each denying on absence: a new opt-in mode that constructs the adapter but keeps every D-005 refusal (absence is still live, D-002); a send-side **recipient allowlist** of literal chat JIDs checked at the bridge behind all §3 send sites, empty = nothing leaves; `shadow snapshot --remap <live-jid>=<test-jid>` rewriting `conversations.external_id` + `message_platform_ids`, refusing an unmapped live JID and deactivating tasks whose space maps to nothing. Without the remap a copied task delivers from the test number to the owner's real DM | test-bot-instance | M1.6, M1.4 | backlog (needs a second bot number and a host — owner's decision) |
102
103
 
103
104
  **Checkpoint:** the §5 list of `docs/live-testing.md` is walked once in the
104
105
  lab space and every item is either ticked with the date or moved into M2 as
@@ -153,6 +154,7 @@ M1.1 ─ M1.5 │
153
154
  M2.1 ──────┬─ M2.2 ─┴─ M2.3 ─ M2.4
154
155
  M1.3 ──────┘ └────────────────────► release-gate R3 (smoke runner)
155
156
  M1.4 ─ M4.1
157
+ M1.6 ─ M4.2 (owner's decision)
156
158
  release-gate R5 (sync-rehearse, ex-M3.3) ◄─ M3.1, M2.3
157
159
  ```
158
160
 
@@ -110,3 +110,89 @@
110
110
  - **Revisit if:** a second operator (not tagula) needs the same policy —
111
111
  then it is a Mercury feature, planned as one.
112
112
  - **Used by:** smoke-runner, release-gate-rules
113
+
114
+ ## G-007: Mercury publishes its own agent image; visibility stays private for now
115
+ - **Category:** architecture / distribution
116
+ - **Status:** **Decided 2026-09-06/07 by the owner in two parts.** *Publish:*
117
+ yes — `release.yml` builds and pushes the image (R0.3, landed).
118
+ *Visibility:* **private, for now** — the owner's pushback on 2026-09-07
119
+ was right that nothing forces public: both consoles already pull private
120
+ GHCR images with a PAT, so private is proven, and a package inherits the
121
+ repo's private visibility on first push, so private is also the
122
+ zero-action outcome. The earlier "public" wording here was written from
123
+ the first half of the answer before the second was given; it is amended,
124
+ and the record of both branches is kept below.
125
+ - **What this commits to:** `release.yml` pushes
126
+ `ghcr.io/avishai-tsabari/mercury-agent:<version>` and `:latest` after
127
+ npm publish (R0.3). The package is NOT flipped to public. A third-party
128
+ install path therefore still does not exist, and the README must say so
129
+ honestly rather than imply one — that is the follow-up private incurs.
130
+ R0.4 may pin the default tag to `:${pkg.version}` once per-version tags
131
+ are being pushed; a pull token stays a requirement of every box, as it
132
+ is today.
133
+ - **What public would buy, recorded for the revisit:** not disclosure — the
134
+ image is built from a tarball anyone can already `npm pack`, and the moat
135
+ is the profiles, in neither artifact. It buys the deletion of a
136
+ credential class from the owner's own fleet: no PAT base64-embedded in
137
+ cloud-init userdata (readable back from the Hetzner metadata service),
138
+ no plaintext copy in the node `.env` for Watchtower, no root-0600 token
139
+ file kept on every box for its lifetime, no manual per-box rotation
140
+ runbook, and no "PAT expired silently, box looks healthy until first
141
+ use" failure class — each of which the two consoles document about
142
+ themselves. Against that: a support surface outsiders can use and will
143
+ expect to keep working, and the end of "the agent image is a private
144
+ product" as a positioning stance.
145
+ - **Do regardless of visibility:** the PAT in `mercury-cloud-console`'s
146
+ userdata and node `.env` reads every package on the account. A dedicated
147
+ machine user with `read:packages` on this one package shrinks the blast
148
+ radius; the fleet's own 2026-08-13 note already recommends it.
149
+ - **Question (as it stood):** Mercury's release workflow publishes the npm package and
150
+ nothing else. The agent image every fleet box runs is built and pushed by
151
+ `agent-image.yml` in **tagula-agents**, from Mercury's published npm
152
+ tarball. Mercury's own `ghcr.io/avishai-tsabari/mercury-agent` is pushed
153
+ by hand (`docs/runbooks/publish-checklist.md`, "manual — CI does not do
154
+ this"), is stale, and returned 401/403 unauthenticated on 2026-09-05.
155
+ Should Mercury publish its own image from `release.yml`, and public or
156
+ private?
157
+ - **Evidence (2026-09-05):** `container/Dockerfile` has zero `LABEL` lines;
158
+ `container/build.sh:20` tags `:latest` only; `release.yml` has no
159
+ `packages: write`; `src/config.ts:59` defaults to
160
+ `ghcr.io/avishai-tsabari/mercury-agent:latest`;
161
+ `container-runner.ts:1131-1136` never refreshes a tag that exists locally;
162
+ `container-runner.ts:1104-1117` skips validation for exactly the
163
+ `/mercury-agent:` names that broke. A fresh outside install therefore
164
+ fails `ensureImage()` and must fall back to a ~2.8 GB local `mercury build`
165
+ the README's install steps never mention. The tagula box could not perform
166
+ that fallback: 2 vCPU, 3 GB RAM and 1.9 GB free before the 2026-09-05
167
+ cleanup.
168
+ - **The two answers, and what each commits to:**
169
+ - **Public.** One publishing path, third-party installs work, R0.4's
170
+ version pin becomes safe, `agent-image.yml`'s version-linking half
171
+ retires. It is still an open-core boundary decision rather than a CI
172
+ one — but the disclosure half of that objection does not survive the
173
+ evidence. **Verified 2026-09-06, unauthenticated:** `npm pack
174
+ mercury-agent@0.18.2` returns a 378-entry tarball carrying all of
175
+ `src/`, `container/Dockerfile` and every source its `COPY` list names.
176
+ The image is built *from that tarball*, so publishing it discloses
177
+ nothing npm already publishes to anyone who asks. The moat is the
178
+ profiles, and they are in neither artifact. What public does commit to
179
+ is a **support surface**: an install path outsiders can use, and will
180
+ therefore expect to keep working.
181
+ - **Private (or no change).** Nothing outside the fleet can install
182
+ Mercury. Then say so: `docs/container-lifecycle.md:225` currently
183
+ claims "Images are published on each release", which is false today
184
+ either way, and the README's install steps imply an independent
185
+ install that does not exist.
186
+ - **Reasoning (publish half):** one publishing path beats two, and the
187
+ path belongs with the artifact it publishes. **(visibility half):** the
188
+ cost private carries is one the fleet already pays for its own image, so
189
+ keeping it costs nothing new today; public is reversible with one click
190
+ later, and the moment to decide it with full information is when
191
+ `agent-image.yml` retires — that is when the pull token could actually
192
+ become deletable.
193
+ - **Not in scope of this decision:** the fleet's idle check, database
194
+ snapshot, rollback and stagger. Those are topology and stay in
195
+ tagula-agents whatever is decided here (G-006).
196
+ - **Revisit if:** `agent-image.yml` in tagula-agents retires (decide visibility then, with the token's future known); a second operator appears; or Mercury is open-sourced.
197
+ - **Used by:** mercury-publishes-image (R0.3), image-refresh-and-visibility (R0.4)
198
+
@@ -1,7 +1,7 @@
1
1
  # Roadmap: Release Gate
2
2
 
3
3
  **Goal**: [release-gate](goal.md)
4
- **Last updated**: 2026-09-03 (created from the VPS cutover review; story slugs are `/f-feature-planning` inputs)
4
+ **Last updated**: 2026-09-07 (G-007 amended: publish yes, private for now. R0.1–R0.3 Mercury side landed; lint drift cleared, release path green)
5
5
 
6
6
  > Order is by leverage against the live risk, not by size: R1 protects the
7
7
  > customer box within a session or two and depends on nothing; R2 is the
@@ -11,6 +11,33 @@
11
11
 
12
12
  ## Milestones
13
13
 
14
+ ### Milestone R0: The agent image is a published, self-describing artifact
15
+
16
+ > R1.1(c) asks the image what mercury version it is and which `COPY`
17
+ > sources it carries. **Nothing puts either there today**, and nothing in
18
+ > this repo publishes the image at all, so R1.1(c) has no artifact to read
19
+ > and no milestone owned the gap. This is that milestone. It is small,
20
+ > it depends on nothing, and R1.1(c) is blocked without R0.1.
21
+ >
22
+ > Established 2026-09-05 while releasing 0.18.1/0.18.2 (evidence in
23
+ > [`decisions.md` G-007](decisions.md)):
24
+ > `container/Dockerfile` has **zero** `LABEL` lines; `container/build.sh`
25
+ > tags `:latest` only, never a version; `release.yml` has no image job and
26
+ > no `packages: write` anywhere; the image every fleet box actually runs is
27
+ > built and pushed by `agent-image.yml` in **tagula-agents**, from
28
+ > Mercury's published npm tarball. Mercury's own
29
+ > `ghcr.io/avishai-tsabari/mercury-agent` is pushed by hand per
30
+ > `docs/runbooks/publish-checklist.md` ("CI does not do this"), is stale,
31
+ > and is **not anonymously pullable** (401/403 verified 2026-09-05).
32
+
33
+ | ID | Story | Slug | Depends on | Status |
34
+ |----|-------|------|------------|--------|
35
+ | R0.1 | **The image describes itself.** `container/Dockerfile` + `container/build.sh` stamp, at build time, (a) `org.opencontainers.image.version` and a mercury-namespaced version label taken from `package.json`, not hand-typed, and (b) the **`COPY`-source manifest** R1.1(c) checks against — generated from the Dockerfile's own `COPY` list so the two cannot drift. `build.sh` tags `:<version>` as well as `:latest`. Labels are inherited by the derived `mercury-agent-ext-*` image with no builder change (verified on the tagula box 2026-09-05: derived and base both report `0.18.0-beta.0`), so the host-side check works through the image actually executing | image-self-describing | — | **landed 2026-09-06** — `src/agent/image-manifest.ts` generates the manifest inside the build from the Dockerfile that is building, `container/Dockerfile` stamps `org.opencontainers.image.version` + `com.mercury.version` from an `ARG` and writes `/app/image-manifest.json`, `build.sh` reads the version from package.json and tags `:<version>` + `:latest`. Parser and the host-side `diffImageManifest` are unit-tested; **the build itself is unverified** — four checks in `docs/pending-verification.md` |
36
+ | R0.2 | **The build-integrity gate moves into this repo.** `agent-image.yml` in tagula-agents verifies Mercury's artifact — every `COPY` source present in the tarball, `bun build container-entry.ts` resolves inside the image — and that check is what would have caught 0.17.0. It belongs in `release.yml` (or `bun run check`) so every consumer building the image gets it, not only this one fleet. The fleet keeps its copy or drops it; either is fine once this exists | build-integrity-in-mercury | — | **landed 2026-09-06** — split by what it needs: `scripts/check-image-sources.ts` runs in `bun run check` with no Docker and proves the source contract (every COPY source exists, is inside the npm `files` allowlist, and every local module reachable from a copied `.ts` is itself copied — verified to reproduce the 0.17.0 miss by deleting the `sanitize-text.ts` COPY); `smoke:image` gained the artifact half from `agent-image.yml` — toolchains, the pi binary, the version label, and `bun build` resolving `container-entry.ts` and `mrctl.ts` inside the image |
37
+ | R0.3 | **Mercury publishes its own image.** A job in `release.yml`, after `publish-npm`, that installs the published tarball, builds, verifies (R0.2) and pushes `:<version>` + `:latest` with the R0.1 labels. Needs `packages: write`, which no job has today. **G-007 is decided: publish yes, visibility private for now** (2026-09-07), so this tidies the internal path; the third-party install path stays closed until visibility is revisited. Makes `agent-image.yml`'s *version-linking* half redundant; its nightly updater, idle check, snapshot, rollback and stagger stay where they are. **Move that workflow rather than rewriting it** — it already builds from the published tarball, which is the shape wanted here — and carry three traps across with it, all read out of it on 2026-09-06: (i) it derives the image name from `GITHUB_REPOSITORY`, which under this repo yields a different path than the fleet's, so set the name explicitly instead of inheriting it; (ii) a GHCR package takes the repo's visibility on first push, so a *public* answer to G-007 needs the package flipped by hand after the first run — it cannot be expressed in the YAML; (iii) the job runs **after** `publish-npm`, so a failed image push leaves npm ahead of the image, which is the 0.17.0 ordering inverted — keep boxes following the image (R0.4/R0.5) and make the job required so the gap is loud rather than latent | mercury-publishes-image | R0.1, R0.2 | **landed 2026-09-06 (Mercury side)** — `release.yml` gained a `publish-image` job with `packages: write`: builds from the published tarball, verifies with `smoke:image`, pushes `:<version>` and (stable only) `:latest`, image name set explicitly so it is not derived from `GITHUB_REPOSITORY`. Inert until the next `v*` tag. **One thing it cannot do itself**: retire `agent-image.yml` in tagula-agents — the owner's call. The package stays private (G-007), which is what it inherits by default, so no visibility step is pending |
38
+ | R0.4 | **A floating tag stops meaning "frozen at first pull".** `ensureImage()` returns the moment `docker image inspect` succeeds, so the default `:latest` is never refreshed and a host upgraded by npm keeps yesterday's container silently — the trap the owner hit on 2026-09-05. Either pin the default to `:${pkg.version}` (safe **only** after R0.3, or every fresh install fails on a tag nobody published) or re-pull a floating tag on a version change. Either way boot logs the resolved image ref, digest and version label so drift is visible without asking. Filed as `docs/bugs/ensure-image-never-refreshes-a-floating-tag.md`; take `docs/bugs/health-reports-stale-env-version.md` with it — `/health` reports `MERCURY_VERSION` from the environment and never the installed `package.json`, so the host half of "what is actually running here" lies for the same reason the image half does, and fixing one without the other leaves the console's version badge still wrong | image-refresh-and-visibility | R0.3 | backlog |
39
+ | R0.5 | **`mercury upgrade` owns both halves.** `upgradeAction` contains no Docker call at all: it stops the service, npm-installs and restarts, leaving the image behind. It should pull the matching image first and refuse when that tag does not exist — image first, host second, the rule the fleet updater already follows and the one 0.17.0 broke | upgrade-pulls-image | R0.1 | backlog |
40
+
14
41
  ### Milestone R1: Preflight
15
42
  > After this, a box that cannot run a container says so at boot and on
16
43
  > demand, with one diagnosis per blocker, and the nightly updater has
@@ -18,7 +45,7 @@
18
45
 
19
46
  | ID | Story | Slug | Depends on | Status |
20
47
  |----|-------|------|------------|--------|
21
- | R1.1 | `mercury preflight [--json] [--space <id>]`: one command that runs, on the production launch shape and the pulled image, (a) the bwrap probe and the runsc host-uds probe (folded from `probeBwrapSandbox` / `probeRunscHostUds`, one implementation, boot calls the same function); (b) an `mrctl whoami` round trip from inside a container; (c) the **host-vs-image contract** — the image carries every file the host version's `container/Dockerfile` `COPY` list names, checked by a manifest baked at build time, and the image's recorded mercury version matches the host's; (d) the model chain's first leg resolves a credential *the way the runtime resolves it* (the `modelProvider` default vs `models.json` gateway case); (e) host deps — `pi` on `PATH` when any host-side pi spawn is configured, timezone not UTC-by-accident when any active task has `timezone:null`, swap present on ≤4 GB; (f) an extension load report with the reason for every skip (the silent `gws` case). Exit non-zero on any *blocked* verdict, zero with warnings otherwise; JSON report with per-check verdict, evidence and remedy. Runs in shadow mode too | preflight | — | backlog — **first** |
48
+ | R1.1 | `mercury preflight [--json] [--space <id>]`: one command that runs, on the production launch shape and the pulled image, (a) the bwrap probe and the runsc host-uds probe (folded from `probeBwrapSandbox` / `probeRunscHostUds`, one implementation, boot calls the same function); (b) an `mrctl whoami` round trip from inside a container; (c) the **host-vs-image contract** — the image carries every file the host version's `container/Dockerfile` `COPY` list names, checked by a manifest baked at build time, and the image's recorded mercury version matches the host's; (d) the model chain's first leg resolves a credential *the way the runtime resolves it* (the `modelProvider` default vs `models.json` gateway case); (e) host deps — `pi` on `PATH` when any host-side pi spawn is configured, timezone not UTC-by-accident when any active task has `timezone:null`, swap present on ≤4 GB; (f) an extension load report with the reason for every skip (the silent `gws` case). Exit non-zero on any *blocked* verdict, zero with warnings otherwise; JSON report with per-check verdict, evidence and remedy. Runs in shadow mode too | preflight | — | in-progress — **first** |
22
49
  | R1.2 | `mercury doctor` calls `loadConfig` and delegates its environment checks to R1.1 so there is one implementation; boot logs the same report once; a *blocked* verdict at boot refuses spawns with the diagnosis (already the case for the two probes — generalised) | doctor-delegates-to-preflight | R1.1 | backlog |
23
50
 
24
51
  **Checkpoint:** on the tagula box (with the owner's approval and in a cron
@@ -94,6 +121,10 @@ manual check in `pending-verification.md` is replaced by the reports.
94
121
  ## Dependency Graph
95
122
 
96
123
  ```
124
+ R0.1 ─┬─ R1.1(c)
125
+ ├─ R0.5
126
+ └─ R0.3 ─ R0.4 (G-007: publish yes, private for now)
127
+ R0.2 ─── R0.3
97
128
  R1.1 ─┬─ R1.2
98
129
  ├─ R2.1 ─ R2.2
99
130
  ├─ R4.1 ─ R5.2
@@ -103,7 +134,9 @@ R3.1 ─ R3.2 ─┬─ R3.3
103
134
  rehearsal-bench M2.2 (function half) ─► R3.2
104
135
  ```
105
136
 
106
- R1.1 first, alone. R2 and R4 can run beside R3.1 in separate worktrees.
137
+ R0.1 is the true first story: it is small, depends on nothing, and R1.1(c)
138
+ cannot be built without it. R0.3 no longer waits on anything but R0.1 and R0.2.
139
+ R1.1 otherwise first, alone. R2 and R4 can run beside R3.1 in separate worktrees.
107
140
  R3.2 is the join with the bench. R5 last.
108
141
 
109
142
  ## Cost note
@@ -243,6 +243,22 @@ restarting, applying, deleting — whoever runs them, agent or owner.
243
243
 
244
244
  ## 7. Until the bench exists — what to do today
245
245
 
246
+ - **`mercury preflight` is today's gate.** One command, run from
247
+ `~/whatsapp-bot`, that proves this box can run a turn: both boot probes on
248
+ demand, an `mrctl` round trip in a container on the production launch shape,
249
+ the host-vs-image contract, the first model leg's credential, host deps and
250
+ the extension load report. Exit 1 on any blocker, 0 with warnings. It makes
251
+ no model call and costs no tokens, writes nothing but its report, and never
252
+ sends to a person — so it is safe on the live box at any hour. Run it after
253
+ every deploy, before declaring one done, and read the JSON report rather
254
+ than the terminal. The check-by-check table — what each id proves and
255
+ which verdict class it can return — is in `docs/container-lifecycle.md`
256
+ § "`mercury preflight`". It needs no configuration of its own: since
257
+ 2026-09-06 the round trip falls back to header identity when no
258
+ `MERCURY_CALLER_TOKEN_KEY` is set, and names the mode it used in the
259
+ evidence (`callerAuth`) — a `headers` pass proves transport, `API_SECRET`
260
+ and identity but not the caller-token leg, so set the key when you want
261
+ that leg covered too.
246
262
  - **Send-free, admin-only, DM-only:** `mercury chat -s <space>` /
247
263
  `POST /chat` runs the real pipeline and returns the reply as JSON; it never
248
264
  sends. It seeds the caller as admin (`routes/chat.ts:241`) and forces
@@ -327,6 +343,12 @@ The doctrine still holds, whoever implements it:
327
343
  connection open`; once a day, `SELECT MAX(created_at) FROM token_usage`
328
344
  — a bot that has not billed a run in a day is a bot nobody is talking
329
345
  to or a bot that is down, and only a person can tell which.
346
+ 6. **The evidence has a file.** `mercury preflight` writes
347
+ `<dataDir>/preflight/<timestamp>.json` and a `latest.json` beside it. That
348
+ file, not the terminal scrollback, is what a deploy attaches — a gate whose
349
+ evidence was not persisted did not pass. It answers "can this box run a
350
+ turn", never "is the customer's scenario still working": that second
351
+ question is the functional harness's, and it is still open.
330
352
 
331
353
  ## 9. Known gaps (kept honest)
332
354
 
@@ -9,6 +9,89 @@ once every line under it is ticked.
9
9
 
10
10
  ---
11
11
 
12
+ ## history window cap 2026-09-03 — deploy-gated re-measure
13
+
14
+ `context.history_max_chars` (default 2,000) clips older messages in the
15
+ `context`-mode window before they reach the container. The projection against
16
+ the live corpus says football's window drops 30.8% on average and 38% at the
17
+ tail; the point of these lines is that the *continuity* did not pay for it.
18
+ The cap itself, its config keys and the dashboard row are host-side — a
19
+ restart deploys them. The 400 KB backstop reader that came with the same
20
+ commit is **not**: `3884506` also edits `src/agent/container-entry.ts`
21
+ (`history_budget_dropped_turns`), which `container/Dockerfile` bakes into the
22
+ agent image, so that half needs a base rebuild
23
+ (`docs/debug/major/2026-09-03-history-window-dominates-the-run-prefix.md`).
24
+ Corrected on top of `dev` @ `7b7bc5f` — S1 of
25
+ `docs/notes/2026-09-06-audit-fix-plan.md`; S4 ticks against this text.
26
+
27
+ - [ ] After the restart, `task_runs.prompt_prefix_tokens` for `football-friends`
28
+ over a full day is below the 2026-09-03 baseline (21 runs, avg 33,343,
29
+ max 53,273). Compare the same query, not a trace.
30
+ - [ ] `mercury service logs` carries `History window: older messages clipped`
31
+ with a non-zero `savedChars` — the operator's half of the reader. If it
32
+ never appears, the cap is not reaching the call site.
33
+ *Not yet seen, and not yet expected: the line fires only when a history
34
+ row exceeds the per-message cap (2,000 chars). The first run on the new
35
+ image (task 436, 2026-09-06 13:33 IDT, `mercury-agent-ext:a5a60d5b20c9`)
36
+ had only short scan posts in its window — `prompt_prefix_tokens` 27,510
37
+ against the 33,343 baseline average. Re-check after the next 09:00
38
+ article is inside the window.*
39
+ - [ ] The dashboard's Context panel for `football-friends` shows a
40
+ `context.history_max_chars` row with the effective value, and setting it
41
+ there changes the next run's window (the key is per-space; a row the
42
+ runtime ignored would be the same invisible limit this fix removes).
43
+ - [ ] A run whose window contains a clipped post shows the
44
+ `[truncated: N of M characters omitted from this older message]` line
45
+ inside `<history>` in that run's trace, and **not** inside
46
+ `<reply_anchor>`.
47
+ - [ ] Continuity did not regress: ask the bot in the football group about
48
+ something it published the *previous* evening (not the newest turn) and
49
+ confirm it still knows what it posted. This is the check the bug doc
50
+ warned about — shrinking the window blind trades a cost problem for a
51
+ memory problem.
52
+ - [ ] A swipe-reply to a long bot post still quotes the post in full; the
53
+ anchor is never clipped.
54
+ - [x] The derived image carries the backstop reader:
55
+ `docker run --rm --entrypoint grep mercury-agent-ext:<hash> -c
56
+ history_budget_dropped_turns /app/src/agent/container-entry.ts` prints
57
+ `1`. Grep the image, never the repo — a restart alone leaves this at `0`.
58
+ *Verified 2026-09-06 13:10 IDT (S4): `dev` `631c3b5` synced, base rebuilt
59
+ 13:00–13:05, derived image `mercury-agent-ext:a5a60d5b20c9` built by the
60
+ service at 13:10; the grep printed `1` inside it (and `2` for
61
+ `shouldOverridePiSystemPrompt`, `2` for `EPISODE_MIN_STEM_CHARS`).*
62
+
63
+ ## club markers 2026-09-03 (D-023) — repo adopts the owner's chat edit; deploy-gated
64
+
65
+ The owner changed the seven club markers and added Maccabi Tel Aviv to the
66
+ standing coverage from the football chat on 2026-09-03 (12:35–12:55 IDT); the
67
+ bot wrote the live `AGENTS.md` and `MEMORY.md` itself, so **nothing changes on
68
+ the box at deploy** — the deploy only stops `apply` from reverting it.
69
+
70
+ - [x] After the merge is deployed: `space-profile check` against `dev` reports
71
+ **no** `AGENTS.md` drift. The only expected line is
72
+ `feed-watch.max_per_hour` (live `3`, set from the dashboard 2026-09-03
73
+ 01:19 IDT; profile `1`) — the owner's call, not this change's.
74
+ *Verified 2026-09-03 14:50 IDT: `dev` merged locally as `6aff14f`, synced
75
+ to the box; the sync's own check printed exactly one difference, the
76
+ `max_per_hour` line. Base image rebuilt (the deploy also carried
77
+ `e8b50e4`/`b9b5c94`, which touch baked files), derived image
78
+ `mercury-agent-ext:eac63950bd82` verified by grep at 14:55.*
79
+ *Post-`0e0a4e0` check, 2026-09-06 12:59 IDT (S4, `dev` `631c3b5`): the
80
+ sync's check printed exactly the two drifts the women's-football fix
81
+ introduced (`AGENTS.md` coverage line, `feed-watch.exclude` five terms)
82
+ and nothing else — `max_per_hour` no longer drifts since the profile
83
+ adopted `3` (D-024). `space-profile apply` at 13:10 wrote both; check
84
+ after apply: clean. Seed addendum (`52eaf67`) copied by hand into the
85
+ live `distill.md`; `diff` against the repo seed is empty.*
86
+ - [ ] The next scan post that names Barcelona or Beitar opens with the pair
87
+ (❤️💙 / 💛🖤) as one marker, and the harness (`bun run check` offline
88
+ lint) reports no `emoji-allowlist` on it. First live post under the new
89
+ set: message #4228, 2026-09-03 14:04 IDT, 👹 for United — already clean.
90
+ - [ ] A Maccabi Tel Aviv item in a scan opens with 💛💙. None has posted yet
91
+ (the club's last item, message #4226 at 13:43 IDT, was a Real Madrid one
92
+ that posted with no marker at all — a separate, pre-existing miss).
93
+
94
+
12
95
  ## Verification pass, 2026-08-31 — what the box could answer on its own
13
96
 
14
97
  Everything below was read from the live box (DB, journal, traces, images,
@@ -820,3 +903,98 @@ Host-side only: deploy = service restart, **no image rebuild** (`container-runne
820
903
  Host-side only: deploy = service restart, **no image rebuild** (`handler.ts` and `debounce.ts` are not baked).
821
904
 
822
905
  - [ ] MANUAL-VERIFY(2026-09-03): **A reply lands in the chat it was asked in.** On the VPS (`tagula-assistant`, space `tomer-ohana` has three linked groups), restart the service, then as one caller address the bot in group A with something slow, and while it is running address it in group B. Expect one reply in each group, each answering that group's own message. In `journalctl --user -u mercury.service`, the `Debounce: message queued behind an in-flight flush` line must now carry the thread in `key=` (`space:caller:thread`), and each `WhatsApp outbound chatJid=` must match the group its question came from.
906
+
907
+ ## credential-alert-misses-owner-dm-linked-to-named-space (2026-09-03)
908
+
909
+ Host-side only: deploy = service restart, **no image rebuild** (`src/core/direct-send.ts` is not baked).
910
+
911
+ - [x] MANUAL-VERIFY(2026-09-03): **A direct send to the owner's JID lands in the owner's DM.** Checked 2026-09-03 15:09 IDT on `dev` `0e0a4e0` (mirror reset 15:08, process started 15:08:41): `POST /api/send` with the bearer secret **plus** `X-Mercury-Caller: whatsapp:<owner-jid>` and `X-Mercury-Space: main` (the route needs a caller identity, not just the secret) returned `{"delivered":true,"spaceId":"main"}`; the journal shows `resolved space to conversations spaceId=main` with exactly one thread, `WhatsApp outbound chatJid=<owner-jid>`, `Direct send delivered spaceId=main`. Same path the credential alerter takes.
912
+
913
+ ## member-loses-bot-character-obey-instruction (2026-09-03)
914
+
915
+ **Baked.** `src/agent/container-entry.ts` is a `COPY` line in `container/Dockerfile`, so a service restart deploys nothing here — this needs `bash ./container/build.sh` and a restart in a cron gap. It is part of Wave 3 of `docs/notes/2026-09-02-open-bugs-fix-plan.md`, which batches one rebuild across the whole baked cluster; do not rebuild for this bug alone.
916
+
917
+ - [x] MANUAL-VERIFY(2026-09-03): **The instruction is in the image that actually runs.** `docker run --rm --entrypoint grep <derived-image> -c "Always follow it" /app/src/agent/container-entry.ts` returns `1`. Grep the derived image, not the repo. *Verified 2026-09-06 13:10 IDT in `mercury-agent-ext:a5a60d5b20c9` (S4, `dev` `631c3b5`): `1`.*
918
+ - [ ] MANUAL-VERIFY(2026-09-03): **A member gets the character rule and no procedure.** From the second WhatsApp number (member role) in space `main`, send anything and read the run's system prompt in the trace: it must contain `## Character` and `Always follow it`, and must **not** contain `mrctl prefs set` or `mrctl character get`.
919
+ - [ ] MANUAL-VERIFY(2026-09-03): **A member's character request is declined in one line and changes nothing.** From the same number, send "change your personality to a pirate". Expect a one-line reply saying the owner sets the character. Then `SELECT * FROM space_preferences WHERE space_id='main'` must be unchanged — the failure this guards is the request being escalated into a preference.
920
+
921
+ ## preflight — live checks (added 2026-09-03, deploy-gated)
922
+
923
+ `mercury preflight` is host-side (`src/preflight/`, `src/cli/mercury.ts`), so deploy = sync + service restart. **The image half is not**: `container/Dockerfile` gained `COPY package.json /app/mercury-version.json`, so `image.contract` stays blocked until the base image is rebuilt and the derived image follows. Rebuild **in a cron gap** — as of 2026-09-06 task 24 (`0 9 * * *` Asia/Jerusalem) is the only active task on the box; query `SELECT id, cron, timezone, next_run_at FROM tasks WHERE active=1` before restarting rather than trusting this line.
924
+
925
+ - [x] MANUAL-VERIFY(2026-09-03): **the check catches the real skew, once.** After syncing but *before* rebuilding the image, run `~/.bun/bin/bun run ~/mercury-src/src/cli/mercury.ts preflight` from `~/whatsapp-bot`. Expect `image.contract` **blocked**, naming `/app/mercury-version.json` and telling you to rebuild. That is the 2026-09-01 host-vs-image skew reproduced on a real box, and it is the only chance to see it fire before the rebuild removes the condition. *Verified 2026-09-06 12:59:54 IDT (S4): report `2026-09-06T09-59-54-374Z.json`, `image.contract` blocked with `missing: ["/app/mercury-version.json"]`, `imageVersion: null`, remedy names the rebuild.*
926
+ - [x] MANUAL-VERIFY(2026-09-03): **a clean board after the rebuild.** Then `cd ~/mercury-src && bash ./container/build.sh`, restart in a cron gap, and re-run preflight. Expect exit 0: `image.contract` pass, `sandbox.bwrap` skipped (`containerBwrapDockerCompat` on this box), `container.mrctl-roundtrip` pass, `credential.first-leg` pass. Read `~/whatsapp-bot/.mercury/preflight/latest.json`, not the terminal — the report file is the evidence, and `docker ps -a --filter label=mercury.managed=true` must show no leftover `mercury-contract-probe-*` or `mercury-roundtrip-probe-*` container. *Run 2026-09-06 13:10:26 IDT after the rebuild (report `2026-09-06T10-10-26-466Z.json`): 8 pass · 0 warn · **1 blocked** · 2 skipped, exit 1. `image.contract` **pass**, `sandbox.bwrap` skipped, `credential.first-leg` pass, no leftover probe container — but `container.mrctl-roundtrip` blocked with `401 Invalid or expired caller token`: this box has no `MERCURY_CALLER_TOKEN_KEY`, so the preflight process signs with its own per-process random key and the service verifies with another. Not a wrong `.env` and not fixable by re-running — filed as `docs/bugs/preflight-roundtrip-cannot-pass-without-configured-caller-token-key.md`. Left unticked until that bug is fixed or the owner sets the key.* **Ticked 2026-09-06 18:20 IDT:** the owner set `MERCURY_CALLER_TOKEN_KEY` in the box's `.env` (one line, value never displayed) and restarted at 18:19:58; `latest.json` at 18:20:34 reads exit 0, **9 pass · 0 warn · 0 blocked · 2 skipped**, `container.mrctl-roundtrip` pass with `answeredCallerId: "preflight"`, `answeredSpaceId: "main"`, `role: "member"`, image `a5a60d5b20c9`; `docker ps -a --filter label=mercury.managed=true` empty. The bug stays open for boxes that never set the key.*
927
+
928
+ The three below close audit findings 1, 2 and 3 (2026-09-06). Each is a one-liner on the box and each fails on the *live* shape the unit tests can only fake.
929
+
930
+ - [x] MANUAL-VERIFY(2026-09-06): **`--json` is parseable on a box that loads extensions.** `cd ~/whatsapp-bot && ~/.bun/bin/bun run ~/mercury-src/src/cli/mercury.ts preflight --json | python3 -m json.tool > /dev/null && echo PARSED`. This box loads real extensions and the fixture only loads one, so it is the honest version of the test. A `Loaded extension` line anywhere in stdout means the `configureLogger({ level: "warn" })` did not take. *Verified 2026-09-06 13:11 IDT: stdout 4,945 bytes, stderr 0 bytes, no non-JSON line, `JSON.parse` in bun returns the ten top-level keys (`schema … reportPath`). `python3 -m json.tool` exits 0 on the same output; the one "NOT PARSEABLE" seen first was `set -o pipefail` surfacing preflight's own exit 1 (the round-trip blocker), not a parse failure.*
931
+ - [x] MANUAL-VERIFY(2026-09-06): **the round trip leaves no member row.** Before the run: `SELECT COUNT(*) FROM space_roles WHERE platform_user_id='preflight'` → whatever it is now (it may be non-zero from runs before the fix — clean those up with `DELETE FROM space_roles WHERE platform_user_id='preflight'` and say so). Run preflight, then the same query → **0**. Also `mrctl roles list` and the dashboard members table must not show `preflight`. *2026-09-06: count was 0 before and 0 after both runs — but the probe was refused at the token check (401, see the round-trip line) before any role resolution ran, so this proves nothing yet; re-check once the round trip actually completes.* **Re-checked 2026-09-06 18:20 IDT after the key was set:** the round trip completed (`answeredCallerId: preflight`, `role: member`) and `SELECT COUNT(*) FROM space_roles WHERE platform_user_id='preflight'` → **0**.*
932
+ - [x] MANUAL-VERIFY(2026-09-06): **the credential source names the passthrough mode.** In `latest.json`, `credential.first-leg` evidence carries `envPassthrough` matching this box's `agent.env_passthrough` (`claimed` here), and — since the leg is Anthropic OAuth — `source` reads `env:MERCURY_ANTHROPIC_OAUTH_TOKEN (parsed, not refreshed)`. The qualifier is the point: preflight does not spend a refresh, so a pass here is "a usable credential is present", not "it works". *Verified 2026-09-06 13:10 IDT, with one correction to this line's premise: evidence `{"envPassthrough":"claimed","credentialEnvVar":"ANTHROPIC_OAUTH_TOKEN","source":"auth.json:oauth","authPath":"…/.mercury/global/auth.json","leg":"anthropic:claude-opus-5"}`. `envPassthrough` matches the box; `source` reads `auth.json:oauth` rather than the `env:…` form because on this box the token lives in `auth.json`, not in the environment — the check named the real source, the expectation here assumed the wrong one.*
933
+
934
+ ## env_passthrough default flipped to `claimed` 2026-09-06
935
+
936
+ Host-side only (`src/config.ts`, `src/main.ts`) — a restart deploys it, no
937
+ image rebuild. This box already pins `agent.env_passthrough: claimed` in
938
+ `~/whatsapp-bot/mercury.yaml`, so the flip is a **no-op here**; that is exactly
939
+ why the default itself has to be proven rather than assumed.
940
+
941
+ - [ ] MANUAL-VERIFY(2026-09-06): **the default is `claimed` with no config saying so.** Comment out the `env_passthrough: claimed` line in `~/whatsapp-bot/mercury.yaml`, confirm by name only that `MERCURY_CONTAINER_ENV_PASSTHROUGH` is absent from `~/whatsapp-bot/.env` (`grep -c` — never print the file), restart **in a cron gap** (`SELECT id, cron, timezone, next_run_at FROM tasks WHERE active=1`), then read `mercury service logs` for the startup line. Expect `Container env passthrough: claimed` with a `withheld=` field naming the unscoped `MERCURY_*` vars (names only — a value in that log is itself a bug). Restore the yaml line and restart, again in a gap.
942
+ - [ ] MANUAL-VERIFY(2026-09-06): **the escape hatch still logs the old warning.** With `agent.env_passthrough: all` set explicitly, the startup line reads ``Container env passthrough: explicit `all` override`` with a `vars=` field. This is the only branch a deployment relying on blind passthrough will ever see, so it has to name the vars it is carrying.
943
+
944
+ ## override_pi_system_prompt parsed strictly 2026-09-06
945
+
946
+ Host-side only (`src/config.ts`, `src/agent/container-runner.ts`) — a restart
947
+ deploys it, no image rebuild. This box sets `override_pi_system_prompt: true` in
948
+ `~/whatsapp-bot/mercury.yaml` and no env var, so the change is a **no-op here**:
949
+ the point of the line below is that it stayed one.
950
+
951
+ - [ ] MANUAL-VERIFY(2026-09-06): **the container still gets `true`.** After the restart, during a live run: `docker inspect $(docker ps -q --filter label=mercury.managed=true | head -1) --format '{{range .Config.Env}}{{println .}}{{end}}' | grep OVERRIDE` prints `OVERRIDE_PI_SYSTEM_PROMPT=true`. Read it off the running container, not off `mercury.yaml` — the value the container was handed is the only thing that decides which prompt mode pi runs in.
952
+ - [ ] MANUAL-VERIFY(2026-09-06): **a typo now stops startup instead of taking the bot down quietly.** Add `MERCURY_OVERRIDE_PI_SYSTEM_PROMPT=no` to `~/whatsapp-bot/.env`, restart **in a cron gap** (`SELECT id, cron, timezone, next_run_at FROM tasks WHERE active=1`), and expect the service to fail to start: `mercury service logs` carries `Startup failed` with an error naming `override_pi_system_prompt` (`main.ts:715` turns a `loadConfig` throw into exit 1) — not a bot that boots and answers every message with a 400. Remove the line and restart, again in a gap.
953
+
954
+ ## swipe-reply to media keeps its quote block 2026-09-06
955
+
956
+ Host-side only (`src/core/reply-context.ts`, `src/core/runtime.ts`) — a restart
957
+ deploys it, no image rebuild. The de-duplication that drops a swipe-reply's
958
+ `<reply_to>` block now makes an exception for a quote whose opening tag carries
959
+ `media_type=`, because that attribute is the only place in the whole prompt
960
+ where the quoted message's media is named. The tests cover the branch; what
961
+ they cannot cover is that a real WhatsApp voice-note quote actually produces the
962
+ attribute in the stanza.
963
+
964
+ - [ ] MANUAL-VERIFY(2026-09-06): **a swipe-reply to a voice note still names the audio.** In the private `main` space (never the football group), send a voice note, then swipe-reply to it with `מה אמרתי בהודעה הקולית?`. The newest `messages` row with `role='user'` contains `media_type="voice"` and `[voice note]`, and the bot's answer addresses the recording rather than an empty quote.
965
+ - [ ] MANUAL-VERIFY(2026-09-06): **a text quote is still de-duplicated.** In the same space, swipe-reply to one of the bot's own text messages. The stored user row is the question alone — no `<reply_to` — so the exception did not widen into "never strip".
966
+
967
+ ## preflight round trip without a caller-token key 2026-09-06
968
+
969
+ Host-side only (`src/preflight/checks/roundtrip.ts`) — a sync + restart deploys
970
+ it, no image rebuild. **This box now has `MERCURY_CALLER_TOKEN_KEY` set** (the
971
+ owner added it at 18:19 IDT as the workaround), so the default path the fix
972
+ exists for is *not* the path this box takes by default. The first line below is
973
+ therefore the whole point: it has to be checked with the key temporarily out of
974
+ the way, or it proves nothing.
975
+
976
+ - [ ] MANUAL-VERIFY(2026-09-06): **the round trip passes with no caller-token key.** From `~/whatsapp-bot`, run preflight with the key blanked for that one process only — `MERCURY_CALLER_TOKEN_KEY= ~/.bun/bin/bun run ~/mercury-src/src/cli/mercury.ts preflight` (an env var beats `mercury.yaml`, and an empty value is read as unconfigured). Never edit the `.env`. In the report, `container.mrctl-roundtrip` must be **pass** with evidence `callerAuth: "headers"` and `answeredCallerId: "preflight"`, and its title must say the token leg was not exercised. Read `~/whatsapp-bot/.mercury/preflight/latest.json`, not the terminal.
977
+ - [ ] MANUAL-VERIFY(2026-09-06): **the configured key is still used.** Run preflight normally (no env override). Same check must be **pass** with evidence `callerAuth: "token"` — the fallback must not have swallowed the real path on a box that has a key.
978
+ - [ ] MANUAL-VERIFY(2026-09-06): **the header round trip still leaves no member row.** After the first run above, `SELECT COUNT(*) FROM space_roles WHERE platform_user_id='preflight'` → **0**, and `mrctl roles list` must not show `preflight`. The ephemeral-caller rule is keyed on the caller id, not on how it was asserted, so this should hold in both modes — which is exactly why it is worth checking once on the mode that never ran on a box before.
979
+
980
+ ## image-self-describing (R0.1, 2026-09-06)
981
+
982
+ Image-side: the Dockerfile changed, so nothing here is live until the image is
983
+ rebuilt. The checks below are the only proof the stamping works — the unit
984
+ tests cover the parser, not the build.
985
+
986
+ - [x] VERIFIED(2026-09-07, dev box, Docker 29.5.3 linux/amd64): built from the tip of `dev` with `--build-arg MERCURY_VERSION=0.19.0` under a scratch tag; `com.mercury.version` and `org.opencontainers.image.version` both read `0.19.0`. Original check: **the build stamps a real version.** On a machine with Docker and this checkout, `bash container/build.sh`, then `docker image inspect mercury-agent:$(bun -e 'console.log(require("./package.json").version)') -f '{{index .Config.Labels "com.mercury.version"}}'` — expect the package.json version, never `unknown`. `unknown` means the `--build-arg` did not reach the `ARG`, and the whole R0.4/R0.5 version check would be reading a constant.
987
+ - [x] VERIFIED(2026-09-07): `/app/image-manifest.json` present, version `0.19.0`, 16 `copySources` including `src/agent/container-entry.ts` and `container/Dockerfile`, zero `--from=` paths; the copied Dockerfile's sha256 matches the repo's LF-normalised file. `bun run smoke:image` against the same image passed all four checks, including the two R0.2 added (version label; `container-entry.ts` and `mrctl.ts` both `bun build` inside the image). Original check: **the manifest is inside the image and describes it.** `docker run --rm --entrypoint cat mercury-agent:latest /app/image-manifest.json` — expect the same version, a `dockerfileSha256`, and `copySources` listing `src/agent/container-entry.ts` and `container/Dockerfile`. Nothing from a `COPY --from=` stage (`/usr/local/bin/bun`, `/usr/local/go`) may appear.
988
+ - [ ] MANUAL-VERIFY(2026-09-06): **the derived image inherits the label.** After a space with extensions has built `mercury-agent-ext-*`, `docker image inspect <derived> -f '{{index .Config.Labels "com.mercury.version"}}'` must report the same version as the base. Verified once by hand on the tagula box 2026-09-05 for the old labels; this re-checks it for the ones the build now sets, because R0.4's host-side check reads the version through the derived image, not the base.
989
+ - [ ] MANUAL-VERIFY(2026-09-06): **both tags point at one image.** `docker image inspect mercury-agent:latest -f '{{.Id}}'` and the same for `:<version>` — identical ids. If they differ, `build.sh` built twice and `:latest` is not the version it claims.
990
+
991
+ ## mercury-publishes-image (R0.3, 2026-09-06)
992
+
993
+ CI-side: `release.yml`'s new `publish-image` job does not run until the next
994
+ `v*` tag, so none of this is proven yet. The first release after this is the
995
+ test.
996
+
997
+ - [ ] MANUAL-VERIFY(2026-09-06): **the job runs and pushes.** On the next `v*` tag, watch the `publish-image` job. It runs after `publish-npm`, so if it fails npm is ahead of GHCR — safe (boxes follow the image) but the release is not finished. Expect `ghcr.io/avishai-tsabari/mercury-agent:<version>` in the repo's Packages.
998
+ - [ ] MANUAL-VERIFY(2026-09-06): **a prerelease does not move `:latest`.** If the next tag is a prerelease, confirm the job log says `:latest untouched` and that `:latest` still resolves to the previous stable id.
999
+ - [ ] MANUAL-VERIFY(2026-09-06): **the package landed private, and the fleet token can read it.** G-007 keeps visibility private. After the first push, GitHub → Packages → `mercury-agent` should show Private (it inherits the repo's visibility; nothing here changes it). Then from a box, or any shell with the fleet's `REGISTRY_PULL_TOKEN`: `docker login ghcr.io -u x-access-token --password-stdin` and `docker pull ghcr.io/avishai-tsabari/mercury-agent:<version>` — must succeed, since that token is what `box-upgrade.ts` will use once it points here.
1000
+ - [ ] MANUAL-VERIFY(2026-09-06): **retire the fleet's duplicate.** Once a Mercury-published image exists for a version, `agent-image.yml` in tagula-agents is building the same thing into a different namespace. Decide deliberately: point `box-upgrade.ts` at the Mercury image and delete that workflow, or keep the fleet image and accept two publishers. Do not leave both running unexamined — two images with the same version label and different contents is worse than either.
@@ -57,7 +57,7 @@ Custom roles can be created by assigning permissions to any role name.
57
57
  | `spaces.rename` | Rename a space and link/unlink conversations |
58
58
  | `spaces.delete` | Delete current space and all related DB data |
59
59
  | `model.list` | List the configured model chain (`mrctl model list`, chat `/model list`) |
60
- | `model.switch` | Switch the active model for the space (`mrctl model switch`, chat `/model switch`) — takes effect on the next agent run |
60
+ | `model.switch` | Pin the space to one model leg (`mrctl model switch`, chat `/model switch`) or clear that pin (`mrctl model reset`, chat `/model reset`) — takes effect on the next agent run. A pinned space has no fallback; reset is refused while the space has an active scheduled task |
61
61
  | `media.receive` | Incoming attachments are saved to `inbox/` and shown to the agent |
62
62
  | `media.send` | Outbox files produced on this caller's turn are delivered back to the chat |
63
63