mercury-agent 0.18.2 → 0.20.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (107) hide show
  1. package/container/Dockerfile +25 -0
  2. package/container/build.sh +23 -3
  3. package/docs/behavior-layers.md +25 -14
  4. package/docs/configuration.md +139 -11
  5. package/docs/container-lifecycle.md +151 -1
  6. package/docs/context-architecture.md +6 -2
  7. package/docs/extensions.md +9 -1
  8. package/docs/goals/football-reporter-profile/decisions.md +79 -0
  9. package/docs/goals/rehearsal-bench/decisions.md +41 -3
  10. package/docs/goals/rehearsal-bench/roadmap.md +3 -1
  11. package/docs/goals/release-gate/decisions.md +86 -0
  12. package/docs/goals/release-gate/roadmap.md +36 -3
  13. package/docs/live-testing.md +22 -0
  14. package/docs/pending-verification.md +178 -0
  15. package/docs/permissions.md +1 -1
  16. package/docs/profile-guide.md +28 -8
  17. package/examples/extensions/archive/backends/local.ts +6 -3
  18. package/examples/extensions/archive/queue.ts +51 -10
  19. package/examples/extensions/feed-watch/config.ts +42 -2
  20. package/examples/extensions/feed-watch/digest.ts +4 -7
  21. package/examples/extensions/feed-watch/items.ts +18 -7
  22. package/examples/extensions/feed-watch/match.ts +125 -9
  23. package/examples/extensions/feed-watch/skill/SKILL.md +25 -0
  24. package/examples/extensions/feed-watch/watch.ts +39 -5
  25. package/examples/extensions/gws/index.ts +126 -8
  26. package/examples/extensions/longview/hook.ts +72 -7
  27. package/examples/extensions/longview/index.ts +2 -0
  28. package/examples/extensions/morning/README.md +26 -15
  29. package/examples/extensions/morning/index.ts +29 -6
  30. package/examples/extensions/morning/lib/hosts.ts +16 -0
  31. package/examples/extensions/morning/lib/morning.ts +78 -0
  32. package/examples/extensions/morning/lib/upload.ts +584 -0
  33. package/examples/extensions/morning/skill/SKILL.md +34 -4
  34. package/examples/extensions/napkin/index.ts +12 -3
  35. package/examples/extensions/napkin/pi-spawn.ts +5 -1
  36. package/examples/extensions/overview/index.ts +20 -0
  37. package/examples/extensions/overview/skill/SKILL.md +9 -1
  38. package/examples/extensions/pinchtab/index.ts +36 -7
  39. package/examples/extensions/pinchtab/skill/SKILL.md +28 -1
  40. package/examples/profiles/_template/AGENTS.md +6 -1
  41. package/examples/profiles/football-reporter/AGENTS.md +16 -4
  42. package/examples/profiles/football-reporter/README.md +1 -1
  43. package/examples/profiles/football-reporter/config.yaml +42 -6
  44. package/examples/profiles/football-reporter/seed/MEMORY.md +1 -1
  45. package/examples/profiles/football-reporter/seed/episodes/beitar-jerusalem-2026-27.md +1 -1
  46. package/examples/profiles/football-reporter/seed/episodes/maccabi-tel-aviv-2026-27.md +24 -0
  47. package/examples/profiles/football-reporter/seed/napkin-distill.md +30 -29
  48. package/examples/profiles/football-reporter/standard.json +45 -8
  49. package/package.json +11 -6
  50. package/resources/skills/tasks/SKILL.md +17 -1
  51. package/resources/templates/AGENTS.md +4 -3
  52. package/resources/templates/mercury.example.yaml +13 -4
  53. package/src/adapters/whatsapp-ingress.ts +15 -3
  54. package/src/agent/container-entry.ts +299 -29
  55. package/src/agent/container-env.ts +26 -8
  56. package/src/agent/container-runner.ts +497 -76
  57. package/src/agent/image-contract.ts +77 -0
  58. package/src/agent/image-manifest.ts +180 -0
  59. package/src/agent/image-refresh.ts +318 -0
  60. package/src/cli/mercury.ts +225 -28
  61. package/src/cli/mrctl-http.ts +5 -0
  62. package/src/cli/mrctl.ts +57 -5
  63. package/src/cli/service-unit.ts +109 -0
  64. package/src/config-file.ts +22 -1
  65. package/src/config.ts +105 -14
  66. package/src/core/api.ts +11 -2
  67. package/src/core/commands.ts +11 -4
  68. package/src/core/connection-health.ts +377 -0
  69. package/src/core/direct-send.ts +235 -14
  70. package/src/core/exec.ts +10 -0
  71. package/src/core/history-window.ts +142 -0
  72. package/src/core/model-command.ts +130 -0
  73. package/src/core/operator-alerts.ts +326 -23
  74. package/src/core/permissions.ts +28 -0
  75. package/src/core/profiles.ts +32 -16
  76. package/src/core/reply-context.ts +31 -0
  77. package/src/core/routes/config-builtin.ts +11 -0
  78. package/src/core/routes/console.ts +98 -16
  79. package/src/core/routes/dashboard.ts +93 -10
  80. package/src/core/routes/model.ts +26 -3
  81. package/src/core/routes/send.ts +1 -1
  82. package/src/core/routes/tasks.ts +74 -0
  83. package/src/core/runtime.ts +302 -21
  84. package/src/core/system-messages.ts +33 -0
  85. package/src/core/task-scheduler.ts +155 -7
  86. package/src/extensions/image-builder.ts +1 -1
  87. package/src/extensions/installer.ts +72 -17
  88. package/src/extensions/load-project.ts +74 -0
  89. package/src/extensions/loader.ts +50 -9
  90. package/src/host-version.ts +32 -0
  91. package/src/main.ts +42 -21
  92. package/src/preflight/checks/credential.ts +300 -0
  93. package/src/preflight/checks/docker.ts +149 -0
  94. package/src/preflight/checks/extensions.ts +90 -0
  95. package/src/preflight/checks/host-deps.ts +247 -0
  96. package/src/preflight/checks/image-contract.ts +228 -0
  97. package/src/preflight/checks/roundtrip.ts +424 -0
  98. package/src/preflight/checks/sandbox.ts +159 -0
  99. package/src/preflight/deps.ts +107 -0
  100. package/src/preflight/probe-container.ts +156 -0
  101. package/src/preflight/report.ts +177 -0
  102. package/src/preflight/run.ts +223 -0
  103. package/src/server.ts +55 -14
  104. package/src/storage/db.ts +41 -2
  105. package/src/storage/models-json.ts +110 -0
  106. package/src/text/reporter-lint.ts +167 -23
  107. package/src/types.ts +22 -0
@@ -203,10 +203,15 @@ win by being specific:
203
203
  - *Presenting tool results* — "simple lists, never commands/JSON". A
204
204
  reporter who must not answer in lists: say so; a profile whose users *are*
205
205
  developers: say the opposite.
206
- - *Character → "set a preference"* — a standing instruction from a member
207
- becomes "only an admin can set that", which in a friends' group reads as
208
- refusing a joke. Say what the bot does with a standing request from a
209
- member (relay, decline in one line, or treat as banter).
206
+ - *Character → "set a preference"* — a standing instruction from an
207
+ admin becomes "only an admin can set that", which in a friends' group reads
208
+ as refusing a joke. A member never sees that procedure at all: from the
209
+ 2026-09 image rebuild (`7d242e2`) the `## Character` heading and its "Always
210
+ follow it" / "only the bot owner can change it" sentences ship to every
211
+ caller, while the `mrctl` procedure under them stays gated on `prefs.set`, so
212
+ a member's request gets the one-line decline. Say what the bot does with a
213
+ standing request from a member (relay, decline in one line, or treat as
214
+ banter).
210
215
  - *"Be concise"*, *"Ask for clarification"* from the pi preamble and the
211
216
  default global `AGENTS.md` — a reporter that must give a full factual
212
217
  answer, or a bot that must not ask before acting, says so.
@@ -360,10 +365,25 @@ What that means for a profile:
360
365
  bot an explicit rule for when to update it. The three DM spaces on the
361
366
  live bot that have one use it; the football space, which needed it most,
362
367
  never had one.
363
- 3. **Episodes need keywords that will recur.** An episode about "Abu Fani's
364
- transfer" with `keywords: ["אבו פאני","hapoel","transfer"]` surfaces on
365
- the next message containing one of those words; without keywords it never
366
- surfaces. Prefer a notebook the bot reads by rule over relying on scoring.
368
+ 3. **Episodes need keywords that will recur, one word each.** An episode
369
+ about "Abu Fani's transfer" with `keywords: ["פאני","fani","hapoel",
370
+ "transfer"]` surfaces on the next message containing one of those words;
371
+ without keywords it never surfaces. The message is split on whitespace,
372
+ ASCII punctuation, the hyphen and the maqaf, so a keyword with a space, an
373
+ apostrophe or a dash — `"אבו פאני"`, `"מנצ'סטר"`, `"man-united"` — can
374
+ never match anything (a geresh or gershayim *inside* a word in the message
375
+ is dropped rather than split, so `לבית"ר` does reach the keyword `ביתר`;
376
+ the keyword itself is never rewritten). A Hebrew prefix on the word still
377
+ matches (`ליונייטד`, `וכשיונייטד`) as long as four letters remain once it
378
+ is stripped — five after a leading `מ`, which also opens whole words. So
379
+ list the bare word once — every extra keyword lowers the note's score —
380
+ and before choosing a short keyword, read it back with each of
381
+ `ו/ה/ב/ל/ש/כ` in front: if that spells a different everyday word, the
382
+ note will surface on that word too, so pick another keyword. Mem is the
383
+ one the host handles itself — it turned `שוער` (goalkeeper) into `משוער`
384
+ (projected) — and the extra letter it costs is deliberate: `מחיפה` no
385
+ longer reaches a note keyed `חיפה`, while `חיפה` and `בחיפה` still do.
386
+ Prefer a notebook the bot reads by rule over relying on scoring.
367
387
  4. **Do not let distillation stand in for a notebook.** On the live bot
368
388
  napkin distils every space daily and has written *zero* entity files for
369
389
  the football space — group banter does not meet its "lasting signal"
@@ -13,7 +13,6 @@
13
13
 
14
14
  import { randomBytes } from "node:crypto";
15
15
  import {
16
- copyFileSync,
17
16
  existsSync,
18
17
  mkdirSync,
19
18
  readdirSync,
@@ -22,6 +21,7 @@ import {
22
21
  statSync,
23
22
  unlinkSync,
24
23
  } from "node:fs";
24
+ import { copyFile } from "node:fs/promises";
25
25
  import { dirname, join } from "node:path";
26
26
  import { type ArchiveBackend, ArchiveConfigError } from "./types.js";
27
27
 
@@ -84,7 +84,9 @@ export class LocalArchiveBackend implements ArchiveBackend {
84
84
  // hash that claims to describe complete content.
85
85
  const tmp = `${dest}.${randomBytes(4).toString("hex")}.tmp`;
86
86
  try {
87
- copyFileSync(srcPath, tmp);
87
+ // Awaited: blobs run to the 25 MB outbox cap and this backend is on the
88
+ // host delivery path — a sync copy stalls every other space.
89
+ await copyFile(srcPath, tmp);
88
90
  renameSync(tmp, dest);
89
91
  } catch (err) {
90
92
  // Leave nothing behind on failure; the queue will retry.
@@ -122,7 +124,8 @@ export class LocalArchiveBackend implements ArchiveBackend {
122
124
  // attached to a reply and sent to the customer as if it were whole.
123
125
  const tmp = `${destPath}.${randomBytes(4).toString("hex")}.tmp`;
124
126
  try {
125
- copyFileSync(src, tmp);
127
+ // Awaited, same reason as `put`.
128
+ await copyFile(src, tmp);
126
129
  renameSync(tmp, destPath);
127
130
  } catch (err) {
128
131
  try {
@@ -29,11 +29,21 @@
29
29
  * bytes that were never stored: `ARCHIVE.md` would advertise a file, and
30
30
  * retrieving it would 404 forever. That asymmetry is why the order is a rule
31
31
  * and not a preference.
32
+ *
33
+ * ## The one window that is not a crash
34
+ *
35
+ * `spool` used to have no `await` in it, so `deleteSpool` — whole-space
36
+ * erasure — could only run wholly before or wholly after it. The copy is
37
+ * awaited now (it was blocking the host event loop), so an erasure can land
38
+ * inside it. That is not a crash window and reconcile does not cover it: the
39
+ * pending directory simply stops existing halfway through. `spool` detects it
40
+ * and fails with a message naming the erasure, rather than recreating the
41
+ * directory — a record written back into an erased spool would drain, and the
42
+ * file the customer asked to erase would be archived again.
32
43
  */
33
44
 
34
45
  import { createHash, randomBytes } from "node:crypto";
35
46
  import {
36
- copyFileSync,
37
47
  createReadStream,
38
48
  existsSync,
39
49
  mkdirSync,
@@ -44,6 +54,7 @@ import {
44
54
  unlinkSync,
45
55
  writeFileSync,
46
56
  } from "node:fs";
57
+ import { copyFile } from "node:fs/promises";
47
58
  import { join } from "node:path";
48
59
  import type { ArchiveBackend } from "./backends/types.js";
49
60
  import { isDeleting } from "./inflight.js";
@@ -121,15 +132,45 @@ export async function spool(
121
132
  mkdirSync(dir, { recursive: true });
122
133
 
123
134
  const blobCopy = join(dir, `${record.id}.blob`);
124
- copyFileSync(srcPath, blobCopy);
125
-
126
- const full: PendingRecord = { v: PENDING_VERSION, attempts: 0, ...record };
127
- const tmp = join(
128
- dir,
129
- `${record.id}.json.${randomBytes(4).toString("hex")}.tmp`,
130
- );
131
- writeFileSync(tmp, JSON.stringify(full), "utf8");
132
- renameSync(tmp, join(dir, `${record.id}.json`));
135
+ try {
136
+ // Awaited, not `copyFileSync`. This runs inside the `delivery` hook, on
137
+ // the host, in-process — the same thread that owns every other space's
138
+ // turn, the dashboard, `/health` and the inner-container API socket.
139
+ // Outbox files are capped at 25 MB, so the synchronous version parked all
140
+ // of them for a 25 MB disk-to-disk copy on the hot path before a reply was
141
+ // sent.
142
+ await copyFile(srcPath, blobCopy);
143
+
144
+ const full: PendingRecord = { v: PENDING_VERSION, attempts: 0, ...record };
145
+ const tmp = join(
146
+ dir,
147
+ `${record.id}.json.${randomBytes(4).toString("hex")}.tmp`,
148
+ );
149
+ writeFileSync(tmp, JSON.stringify(full), "utf8");
150
+ renameSync(tmp, join(dir, `${record.id}.json`));
151
+ } catch (err) {
152
+ throw erasedMidSpool(err, dir, record);
153
+ }
154
+ }
155
+
156
+ /**
157
+ * Translate "the spool directory vanished under us" — the header's one
158
+ * non-crash window, opened by making the copy awaited — into a message that
159
+ * names what happened. Every other error is passed through untouched: nothing
160
+ * is swallowed here, and the `delivery` hook logs whichever comes out.
161
+ */
162
+ function erasedMidSpool(
163
+ err: unknown,
164
+ dir: string,
165
+ record: { space: string; name: string },
166
+ ): Error {
167
+ const code = (err as NodeJS.ErrnoException | undefined)?.code;
168
+ if (code === "ENOENT" && !existsSync(dir)) {
169
+ return new Error(
170
+ `the spool for space ${record.space} was erased while ${record.name} was being copied into it — not spooled`,
171
+ );
172
+ }
173
+ return err instanceof Error ? err : new Error(String(err));
133
174
  }
134
175
 
135
176
  /** Every spooled record for a space, oldest first. `onCorrupt` is called per unreadable file. */
@@ -1,6 +1,6 @@
1
1
  /**
2
- * The eight per-space config keys, their validators, and the reader that turns
3
- * eight strings into one typed `WatchConfig`.
2
+ * The ten per-space config keys, their validators, and the reader that turns
3
+ * ten strings into one typed `WatchConfig`.
4
4
  *
5
5
  * Two audiences, deliberately held apart:
6
6
  *
@@ -31,6 +31,19 @@ export const MAX_BATCH_MINUTES = 30;
31
31
  export const DEFAULT_DIGEST_MAX_ITEMS = 200;
32
32
  export const DIGEST_MAX_ITEMS_CEILING = 400;
33
33
 
34
+ /**
35
+ * Most items one verify run carries. The other half of the same lesson, on the
36
+ * other prompt: `prompts/verify.md` asks the agent to open the link for every
37
+ * item, which is a per-item instruction, so the batch has to stay the size that
38
+ * instruction was written for. Nothing bounded it until 2026-09-06, when the
39
+ * first poll after a 29-hour outage opened a single run with **116** items and
40
+ * the model answered with one line — 115 items neither verified nor deferred.
41
+ * Overflow is not dropped: it stays in the pending batch for the next tick and
42
+ * is in the daily digest either way.
43
+ */
44
+ export const DEFAULT_VERIFY_MAX_ITEMS = 20;
45
+ export const VERIFY_MAX_ITEMS_CEILING = 50;
46
+
34
47
  /** An RSS/Atom URL, polled as-is. */
35
48
  export interface RssSource {
36
49
  type: "rss";
@@ -70,6 +83,8 @@ export interface WatchConfig {
70
83
  quietHours: QuietHours | null;
71
84
  maxPerHour: number;
72
85
  batchMinutes: number;
86
+ /** Hard ceiling on how many items one verify run carries, newest first. */
87
+ verifyMaxItems: number;
73
88
  /** Hard ceiling on how many items the daily digest hands the article. */
74
89
  digestMaxItems: number;
75
90
  /** IANA zone, or "" to fall back to the deployment default and then UTC. */
@@ -178,6 +193,15 @@ export function isValidDigestMaxItems(value: string): boolean {
178
193
  return isIntInRange(value, 1, DIGEST_MAX_ITEMS_CEILING);
179
194
  }
180
195
 
196
+ /**
197
+ * At least one, because zero is already spelled `max_per_hour: 0` — a run that
198
+ * carries no items is not a quieter watcher, it is an empty prompt. The ceiling
199
+ * is where a per-item "open the link" instruction stops being followable.
200
+ */
201
+ export function isValidVerifyMaxItems(value: string): boolean {
202
+ return isIntInRange(value, 1, VERIFY_MAX_ITEMS_CEILING);
203
+ }
204
+
181
205
  /**
182
206
  * An IANA zone name, checked by asking `Intl` to use it. Empty means "unset",
183
207
  * which is a valid state — the poller falls back to the deployment default.
@@ -290,6 +314,16 @@ export function readSpaceConfig(
290
314
  log,
291
315
  ),
292
316
  ),
317
+ verifyMaxItems: Number(
318
+ read(
319
+ getConfig,
320
+ spaceId,
321
+ "verify_max_items",
322
+ String(DEFAULT_VERIFY_MAX_ITEMS),
323
+ isValidVerifyMaxItems,
324
+ log,
325
+ ),
326
+ ),
293
327
  digestMaxItems: Number(
294
328
  read(
295
329
  getConfig,
@@ -374,6 +408,12 @@ export const CONFIG_KEYS: ConfigKeyDef[] = [
374
408
  default: String(DEFAULT_BATCH_MINUTES),
375
409
  validate: isValidBatchMinutes,
376
410
  },
411
+ {
412
+ key: "verify_max_items",
413
+ description: `Most items one verify run carries, newest first (1–${VERIFY_MAX_ITEMS_CEILING}, default ${DEFAULT_VERIFY_MAX_ITEMS}). The rest stay in the batch for the next tick and reach the daily digest either way — a bound on how much one run is asked to check, not a filter.`,
414
+ default: String(DEFAULT_VERIFY_MAX_ITEMS),
415
+ validate: isValidVerifyMaxItems,
416
+ },
377
417
  {
378
418
  key: "digest_max_items",
379
419
  description: `Most items the daily digest carries into the article's prompt, newest first (1–${DIGEST_MAX_ITEMS_CEILING}, default ${DEFAULT_DIGEST_MAX_ITEMS}). Older matches stay in the workspace digest file. A bound on prompt size, not a filter.`,
@@ -18,6 +18,7 @@ import {
18
18
  formatItemLines,
19
19
  formatLocal,
20
20
  isAggregatorLink,
21
+ newestFirst,
21
22
  resolveTimezone,
22
23
  type StoredItem,
23
24
  } from "./items.js";
@@ -200,13 +201,9 @@ export function buildDigest(
200
201
  return !Number.isFinite(untilMs) || seen <= untilMs;
201
202
  });
202
203
 
203
- // Newest first, then cut. `formatItemLines` sorts the same way, so the lines
204
- // and the cut agree about which items are the freshest.
205
- const byRecency = [...items].sort(
206
- (a, b) =>
207
- Date.parse(b.publishedAt ?? b.seenAt) -
208
- Date.parse(a.publishedAt ?? a.seenAt),
209
- );
204
+ // Newest first, then cut. `formatItemLines` sorts through the same helper, so
205
+ // the lines and the cut agree about which items are the freshest.
206
+ const byRecency = newestFirst(items);
210
207
  const cap = maxItems && maxItems > 0 ? maxItems : byRecency.length;
211
208
  const shown = byRecency.slice(0, cap);
212
209
  const omitted = byRecency.length - shown.length;
@@ -191,25 +191,36 @@ export function formatItemLine(
191
191
  return `- ${parts.join(" | ")}`;
192
192
  }
193
193
 
194
+ /**
195
+ * A copy of `items` ordered **newest first**, leaving the input untouched.
196
+ *
197
+ * Sorted by publication time where the outlet gave one, and by when we saw it
198
+ * where it did not — an undated item's `seenAt` is the closest honest stand-in,
199
+ * and it keeps the order total rather than leaving undated items bunched.
200
+ *
201
+ * Three callers share it and they must agree: the rendered lines, the digest's
202
+ * `digest_max_items` cut and the verify run's `verify_max_items` cut. A cap
203
+ * that cuts on one order while the lines are rendered in another would drop
204
+ * items nobody could name — so the order lives here once.
205
+ */
206
+ export function newestFirst(items: StoredItem[]): StoredItem[] {
207
+ const at = (i: StoredItem) => Date.parse(i.publishedAt ?? i.seenAt);
208
+ return [...items].sort((a, b) => at(b) - at(a));
209
+ }
210
+
194
211
  /**
195
212
  * Render a list of items as lines, **newest first**.
196
213
  *
197
214
  * The buffer is append-ordered, so without this both the verify prompt and the
198
215
  * digest would lead with the oldest headline. The agent is told to trust these
199
216
  * lines over its own memory; the freshest one is the one it should read first.
200
- *
201
- * Sorted by publication time where the outlet gave one, and by when we saw it
202
- * where it did not — an undated item's `seenAt` is the closest honest stand-in,
203
- * and it keeps the order total rather than leaving undated items bunched.
204
217
  */
205
218
  export function formatItemLines(
206
219
  items: StoredItem[],
207
220
  zone: string,
208
221
  opts: FormatOptions = {},
209
222
  ): string {
210
- const at = (i: StoredItem) => Date.parse(i.publishedAt ?? i.seenAt);
211
- return [...items]
212
- .sort((a, b) => at(b) - at(a))
223
+ return newestFirst(items)
213
224
  .map((i) => formatItemLine(i, zone, opts))
214
225
  .join("\n");
215
226
  }
@@ -7,6 +7,11 @@
7
7
  * phone will carry none of those forms. So matching normalizes aggressively
8
8
  * and then asks a substring question, which is the only question that survives
9
9
  * a prefixed, pointed, hyphenated language.
10
+ *
11
+ * Exclusions ask a narrower one — the same substring, but it has to start a
12
+ * word, looking through the attached Hebrew prefixes. The asymmetry is the
13
+ * point: an over-matching watch term costs a line in a prompt that then
14
+ * rejects it, an over-matching exclusion drops the story and says nothing.
10
15
  */
11
16
 
12
17
  import { createHash } from "node:crypto";
@@ -59,20 +64,129 @@ export interface MatchResult {
59
64
  terms: string[];
60
65
  /** True when an exclude term hit — the item is dropped whatever matched. */
61
66
  excluded: boolean;
67
+ /**
68
+ * The exclude term that dropped it, as the admin wrote it. Present only when
69
+ * `excluded` is true — an item dropped here leaves no other trace, so the
70
+ * needle is carried out for the caller to log.
71
+ */
72
+ excludedBy?: string;
73
+ }
74
+
75
+ /**
76
+ * The Hebrew prefix letters that attach to the word they govern: ו (and),
77
+ * ה (the), ב (in), ל (to), מ (from), ש (that), כ (as). Up to three stack —
78
+ * `וכשנשים` is "and when women".
79
+ *
80
+ * The same set, and the same stacking limit, as `HEBREW_PREFIX_RE` in
81
+ * `src/agent/container-entry.ts`, which does this for episode keywords. It is
82
+ * restated rather than imported because this extension is installed as a
83
+ * snapshot and imports nothing from the host; when one moves, move both.
84
+ */
85
+ const HEBREW_PREFIXES = "ובהלמשכ";
86
+ const MAX_STACKED_PREFIXES = 3;
87
+
88
+ /** A letter or digit in any script. What "inside a word" means here. */
89
+ const WORD_CHAR = /[\p{L}\p{N}]/u;
90
+
91
+ /**
92
+ * A *Hebrew* needle shorter than this is matched at a bare word boundary only
93
+ * — the prefix letters are not stripped for it. Three-letter Hebrew words are
94
+ * where the false positives live (`container-entry.ts` carries the same floor
95
+ * for the same reason: `בבית` must not select a note keyed on `בית`), and an
96
+ * exclusion that fires wrongly drops a story with no trace at all.
97
+ *
98
+ * The floor is about Hebrew stems, so it is gated on the needle's script and
99
+ * not on its length alone. A Latin needle is never a short Hebrew word, and it
100
+ * still needs the walk: Hebrew outlets write `ה-WSL`, which `normalize` folds
101
+ * to `הwsl` — three-letter Latin acronyms (`WSL`, `U21`, `U19`) are exactly
102
+ * the entries on the live list that arrive glued to a prefix.
103
+ */
104
+ const PREFIX_STRIP_MIN_CHARS = 4;
105
+
106
+ /**
107
+ * Does the term open with a character from the Hebrew block (U+0590–U+05FF)?
108
+ *
109
+ * Written as code points rather than a character-class range: a pasted range
110
+ * of two invisible-ish boundary characters is one format-on-save away from
111
+ * meaning something else.
112
+ */
113
+ function opensWithHebrew(term: string): boolean {
114
+ const cp = term.codePointAt(0) ?? 0;
115
+ return cp >= 0x0590 && cp <= 0x05ff;
116
+ }
117
+
118
+ /**
119
+ * Does the occurrence at `index` start a word?
120
+ *
121
+ * "Start of a word" in Hebrew has to look through the attached prefixes: the
122
+ * feeds publish `הנשים`, `לנשים` and `בנשים` for the same word an admin typed
123
+ * as `נשים`, and nothing separates the letter from the noun. So up to three
124
+ * prefix letters are walked back over before the boundary is tested — which
125
+ * is what keeps `נשים` out of `אנשים`, where the preceding letter (א) is not
126
+ * one of them.
127
+ */
128
+ function startsWord(
129
+ text: string,
130
+ index: number,
131
+ allowPrefixes: boolean,
132
+ ): boolean {
133
+ let start = index;
134
+ if (allowPrefixes) {
135
+ let stripped = 0;
136
+ while (start > 0 && stripped < MAX_STACKED_PREFIXES) {
137
+ // Spelled out rather than `includes(text[i] ?? "")`: `"…".includes("")`
138
+ // is *true*, so a missing character would authorize the strip instead of
139
+ // refusing it. Absence denies.
140
+ const before = text[start - 1];
141
+ if (!before || !HEBREW_PREFIXES.includes(before)) break;
142
+ start--;
143
+ stripped++;
144
+ }
145
+ }
146
+ return start === 0 || !WORD_CHAR.test(text[start - 1] ?? "");
147
+ }
148
+
149
+ /**
150
+ * Does `needle` occur in `text` at the start of a word (prefixes allowed)?
151
+ *
152
+ * Every occurrence is tried, not just the first: `אנשים במכבי חיפה נשים` has
153
+ * one that fails the test and one that passes, and the item is excluded.
154
+ *
155
+ * The *end* of the needle is deliberately not tested. `Women` has to keep
156
+ * dropping `Women's` — which normalizes to `womens` — and Hebrew inflects at
157
+ * the end as freely as it prefixes at the front. The bug this answers is
158
+ * over-matching to the left; the right-hand side is what the substring design
159
+ * was right about.
160
+ */
161
+ function matchesAtWordStart(text: string, needle: string): boolean {
162
+ const allowPrefixes =
163
+ !opensWithHebrew(needle) || needle.length >= PREFIX_STRIP_MIN_CHARS;
164
+ for (
165
+ let at = text.indexOf(needle);
166
+ at !== -1;
167
+ at = text.indexOf(needle, at + 1)
168
+ ) {
169
+ if (startsWord(text, at, allowPrefixes)) return true;
170
+ }
171
+ return false;
62
172
  }
63
173
 
64
174
  /**
65
175
  * Match an item against a watchlist.
66
176
  *
67
- * Substring, not word-boundary. A boundary test is wrong for Hebrew — the
68
- * prefixes ל/ב/ה/ו/ש/כ/מ attach directly to the noun, so `למכבי` would fail a
69
- * boundary test against `מכבי` and the story would be missed. The cost is that
70
- * a short Latin term over-matches (`City` inside `Capacity`); that costs one
71
- * extra line in a prompt the verify run then rejects, where the Hebrew miss
72
- * costs the story itself. The skill tells admins to keep Latin terms specific.
177
+ * **Watch terms are plain substrings.** A naive boundary test is wrong for
178
+ * Hebrew — the prefixes ל/ב/ה/ו/ש/כ/מ attach directly to the noun, so `למכבי`
179
+ * would fail one against `מכבי` and the story would be missed. The cost is
180
+ * that a short Latin term over-matches (`City` inside `Capacity`); that costs
181
+ * one extra line in a prompt the verify run then rejects, where the Hebrew
182
+ * miss costs the story itself. The skill tells admins to keep terms specific.
73
183
  *
74
- * Exclusion is checked first and is unconditional: an excluded item is dropped
75
- * even when three watch terms hit it.
184
+ * **Exclusions match at a word start**, prefixes looked through. Exclusion is
185
+ * checked first and is unconditional: an excluded item is dropped even when
186
+ * three watch terms hit it, and it is dropped before any run or digest can
187
+ * show it — so the two directions are not symmetric, and the exclusion is the
188
+ * one that must not fire on an accident. `נשים` ("women") inside `אנשים`
189
+ * ("people") dropped ordinary football headlines until 2026-09-06.
76
190
  */
77
191
  export function matchItem(
78
192
  item: Pick<FeedItem, "title" | "description">,
@@ -83,7 +197,9 @@ export function matchItem(
83
197
 
84
198
  for (const raw of exclude) {
85
199
  const needle = normalize(raw);
86
- if (needle && text.includes(needle)) return { terms: [], excluded: true };
200
+ if (needle && matchesAtWordStart(text, needle)) {
201
+ return { terms: [], excluded: true, excludedBy: raw };
202
+ }
87
203
  }
88
204
 
89
205
  const hits: string[] = [];
@@ -60,6 +60,8 @@ Settings are per-space config keys under `feed-watch.`:
60
60
  | `quiet_hours` | `HH:MM-HH:MM` local, or empty. No runs are opened inside it. |
61
61
  | `max_per_hour` | `0`–`12`. `0` means digest-only. |
62
62
  | `batch_minutes` | `0`–`30`. How long to wait for other outlets to catch up. |
63
+ | `verify_max_items` | `1`–`50`, default `20`. Most items one verify run carries. |
64
+ | `digest_max_items` | `1`–`400`, default `200`. Most items the daily digest carries. |
63
65
  | `timezone` | IANA zone for quiet hours and timestamps. |
64
66
 
65
67
  Read one with:
@@ -103,6 +105,21 @@ also hits *Capacity*. Prefer specific terms (`Man City`, `Real Madrid`) and use
103
105
  `exclude` for the categories the space does not care about (`נוער`, `נשים`,
104
106
  `fantasy`, `WSL`).
105
107
 
108
+ `exclude` is matched **at the start of a word**, not anywhere in it. The Hebrew
109
+ prefixes are looked through, so `נשים` still drops `הנשים`, `לנשים` and
110
+ `בנשים` — but not `אנשים` ("people"), which merely contains the letters. The
111
+ end of the word is still open, so `Women` drops `Women's` and `הימור` drops
112
+ `הימורים`. An exclusion shorter than four characters is matched at a bare word
113
+ boundary with no prefixes looked through: at three letters the prefixed forms
114
+ are ordinary words.
115
+
116
+ An exclusion still only drops headlines in the language it was written in:
117
+ `נשים` does not touch `Women's Champions League`. When a category is genuinely
118
+ unwanted, list every form the feeds publish it in — and remember the list is
119
+ only ever as good as its substrings. A category that must never be reported
120
+ also needs a line in the space's `AGENTS.md`, which is what catches the item
121
+ the substrings miss.
122
+
106
123
  ### Who may change it
107
124
 
108
125
  These keys are behind the `config.set` permission, which is **admin-only** by
@@ -126,3 +143,11 @@ default.
126
143
  The next new item is the first one that can wake anything.
127
144
  - `max_per_hour` may be spent, or the clock may be inside `quiet_hours`. Neither
128
145
  loses anything: those items still appear in the next daily digest.
146
+
147
+ ### If a run carries fewer items than you expected
148
+
149
+ A verify run stops at `verify_max_items` (20 by default), newest first. The
150
+ overflow is not lost — it waits for the next run and it is in the daily digest
151
+ regardless — but it means a backlog drains over several runs rather than
152
+ arriving in one. The poller says so in its log: `items=116 verified=20
153
+ deferred=96`. This is what a first poll after a long outage looks like.
@@ -19,6 +19,7 @@ import {
19
19
  formatItemLines,
20
20
  inWindow,
21
21
  localHhMm,
22
+ newestFirst,
22
23
  resolveTimezone,
23
24
  type StoredItem,
24
25
  toStoredItem,
@@ -389,7 +390,19 @@ export async function pollSpace(
389
390
  if (olderThanBuffer(item, now)) continue;
390
391
 
391
392
  const match = matchItem(item, cfg.terms, cfg.exclude);
392
- if (match.excluded) continue;
393
+ if (match.excluded) {
394
+ // The one drop that leaves no other trace: an excluded item never
395
+ // reaches a run, a digest or the buffer, so without this line an
396
+ // exclusion that fires wrongly is invisible from chat and from the
397
+ // journal alike. Debug, and only on the reporting path — the seed tick
398
+ // would write one of these per item of a whole feed.
399
+ log.debug("feed-watch: item dropped by an exclude term", {
400
+ spaceId,
401
+ needle: match.excludedBy,
402
+ title: item.title,
403
+ });
404
+ continue;
405
+ }
393
406
  const terms = mergeTerms(match.terms, forcedTerm);
394
407
  if (terms.length === 0) continue;
395
408
 
@@ -456,16 +469,26 @@ export async function pollSpace(
456
469
  deferred = "task_pending";
457
470
  } else {
458
471
  const byId = new Map(buffered.map((b) => [b.id, b]));
459
- const items = pending.itemIds
472
+ const matured = pending.itemIds
460
473
  .map((id) => byId.get(id))
461
474
  .filter((i): i is StoredItem => i !== undefined);
462
- if (items.length === 0) {
475
+ if (matured.length === 0) {
463
476
  // Every item in the batch aged out of the buffer before it matured.
464
477
  // There is nothing to verify; drop the batch rather than open a run
465
478
  // with an empty item list.
466
479
  log.debug("feed-watch: batch expired before it matured", { spaceId });
467
480
  pending = null;
468
481
  } else {
482
+ // Newest first, then cut. The template asks the agent to open the link
483
+ // for every line, so a batch that outgrew `verify_max_items` is not a
484
+ // bigger run, it is an unfollowable one — 116 items in one run on
485
+ // 2026-09-06 produced a single line and 115 unexamined items. The
486
+ // overflow is kept, not dropped: it stays in `pending` (already mature,
487
+ // so the next tick opens it under `max_per_hour`) and it is in the
488
+ // buffer the daily digest reads either way.
489
+ const ordered = newestFirst(matured);
490
+ const items = ordered.slice(0, cfg.verifyMaxItems);
491
+ const overflow = ordered.slice(cfg.verifyMaxItems);
469
492
  const runAs = resolveRunAs(
470
493
  deps.configUpdatedBy(spaceId, "feed-watch.enabled"),
471
494
  log,
@@ -475,11 +498,22 @@ export async function pollSpace(
475
498
  const taskId = deps.createTask(spaceId, prompt, runAs);
476
499
  recentRuns.push(now);
477
500
  created = { taskId, itemCount: items.length };
478
- pending = null;
501
+ // Same `startedAt`: this batch has already served its `batch_minutes`,
502
+ // and restarting the window would hold the overflow another round for
503
+ // outlets that have long since filed.
504
+ pending =
505
+ overflow.length > 0
506
+ ? {
507
+ startedAt: pending.startedAt,
508
+ itemIds: overflow.map((i) => i.id),
509
+ }
510
+ : null;
479
511
  log.info("feed-watch: opened a verify run", {
480
512
  spaceId,
481
513
  taskId,
482
- items: items.length,
514
+ items: matured.length,
515
+ verified: items.length,
516
+ deferred: overflow.length,
483
517
  terms: [...new Set(items.flatMap((i) => i.terms))],
484
518
  });
485
519
  }